@vellumai/assistant 0.8.6 → 0.8.7-dev.202606052118.34cd356

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (1078) hide show
  1. package/AGENTS.md +4 -4
  2. package/Dockerfile +21 -4
  3. package/bun.lock +13 -4
  4. package/docker-entrypoint.sh +12 -8
  5. package/docker-init-apt-root.sh +3 -1
  6. package/docker-kata-apt-env.sh +3 -1
  7. package/docker-kata-runtime-family.sh +12 -0
  8. package/docs/architecture/memory.md +1 -1
  9. package/docs/plugins.md +110 -83
  10. package/examples/plugins/echo/README.md +13 -12
  11. package/examples/plugins/echo/register.ts +0 -54
  12. package/knip.json +1 -0
  13. package/node_modules/@vellumai/environments/bun.lock +24 -0
  14. package/node_modules/@vellumai/environments/package.json +18 -0
  15. package/node_modules/@vellumai/environments/src/__tests__/package-boundary.test.ts +95 -0
  16. package/node_modules/@vellumai/environments/src/index.ts +11 -0
  17. package/node_modules/@vellumai/environments/src/seeds.ts +73 -0
  18. package/node_modules/@vellumai/environments/src/types.ts +70 -0
  19. package/node_modules/@vellumai/environments/tsconfig.json +20 -0
  20. package/node_modules/@vellumai/skill-host-contracts/src/assistant-event.ts +11 -0
  21. package/node_modules/@vellumai/skill-host-contracts/src/client.ts +3 -4
  22. package/node_modules/@vellumai/skill-host-contracts/src/server-message.ts +3 -3
  23. package/node_modules/@vellumai/skill-host-contracts/src/skill-host.ts +13 -8
  24. package/openapi.yaml +6964 -539
  25. package/package.json +8 -4
  26. package/scripts/generate-openapi.ts +88 -54
  27. package/src/__tests__/agent-loop-callsite-precedence.test.ts +42 -80
  28. package/src/__tests__/agent-loop-exit-reason.test.ts +188 -45
  29. package/src/__tests__/agent-loop-mutable-latest-user-message.test.ts +141 -0
  30. package/src/__tests__/agent-loop-override-profile.test.ts +19 -32
  31. package/src/__tests__/agent-loop-provider-error-recording.test.ts +7 -5
  32. package/src/__tests__/agent-loop-thinking.test.ts +17 -12
  33. package/src/__tests__/agent-loop.test.ts +238 -422
  34. package/src/__tests__/agent-wake-disk-pressure-callsite.test.ts +6 -2
  35. package/src/__tests__/agent-wake-override-profile.test.ts +22 -40
  36. package/src/__tests__/annotate-activity-metadata.test.ts +262 -0
  37. package/src/__tests__/annotate-risk-options.test.ts +2 -3
  38. package/src/__tests__/anthropic-provider.test.ts +296 -57
  39. package/src/__tests__/app-builder-skill-instructions.test.ts +22 -0
  40. package/src/__tests__/app-control-flow.test.ts +6 -1
  41. package/src/__tests__/app-dir-path-guard.test.ts +1 -0
  42. package/src/__tests__/approval-cascade.test.ts +4 -11
  43. package/src/__tests__/approval-routes-http.test.ts +8 -3
  44. package/src/__tests__/assistant-event-hub.test.ts +25 -0
  45. package/src/__tests__/assistant-event.test.ts +15 -0
  46. package/src/__tests__/assistant-events-sse-shed.test.ts +8 -0
  47. package/src/__tests__/assistant-feature-flags-integration.test.ts +2 -2
  48. package/src/__tests__/assistant-stream-state.test.ts +645 -0
  49. package/src/__tests__/auth-fallback-events-store.test.ts +116 -0
  50. package/src/__tests__/avatar-e2e.test.ts +7 -37
  51. package/src/__tests__/avatar-generator.test.ts +12 -42
  52. package/src/__tests__/avatar-identity-sync.test.ts +28 -3
  53. package/src/__tests__/background-shell-bash.test.ts +3 -7
  54. package/src/__tests__/background-workers-disk-pressure.test.ts +6 -0
  55. package/src/__tests__/btw-routes.test.ts +69 -15
  56. package/src/__tests__/build-persisted-content.test.ts +184 -0
  57. package/src/__tests__/call-pointer-messages.test.ts +5 -3
  58. package/src/__tests__/call-site-routing-provider.test.ts +22 -40
  59. package/src/__tests__/catalog-files.test.ts +1 -0
  60. package/src/__tests__/channel-approval-routes.test.ts +49 -21
  61. package/src/__tests__/channel-approvals.test.ts +4 -2
  62. package/src/__tests__/channel-invite-transport.test.ts +1 -5
  63. package/src/__tests__/channel-readiness-routes.test.ts +0 -4
  64. package/src/__tests__/channel-readiness-slack-remote.test.ts +2 -7
  65. package/src/__tests__/channel-retry-sweep.test.ts +71 -79
  66. package/src/__tests__/clawhub-files.test.ts +1 -0
  67. package/src/__tests__/compaction-circuit.test.ts +258 -0
  68. package/src/__tests__/compaction-direct.test.ts +132 -0
  69. package/src/__tests__/compaction-events.test.ts +5 -17
  70. package/src/__tests__/compaction-trail-store.test.ts +1 -79
  71. package/src/__tests__/compaction.benchmark.test.ts +0 -30
  72. package/src/__tests__/compactor-image-manifest-trust.test.ts +112 -0
  73. package/src/__tests__/computer-use-tools.test.ts +2 -2
  74. package/src/__tests__/config-watcher.test.ts +28 -0
  75. package/src/__tests__/context-search-agent-runner.test.ts +6 -3
  76. package/src/__tests__/context-token-estimator.test.ts +34 -0
  77. package/src/__tests__/context-window-manager-compact-retry.test.ts +291 -0
  78. package/src/__tests__/conversation-abort-tool-results.test.ts +70 -25
  79. package/src/__tests__/conversation-agent-loop-disk-pressure.test.ts +9 -7
  80. package/src/__tests__/conversation-agent-loop-inference-profile.test.ts +22 -34
  81. package/src/__tests__/conversation-agent-loop-overflow.test.ts +476 -963
  82. package/src/__tests__/conversation-agent-loop.test.ts +823 -1321
  83. package/src/__tests__/conversation-analysis-routes.test.ts +7 -3
  84. package/src/__tests__/conversation-app-control-lifecycle.test.ts +1 -1
  85. package/src/__tests__/conversation-clean-command.test.ts +5 -2
  86. package/src/__tests__/conversation-clear-safety.test.ts +20 -10
  87. package/src/__tests__/conversation-confirmation-signals.test.ts +15 -45
  88. package/src/__tests__/conversation-disk-view-integration.test.ts +2 -2
  89. package/src/__tests__/conversation-disk-view.test.ts +10 -17
  90. package/src/__tests__/conversation-fork-crud.test.ts +86 -172
  91. package/src/__tests__/conversation-fork-route.test.ts +16 -14
  92. package/src/__tests__/conversation-history-web-search.test.ts +11 -1
  93. package/src/__tests__/conversation-init.benchmark.test.ts +6 -6
  94. package/src/__tests__/conversation-lifecycle.test.ts +3 -2
  95. package/src/__tests__/conversation-load-history-repair.test.ts +3 -2
  96. package/src/__tests__/conversation-load-history-stripped.test.ts +1 -1
  97. package/src/__tests__/conversation-message-sync-tags.test.ts +3 -4
  98. package/src/__tests__/conversation-pairing.test.ts +10 -7
  99. package/src/__tests__/conversation-pre-run-repair.test.ts +1 -1
  100. package/src/__tests__/conversation-process-app-control-preactivation.test.ts +10 -0
  101. package/src/__tests__/conversation-process-callsite.test.ts +27 -30
  102. package/src/__tests__/conversation-provider-retry-repair.test.ts +80 -51
  103. package/src/__tests__/conversation-queue.test.ts +272 -164
  104. package/src/__tests__/conversation-routes-disk-view.test.ts +6 -2
  105. package/src/__tests__/conversation-routes-guardian-reply.test.ts +2 -2
  106. package/src/__tests__/conversation-routes-slash-commands.test.ts +8 -7
  107. package/src/__tests__/conversation-runtime-assembly.test.ts +317 -313
  108. package/src/__tests__/conversation-runtime-workspace.test.ts +114 -36
  109. package/src/__tests__/conversation-slash-commands.test.ts +8 -42
  110. package/src/__tests__/conversation-slash-queue.test.ts +42 -31
  111. package/src/__tests__/conversation-slash-unknown.test.ts +13 -15
  112. package/src/__tests__/conversation-speed-override.test.ts +8 -22
  113. package/src/__tests__/conversation-starter-routes.test.ts +14 -6
  114. package/src/__tests__/conversation-surfaces-action-delivery.test.ts +90 -15
  115. package/src/__tests__/conversation-surfaces-app-control.test.ts +32 -4
  116. package/src/__tests__/conversation-surfaces-state-update.test.ts +5 -2
  117. package/src/__tests__/conversation-surfaces-table-action.test.ts +6 -15
  118. package/src/__tests__/conversation-sync-tags.test.ts +27 -15
  119. package/src/__tests__/conversation-title-service.test.ts +135 -2
  120. package/src/__tests__/conversation-tool-setup-app-refresh.test.ts +23 -11
  121. package/src/__tests__/conversation-unread-route.test.ts +14 -2
  122. package/src/__tests__/conversation-usage.test.ts +0 -2
  123. package/src/__tests__/conversation-wipe.test.ts +1 -1
  124. package/src/__tests__/conversation-workspace-cache-state.test.ts +20 -17
  125. package/src/__tests__/conversation-workspace-injection.test.ts +114 -23
  126. package/src/__tests__/conversation-workspace-tool-tracking.test.ts +34 -13
  127. package/src/__tests__/conversations-import-system-filter.test.ts +101 -0
  128. package/src/__tests__/credential-execution-tools.test.ts +1 -2
  129. package/src/__tests__/credential-security-invariants.test.ts +0 -1
  130. package/src/__tests__/cross-provider-web-search.test.ts +220 -3
  131. package/src/__tests__/cu-unified-flow.test.ts +26 -1
  132. package/src/__tests__/db-acp-history.test.ts +101 -0
  133. package/src/__tests__/db-schedule-syntax-migration.test.ts +16 -0
  134. package/src/__tests__/disk-pressure-guard.test.ts +66 -0
  135. package/src/__tests__/disk-pressure-routes.test.ts +9 -2
  136. package/src/__tests__/dm-persistence.test.ts +12 -3
  137. package/src/__tests__/dynamic-page-surface.test.ts +99 -0
  138. package/src/__tests__/edit-propagation.test.ts +1 -2
  139. package/src/__tests__/empty-response-hook.test.ts +304 -0
  140. package/src/__tests__/feature-flag-test-helpers.ts +2 -2
  141. package/src/__tests__/file-write-tool.test.ts +63 -0
  142. package/src/__tests__/filing-service.test.ts +2 -2
  143. package/src/__tests__/first-greeting.test.ts +55 -14
  144. package/src/__tests__/gemini-image-service.test.ts +13 -0
  145. package/src/__tests__/gemini-inline-media.test.ts +78 -0
  146. package/src/__tests__/gemini-provider.test.ts +351 -28
  147. package/src/__tests__/guardian-grant-minting.test.ts +1 -1
  148. package/src/__tests__/guardian-routing-invariants.test.ts +2 -4
  149. package/src/__tests__/guardian-routing-state.test.ts +60 -71
  150. package/src/__tests__/handlers-user-message-approval-consumption.test.ts +10 -8
  151. package/src/__tests__/heartbeat-disk-pressure.test.ts +2 -0
  152. package/src/__tests__/heartbeat-service.test.ts +3 -1
  153. package/src/__tests__/helpers/mock-provider.ts +110 -0
  154. package/src/__tests__/helpers/native-web-search-harness.ts +129 -0
  155. package/src/__tests__/history-repair-hook.test.ts +162 -0
  156. package/src/__tests__/history-repair-observability.test.ts +1 -1
  157. package/src/__tests__/history-repair.test.ts +2 -1
  158. package/src/__tests__/host-app-control-proxy.test.ts +2 -0
  159. package/src/__tests__/host-app-control-routes.test.ts +1 -1
  160. package/src/__tests__/host-cu-proxy.test.ts +2 -0
  161. package/src/__tests__/host-cu-routes-targeted.test.ts +3 -3
  162. package/src/__tests__/host-file-edit-tool.test.ts +4 -2
  163. package/src/__tests__/host-file-proxy.test.ts +31 -0
  164. package/src/__tests__/host-file-read-tool.test.ts +4 -2
  165. package/src/__tests__/host-file-write-tool.test.ts +9 -3
  166. package/src/__tests__/host-proxy-preactivation.test.ts +53 -14
  167. package/src/__tests__/host-shell-tool.test.ts +9 -4
  168. package/src/__tests__/http-user-message-parity.test.ts +2 -2
  169. package/src/__tests__/identity-intro-cache.test.ts +47 -114
  170. package/src/__tests__/identity-routes.test.ts +248 -7
  171. package/src/__tests__/inbound-slack-persistence.test.ts +12 -3
  172. package/src/__tests__/injector-background-turn.test.ts +3 -9
  173. package/src/__tests__/injector-chain.test.ts +139 -275
  174. package/src/__tests__/injector-disk-pressure.test.ts +75 -41
  175. package/src/__tests__/injector-document-comments.test.ts +3 -3
  176. package/src/__tests__/injector-pkb-v2-silenced.test.ts +30 -22
  177. package/src/__tests__/injector-v3-suppression.test.ts +214 -0
  178. package/src/__tests__/internal-telemetry-routes.test.ts +109 -0
  179. package/src/__tests__/list-messages-attachments.test.ts +7 -8
  180. package/src/__tests__/list-messages-hidden-metadata.test.ts +55 -15
  181. package/src/__tests__/list-messages-page-latest.test.ts +60 -1
  182. package/src/__tests__/list-messages-tool-merge.test.ts +56 -6
  183. package/src/__tests__/llm-request-log-turn-query.test.ts +42 -86
  184. package/src/__tests__/llm-resolver.test.ts +23 -47
  185. package/src/__tests__/llm-usage-store.test.ts +268 -1
  186. package/src/__tests__/log-export-routes.test.ts +59 -0
  187. package/src/__tests__/managed-skill-lifecycle.test.ts +1 -8
  188. package/src/__tests__/mcp-auth-routes.test.ts +15 -10
  189. package/src/__tests__/mcp-health-check.test.ts +18 -13
  190. package/src/__tests__/memory-retrieval-hook.test.ts +297 -0
  191. package/src/__tests__/memory-v2-static-injector.test.ts +103 -35
  192. package/src/__tests__/messaging-send-tool.test.ts +8 -4
  193. package/src/__tests__/migration-export-http.test.ts +12 -12
  194. package/src/__tests__/migration-import-commit-http.test.ts +8 -8
  195. package/src/__tests__/migration-import-preflight-http.test.ts +7 -7
  196. package/src/__tests__/migration-validate-http.test.ts +3 -3
  197. package/src/__tests__/native-web-search.test.ts +205 -20
  198. package/src/__tests__/notification-decision-identity.test.ts +9 -18
  199. package/src/__tests__/notification-decision-recipient-context.test.ts +3 -6
  200. package/src/__tests__/oauth-commands-routes.test.ts +1 -1
  201. package/src/__tests__/onboarding-template-contract.test.ts +12 -0
  202. package/src/__tests__/openai-image-service.test.ts +17 -0
  203. package/src/__tests__/openai-provider.test.ts +97 -71
  204. package/src/__tests__/openai-responses-provider.test.ts +21 -77
  205. package/src/__tests__/outbound-slack-persistence.test.ts +2 -1
  206. package/src/__tests__/{overflow-reduce-pipeline.test.ts → overflow-reduction-loop.test.ts} +64 -286
  207. package/src/__tests__/parallel-tool.benchmark.test.ts +24 -36
  208. package/src/__tests__/persist-unsendable-image.test.ts +215 -0
  209. package/src/__tests__/persistence-secret-redaction.test.ts +3 -1
  210. package/src/__tests__/pipeline-runner.test.ts +31 -43
  211. package/src/__tests__/pkb-autoinject.test.ts +2 -5
  212. package/src/__tests__/plugin-bootstrap.test.ts +62 -51
  213. package/src/__tests__/plugin-registry.test.ts +0 -27
  214. package/src/__tests__/plugin-route-contribution.test.ts +6 -16
  215. package/src/__tests__/plugin-skill-contribution.test.ts +7 -17
  216. package/src/__tests__/plugin-tool-contribution.test.ts +10 -26
  217. package/src/__tests__/plugin-types.test.ts +8 -173
  218. package/src/__tests__/prechat-onboarding-contract.test.ts +23 -0
  219. package/src/__tests__/process-message-background-slack.test.ts +17 -16
  220. package/src/__tests__/process-message-display-content.test.ts +36 -44
  221. package/src/__tests__/provider-commit-message-generator.test.ts +19 -14
  222. package/src/__tests__/provider-error-scenarios.test.ts +7 -6
  223. package/src/__tests__/provider-platform-proxy-integration.test.ts +3 -8
  224. package/src/__tests__/provider-send-message-override-profile.test.ts +9 -25
  225. package/src/__tests__/provider-streaming.benchmark.test.ts +12 -22
  226. package/src/__tests__/provider-usage-tracking.test.ts +0 -6
  227. package/src/__tests__/ratelimit.test.ts +9 -4
  228. package/src/__tests__/reaction-persistence.test.ts +1 -1
  229. package/src/__tests__/regenerate-fire-and-forget-trace.test.ts +5 -1
  230. package/src/__tests__/relay-server.test.ts +20 -13
  231. package/src/__tests__/resolve-trust-class.test.ts +4 -4
  232. package/src/__tests__/retry-openrouter-only-normalization.test.ts +5 -8
  233. package/src/__tests__/retry-thinking-tool-choice.test.ts +10 -13
  234. package/src/__tests__/retry-verbosity-normalization.test.ts +5 -8
  235. package/src/__tests__/runtime-events-sse-reconnect.test.ts +390 -0
  236. package/src/__tests__/schedule-routes.test.ts +683 -12
  237. package/src/__tests__/schedule-store.test.ts +108 -0
  238. package/src/__tests__/schedule-tools.test.ts +160 -0
  239. package/src/__tests__/secret-ingress-http.test.ts +2 -2
  240. package/src/__tests__/secret-prompt-log-hygiene.test.ts +11 -7
  241. package/src/__tests__/secret-prompter-channel-fallback.test.ts +11 -9
  242. package/src/__tests__/secret-response-routing.test.ts +13 -11
  243. package/src/__tests__/send-endpoint-busy.test.ts +6 -2
  244. package/src/__tests__/server-history-render.test.ts +314 -1
  245. package/src/__tests__/shell-observability.test.ts +249 -0
  246. package/src/__tests__/skill-feature-flags-integration.test.ts +44 -11
  247. package/src/__tests__/skill-feature-flags.test.ts +6 -6
  248. package/src/__tests__/skill-load-feature-flag.test.ts +10 -10
  249. package/src/__tests__/skills-files-catalog-fallback.test.ts +10 -0
  250. package/src/__tests__/skillssh-files.test.ts +1 -0
  251. package/src/__tests__/starter-task-flow.test.ts +6 -6
  252. package/src/__tests__/strip-memory-injections.test.ts +102 -14
  253. package/src/__tests__/subagent-call-site-routing.test.ts +3 -3
  254. package/src/__tests__/subagent-fork-notifications.test.ts +1 -3
  255. package/src/__tests__/subagent-fork-spawn.test.ts +1 -1
  256. package/src/__tests__/subagent-manager-notify.test.ts +1 -3
  257. package/src/__tests__/subagent-notify-parent.test.ts +1 -3
  258. package/src/__tests__/subagent-spawn-tool-fork.test.ts +1 -1
  259. package/src/__tests__/suggestion-routes.test.ts +3 -3
  260. package/src/__tests__/sync-message-contract.test.ts +19 -16
  261. package/src/__tests__/system-prompt.test.ts +74 -0
  262. package/src/__tests__/task-scheduler.test.ts +162 -1
  263. package/src/__tests__/terminal-tools.test.ts +9 -25
  264. package/src/__tests__/thread-backfill.test.ts +4 -9
  265. package/src/__tests__/title-generate-hook.test.ts +319 -0
  266. package/src/__tests__/tool-error-hook.test.ts +278 -0
  267. package/src/__tests__/tool-preview-lifecycle.test.ts +481 -16
  268. package/src/__tests__/tool-result-metadata-plumbing.test.ts +1 -0
  269. package/src/__tests__/tool-result-truncate-hook.test.ts +127 -0
  270. package/src/__tests__/tool-result-truncation.test.ts +1 -1
  271. package/src/__tests__/tools-audio-read.test.ts +113 -0
  272. package/src/__tests__/turn-boundary-resolution.test.ts +44 -84
  273. package/src/__tests__/turn-events-store.test.ts +11 -7
  274. package/src/__tests__/ui-choice-copy-surfaces.test.ts +254 -0
  275. package/src/__tests__/ui-work-result-surface.test.ts +159 -0
  276. package/src/__tests__/usage-routes.test.ts +285 -1
  277. package/src/__tests__/user-plugin-loader.test.ts +2 -2
  278. package/src/__tests__/voice-scoped-grant-consumer.test.ts +8 -6
  279. package/src/__tests__/voice-session-bridge.test.ts +19 -10
  280. package/src/__tests__/web-search-backend-failure.test.ts +166 -0
  281. package/src/acp/__tests__/agent-process.test.ts +161 -0
  282. package/src/acp/__tests__/client-handler.test.ts +40 -0
  283. package/src/acp/__tests__/helpers/acp-history-db.ts +82 -0
  284. package/src/acp/__tests__/helpers/exec-file-stub.ts +101 -0
  285. package/src/acp/__tests__/prepare-agent-env.test.ts +143 -31
  286. package/src/acp/__tests__/session-manager-persistence.test.ts +95 -28
  287. package/src/acp/__tests__/session-manager-resume.test.ts +695 -0
  288. package/src/acp/agent-process.ts +61 -1
  289. package/src/acp/auto-install.test.ts +125 -0
  290. package/src/acp/auto-install.ts +174 -0
  291. package/src/acp/client-handler.ts +31 -0
  292. package/src/acp/feature-gate.test.ts +48 -0
  293. package/src/acp/feature-gate.ts +34 -0
  294. package/src/acp/prepare-agent-env.ts +52 -11
  295. package/src/acp/resolve-agent.test.ts +147 -6
  296. package/src/acp/resolve-agent.ts +81 -7
  297. package/src/acp/resume-hint.ts +22 -0
  298. package/src/acp/session-manager.ts +487 -71
  299. package/src/agent/compaction-circuit.ts +98 -0
  300. package/src/agent/loop.ts +651 -450
  301. package/src/api/README.md +19 -17
  302. package/src/api/constants/tool-execution.ts +21 -0
  303. package/src/api/events/assistant-activity-state.ts +75 -0
  304. package/src/api/events/assistant-outbound-attachment.ts +25 -27
  305. package/src/api/events/assistant-text-delta.ts +6 -8
  306. package/src/api/events/assistant-thinking-delta.ts +33 -0
  307. package/src/api/events/assistant-turn-start.ts +5 -7
  308. package/src/api/events/avatar-updated.ts +24 -0
  309. package/src/api/events/compaction-circuit-closed.ts +26 -0
  310. package/src/api/events/compaction-circuit-open.ts +28 -0
  311. package/src/api/events/confirmation-request.ts +114 -0
  312. package/src/api/events/contact-request.ts +33 -0
  313. package/src/api/events/conversation-error.ts +77 -0
  314. package/src/api/events/conversation-list-invalidated.ts +38 -0
  315. package/src/api/events/conversation-title-updated.ts +24 -0
  316. package/src/api/events/disk-pressure-status-changed.ts +61 -0
  317. package/src/api/events/document-comment-created.ts +24 -28
  318. package/src/api/events/document-comment-deleted.ts +6 -8
  319. package/src/api/events/document-comment-reopened.ts +6 -8
  320. package/src/api/events/document-comment-resolved.ts +8 -10
  321. package/src/api/events/document-editor-update.ts +27 -0
  322. package/src/api/events/error.ts +32 -0
  323. package/src/api/events/generation-cancelled.ts +4 -6
  324. package/src/api/events/generation-handoff.ts +13 -15
  325. package/src/api/events/home-feed-updated.ts +26 -0
  326. package/src/api/events/identity-changed.ts +32 -0
  327. package/src/api/events/interaction-resolved.ts +50 -0
  328. package/src/api/events/message-complete.ts +10 -12
  329. package/src/api/events/message-dequeued.ts +21 -0
  330. package/src/api/events/message-queued-deleted.ts +23 -0
  331. package/src/api/events/message-queued.ts +22 -0
  332. package/src/api/events/message-request-complete.ts +29 -0
  333. package/src/api/events/navigate-settings.ts +20 -0
  334. package/src/api/events/notification-intent.ts +33 -0
  335. package/src/api/events/open-url.ts +6 -8
  336. package/src/api/events/question-request.ts +67 -0
  337. package/src/api/events/relationship-state-updated.ts +4 -6
  338. package/src/api/events/secret-request.ts +42 -0
  339. package/src/api/events/subagent-event.ts +79 -0
  340. package/src/api/events/subagent-spawned.ts +40 -0
  341. package/src/api/events/subagent-status-changed.ts +65 -0
  342. package/src/api/events/sync-changed.ts +29 -0
  343. package/src/api/events/tool-output-chunk.ts +45 -0
  344. package/src/api/events/tool-result.ts +129 -0
  345. package/src/api/events/tool-use-preview-start.ts +32 -0
  346. package/src/api/events/tool-use-start.ts +8 -10
  347. package/src/api/events/trace-event.ts +69 -0
  348. package/src/api/events/turn-profile-auto-routed.ts +28 -0
  349. package/src/api/events/ui-surface-complete.ts +30 -0
  350. package/src/api/events/ui-surface-dismiss.ts +22 -0
  351. package/src/api/events/ui-surface-show.ts +67 -0
  352. package/src/api/events/ui-surface-update.ts +26 -0
  353. package/src/api/events/usage-update.ts +34 -0
  354. package/src/api/events/user-message-echo.ts +35 -0
  355. package/src/api/index.ts +389 -0
  356. package/src/api/requests/dictation.ts +45 -0
  357. package/src/api/responses/conversation-message.ts +374 -0
  358. package/src/api/responses/disk-pressure-status.ts +26 -0
  359. package/src/api/responses/home.ts +217 -0
  360. package/src/api/responses/llm-context-response.ts +2 -0
  361. package/src/api/responses/memory-v3-selection-log.ts +50 -0
  362. package/src/api/responses/subagent-detail.ts +48 -0
  363. package/src/approvals/guardian-decision-primitive.ts +7 -15
  364. package/src/approvals/guardian-request-resolvers.ts +7 -10
  365. package/src/avatar/__tests__/avatar-manifest.test.ts +236 -0
  366. package/src/avatar/__tests__/avatar-store.test.ts +198 -0
  367. package/src/avatar/avatar-manifest.ts +195 -0
  368. package/src/avatar/avatar-store.ts +113 -0
  369. package/src/avatar/traits-png-sync.ts +8 -2
  370. package/src/background-wake/next-wake.test.ts +31 -1
  371. package/src/background-wake/next-wake.ts +5 -1
  372. package/src/calls/call-conversation-messages.ts +6 -4
  373. package/src/calls/guardian-action-sweep.ts +6 -4
  374. package/src/calls/relay-server.ts +12 -8
  375. package/src/calls/voice-session-bridge.ts +13 -27
  376. package/src/cli/commands/__tests__/memory-v3.test.ts +245 -0
  377. package/src/cli/commands/__tests__/notifications.test.ts +58 -14
  378. package/src/cli/commands/avatar.ts +17 -11
  379. package/src/cli/commands/conversations.ts +15 -1
  380. package/src/cli/commands/db/__tests__/repair.test.ts +540 -0
  381. package/src/cli/commands/db/__tests__/status.test.ts +253 -0
  382. package/src/cli/commands/db/format.ts +48 -0
  383. package/src/cli/commands/db/index.ts +29 -0
  384. package/src/cli/commands/db/repair-step-conversation-backfill.ts +345 -0
  385. package/src/cli/commands/db/repair-step-integrity.ts +146 -0
  386. package/src/cli/commands/db/repair-steps.ts +164 -0
  387. package/src/cli/commands/db/repair.ts +141 -0
  388. package/src/cli/commands/db/status.ts +366 -0
  389. package/src/cli/commands/memory-v3.ts +159 -445
  390. package/src/cli/commands/notifications.ts +112 -60
  391. package/src/cli/lib/cli-colors.ts +24 -6
  392. package/src/cli/program.ts +4 -5
  393. package/src/config/__tests__/feature-flag-registry-guard.test.ts +4 -4
  394. package/src/config/acp-defaults.test.ts +10 -0
  395. package/src/config/acp-defaults.ts +6 -0
  396. package/src/config/assistant-feature-flags.ts +24 -13
  397. package/src/config/bundled-skills/acp/SKILL.md +64 -30
  398. package/src/config/bundled-skills/acp/TOOLS.json +4 -4
  399. package/src/config/bundled-skills/app-builder/SKILL.md +224 -387
  400. package/src/config/bundled-skills/app-builder/TOOLS.json +29 -0
  401. package/src/config/bundled-skills/app-builder/references/DESIGN_SYSTEM.md +48 -0
  402. package/src/config/bundled-skills/app-builder/references/RESPONSIVE.md +57 -0
  403. package/src/config/bundled-skills/app-builder/references/SLIDES.md +38 -0
  404. package/src/config/bundled-skills/app-builder/references/examples/README.md +17 -0
  405. package/src/config/bundled-skills/app-builder/references/examples/expense-tracker.md +515 -0
  406. package/src/config/bundled-skills/app-builder/references/examples/focus-timer.md +342 -0
  407. package/src/config/bundled-skills/app-builder/references/examples/habit-tracker.md +490 -0
  408. package/src/config/bundled-skills/app-builder/tools/app-list.ts +62 -0
  409. package/src/config/bundled-skills/document-editor/SKILL.md +28 -23
  410. package/src/config/bundled-skills/document-editor/TOOLS.json +1 -1
  411. package/src/config/bundled-skills/media-processing/services/reduce.ts +6 -9
  412. package/src/config/bundled-skills/messaging/SKILL.md +0 -7
  413. package/src/config/bundled-skills/messaging/tools/messaging-send.ts +7 -2
  414. package/src/config/bundled-skills/schedule/SKILL.md +1 -1
  415. package/src/config/bundled-skills/schedule/TOOLS.json +8 -0
  416. package/src/config/bundled-tool-registry.ts +2 -0
  417. package/src/config/call-site-defaults.ts +2 -7
  418. package/src/config/feature-flag-cache.ts +3 -3
  419. package/src/config/feature-flag-registry.json +68 -12
  420. package/src/config/schemas/__tests__/memory-v2.test.ts +2 -226
  421. package/src/config/schemas/__tests__/memory-v3.test.ts +25 -0
  422. package/src/config/schemas/call-site-catalog.ts +8 -15
  423. package/src/config/schemas/heartbeat.ts +9 -0
  424. package/src/config/schemas/llm.ts +3 -3
  425. package/src/config/schemas/memory-lifecycle.ts +24 -0
  426. package/src/config/schemas/memory-v2.ts +8 -253
  427. package/src/config/schemas/memory-v3.ts +47 -0
  428. package/src/config/schemas/memory.ts +6 -1
  429. package/src/config/schemas/platform.ts +8 -0
  430. package/src/config/schemas/timeouts.ts +3 -1
  431. package/src/config/seed-inference-profiles.ts +2 -2
  432. package/src/config/skills.ts +13 -0
  433. package/src/context/compactor.ts +55 -32
  434. package/src/context/strip-injections.ts +128 -0
  435. package/src/context/token-estimator.ts +42 -0
  436. package/src/context/tool-result-truncation.ts +1 -66
  437. package/src/context/window-manager.ts +141 -26
  438. package/src/credential-execution/executable-discovery.ts +16 -0
  439. package/src/daemon/__tests__/conversation-lifecycle-auto-analyze.test.ts +6 -0
  440. package/src/daemon/__tests__/conversation-surfaces-launch.test.ts +2 -2
  441. package/src/daemon/__tests__/inference-profile-notification.test.ts +153 -0
  442. package/src/daemon/__tests__/native-web-search-metadata.test.ts +10 -8
  443. package/src/daemon/__tests__/web-search-status-text.test.ts +10 -6
  444. package/src/daemon/approval-generators.ts +4 -4
  445. package/src/daemon/assistant-attachments.ts +1 -1
  446. package/src/daemon/config-watcher.ts +7 -1
  447. package/src/daemon/context-overflow-reducer.ts +0 -1
  448. package/src/daemon/conversation-agent-loop-handlers.ts +793 -215
  449. package/src/daemon/conversation-agent-loop.ts +487 -1478
  450. package/src/daemon/conversation-error.ts +7 -7
  451. package/src/daemon/conversation-history.ts +27 -10
  452. package/src/daemon/conversation-launch.ts +4 -8
  453. package/src/daemon/conversation-lifecycle.ts +13 -42
  454. package/src/daemon/conversation-messaging.ts +8 -9
  455. package/src/daemon/conversation-notifiers.ts +7 -5
  456. package/src/daemon/conversation-process.ts +109 -93
  457. package/src/daemon/conversation-registry.ts +159 -0
  458. package/src/daemon/conversation-runtime-assembly.ts +209 -382
  459. package/src/daemon/conversation-slash.ts +6 -25
  460. package/src/daemon/conversation-store.ts +15 -95
  461. package/src/daemon/conversation-surfaces.ts +277 -73
  462. package/src/daemon/conversation-tool-setup.ts +5 -29
  463. package/src/daemon/conversation-workspace.ts +17 -0
  464. package/src/daemon/conversation.ts +123 -146
  465. package/src/daemon/daemon-skill-host.ts +2 -6
  466. package/src/daemon/disk-pressure-guard.ts +35 -29
  467. package/src/daemon/external-plugins-bootstrap.ts +53 -32
  468. package/src/daemon/first-greeting.ts +26 -4
  469. package/src/daemon/guardian-action-generators.ts +2 -2
  470. package/src/daemon/handlers/config-a2a.ts +51 -36
  471. package/src/daemon/handlers/config-slack-channel.ts +20 -14
  472. package/src/daemon/handlers/config-telegram.ts +16 -2
  473. package/src/daemon/handlers/conversations.ts +9 -23
  474. package/src/daemon/handlers/shared.ts +158 -82
  475. package/src/daemon/handlers/skills.ts +53 -20
  476. package/src/daemon/host-app-control-proxy.ts +54 -1
  477. package/src/daemon/host-cu-proxy.ts +46 -22
  478. package/src/daemon/host-file-proxy.ts +25 -1
  479. package/src/daemon/host-proxy-preactivation.ts +25 -6
  480. package/src/daemon/lifecycle.ts +53 -55
  481. package/src/daemon/message-protocol.ts +2 -3
  482. package/src/daemon/message-provenance.ts +49 -0
  483. package/src/daemon/message-types/apps.ts +1 -29
  484. package/src/daemon/message-types/contacts.ts +3 -20
  485. package/src/daemon/message-types/conversations.ts +13 -111
  486. package/src/daemon/message-types/documents.ts +3 -9
  487. package/src/daemon/message-types/home.ts +4 -17
  488. package/src/daemon/message-types/integrations.ts +2 -6
  489. package/src/daemon/message-types/messages.ts +37 -400
  490. package/src/daemon/message-types/notifications.ts +2 -32
  491. package/src/daemon/message-types/settings.ts +3 -8
  492. package/src/daemon/message-types/skills.ts +4 -0
  493. package/src/daemon/message-types/surfaces.ts +138 -3
  494. package/src/daemon/message-types/sync.ts +12 -25
  495. package/src/daemon/message-types/workspace.ts +3 -11
  496. package/src/daemon/now-scratchpad.ts +21 -0
  497. package/src/daemon/orphan-reaper.test.ts +210 -0
  498. package/src/daemon/orphan-reaper.ts +240 -0
  499. package/src/daemon/overflow-reduction-loop.ts +230 -0
  500. package/src/daemon/persist-unsendable-image.ts +117 -0
  501. package/src/daemon/process-message.ts +50 -49
  502. package/src/daemon/server.ts +14 -0
  503. package/src/daemon/tool-side-effects.ts +10 -7
  504. package/src/daemon/trace-emitter.ts +6 -4
  505. package/src/daemon/trust-context.ts +32 -0
  506. package/src/daemon/wake-target-adapter.ts +14 -2
  507. package/src/heartbeat/__tests__/heartbeat-service.test.ts +6 -1
  508. package/src/heartbeat/heartbeat-run-store.ts +54 -1
  509. package/src/heartbeat/heartbeat-service.ts +42 -0
  510. package/src/home/feed-types.ts +36 -221
  511. package/src/home/home-greeting-cache.ts +24 -1
  512. package/src/ipc/__tests__/browser-ipc.test.ts +1 -1
  513. package/src/ipc/__tests__/email-ipc.test.ts +0 -9
  514. package/src/ipc/__tests__/ui-request-route.test.ts +3 -3
  515. package/src/ipc/gateway-client.test.ts +2 -2
  516. package/src/ipc/gateway-client.ts +3 -3
  517. package/src/ipc/routes/__tests__/route-adapter.test.ts +244 -0
  518. package/src/ipc/routes/route-adapter.ts +45 -6
  519. package/src/ipc/skill-routes/__tests__/memory.test.ts +33 -9
  520. package/src/ipc/skill-routes/__tests__/providers.test.ts +10 -10
  521. package/src/ipc/skill-routes/__tests__/registries.test.ts +28 -18
  522. package/src/ipc/skill-routes/memory.ts +29 -14
  523. package/src/ipc/skill-routes/providers.ts +5 -6
  524. package/src/ipc/skill-routes/registries.ts +13 -61
  525. package/src/live-voice/__tests__/live-voice-archive.test.ts +24 -11
  526. package/src/media/gemini-image-service.ts +15 -0
  527. package/src/media/openai-image-service.ts +14 -0
  528. package/src/media/types.ts +34 -0
  529. package/src/memory/__tests__/conversation-queries.test.ts +192 -8
  530. package/src/memory/__tests__/db-maintenance.test.ts +128 -0
  531. package/src/memory/__tests__/jobs-store-job-classes.test.ts +5 -4
  532. package/src/memory/__tests__/jobs-worker-v2-schedule.test.ts +56 -0
  533. package/src/memory/__tests__/memory-retrospective-job.test.ts +10 -6
  534. package/src/memory/__tests__/memory-v3-selections-migration.test.ts +103 -0
  535. package/src/memory/auth-fallback-events-store.ts +94 -0
  536. package/src/memory/context-search/agent-runner.ts +2 -4
  537. package/src/memory/conversation-crud.ts +39 -8
  538. package/src/memory/conversation-queries.ts +78 -22
  539. package/src/memory/conversation-starter-checkpoints.ts +1 -0
  540. package/src/memory/conversation-title-service.ts +65 -41
  541. package/src/memory/db-init.ts +14 -0
  542. package/src/memory/db-maintenance.ts +18 -2
  543. package/src/memory/graph/__tests__/conversation-graph-memory-registry.test.ts +119 -0
  544. package/src/memory/graph/consolidation.ts +8 -11
  545. package/src/memory/graph/conversation-graph-memory.ts +106 -8
  546. package/src/memory/graph/extraction.ts +6 -9
  547. package/src/memory/graph/narrative.ts +2 -2
  548. package/src/memory/graph/pattern-scan.ts +2 -2
  549. package/src/memory/graph/retriever.ts +20 -26
  550. package/src/memory/graph/tools.ts +4 -4
  551. package/src/memory/job-handlers/conversation-starters.ts +45 -34
  552. package/src/memory/job-handlers/summarization.ts +1 -2
  553. package/src/memory/jobs-store.ts +36 -1
  554. package/src/memory/jobs-worker.ts +82 -43
  555. package/src/memory/llm-request-log-source-clickhouse.ts +5 -31
  556. package/src/memory/llm-request-log-source-local.ts +0 -11
  557. package/src/memory/llm-request-log-source.ts +9 -25
  558. package/src/memory/llm-request-log-store.ts +0 -41
  559. package/src/memory/llm-usage-store.ts +234 -50
  560. package/src/memory/memory-marker.ts +17 -0
  561. package/src/memory/memory-retrospective-job.ts +6 -2
  562. package/src/memory/memory-v2-activation-log-store.ts +1 -83
  563. package/src/memory/migrations/222-strip-placeholder-sentinels-from-messages.ts +6 -5
  564. package/src/memory/migrations/267-llm-usage-events-add-assistant-version.ts +46 -0
  565. package/src/memory/migrations/268-add-memory-v3-selections.ts +28 -0
  566. package/src/memory/migrations/269-schedule-script-timeout.ts +11 -0
  567. package/src/memory/migrations/270-messages-role-created-at-index.ts +18 -0
  568. package/src/memory/migrations/270-schedule-source-conversation.ts +13 -0
  569. package/src/memory/migrations/271-create-auth-fallback-events.ts +21 -0
  570. package/src/memory/migrations/272-acp-session-history-cwd.ts +36 -0
  571. package/src/memory/migrations/__tests__/267-llm-usage-events-add-assistant-version.test.ts +117 -0
  572. package/src/memory/migrations/index.ts +7 -0
  573. package/src/memory/pkb/autoinject.ts +61 -0
  574. package/src/memory/pkb/context.ts +50 -0
  575. package/src/memory/pkb/types.ts +14 -0
  576. package/src/memory/schedule-attribution-sql.ts +104 -0
  577. package/src/memory/schema/acp.ts +4 -0
  578. package/src/memory/schema/infrastructure.ts +27 -0
  579. package/src/memory/usage-grouped-buckets.ts +6 -1
  580. package/src/memory/v2/__tests__/consolidation-job.test.ts +125 -1
  581. package/src/memory/v2/__tests__/migration.test.ts +11 -3
  582. package/src/memory/v2/__tests__/page-index.test.ts +37 -1
  583. package/src/memory/v2/__tests__/router.test.ts +14 -4
  584. package/src/memory/v2/__tests__/sweep-job.test.ts +6 -5
  585. package/src/memory/v2/backfill-jobs.ts +6 -0
  586. package/src/memory/v2/consolidation-job.ts +99 -10
  587. package/src/memory/v2/migration.ts +5 -3
  588. package/src/memory/v2/page-index.ts +11 -0
  589. package/src/memory/v2/router.ts +8 -11
  590. package/src/memory/v2/sweep-job.ts +8 -11
  591. package/src/memory/v2/types.ts +1 -0
  592. package/src/messaging/providers/slack/render-transcript.test.ts +1 -1
  593. package/src/messaging/providers/slack/render-transcript.ts +2 -2
  594. package/src/messaging/style-analyzer.ts +8 -11
  595. package/src/notifications/conversation-pairing.ts +8 -13
  596. package/src/notifications/decision-engine.ts +16 -16
  597. package/src/notifications/home-feed-side-effect.ts +12 -1
  598. package/src/notifications/preference-extractor.ts +11 -14
  599. package/src/permissions/prompter.ts +46 -36
  600. package/src/permissions/question-prompter.test.ts +35 -26
  601. package/src/permissions/question-prompter.ts +6 -10
  602. package/src/plugin-api/constants.ts +4 -0
  603. package/src/plugin-api/index.ts +10 -1
  604. package/src/plugin-api/types.ts +176 -4
  605. package/src/plugins/defaults/compaction/compact.ts +59 -0
  606. package/src/plugins/defaults/compaction/package.json +15 -0
  607. package/src/plugins/defaults/compaction/register.ts +24 -0
  608. package/src/plugins/defaults/empty-response/hooks/stop.ts +126 -0
  609. package/src/plugins/defaults/empty-response/package.json +15 -0
  610. package/src/plugins/defaults/empty-response/register.ts +23 -0
  611. package/src/plugins/defaults/history-repair/hooks/user-prompt-submit.ts +35 -0
  612. package/src/plugins/defaults/history-repair/package.json +15 -0
  613. package/src/plugins/defaults/history-repair/register.ts +24 -0
  614. package/src/{daemon/history-repair.ts → plugins/defaults/history-repair/terminal.ts} +48 -35
  615. package/src/plugins/defaults/index.ts +22 -49
  616. package/src/plugins/defaults/memory-retrieval/hooks/post-compact.ts +95 -0
  617. package/src/plugins/defaults/memory-retrieval/hooks/user-prompt-submit-temp.ts +216 -0
  618. package/src/plugins/defaults/memory-retrieval/injector-chain.ts +35 -0
  619. package/src/plugins/defaults/{injectors.ts → memory-retrieval/injectors.ts} +295 -112
  620. package/src/plugins/defaults/memory-v3-shadow/__tests__/assign.test.ts +242 -0
  621. package/src/plugins/defaults/memory-v3-shadow/__tests__/capabilities.test.ts +118 -0
  622. package/src/plugins/defaults/memory-v3-shadow/__tests__/core.test.ts +39 -0
  623. package/src/plugins/defaults/memory-v3-shadow/__tests__/fixtures/eval-turns.json +36 -0
  624. package/src/plugins/defaults/memory-v3-shadow/__tests__/fixtures/live-turns.json +37 -0
  625. package/src/plugins/defaults/memory-v3-shadow/__tests__/health.test.ts +219 -0
  626. package/src/plugins/defaults/memory-v3-shadow/__tests__/live-integration.test.ts +330 -0
  627. package/src/plugins/defaults/memory-v3-shadow/__tests__/maintain-job.test.ts +288 -0
  628. package/src/plugins/defaults/memory-v3-shadow/__tests__/needle.test.ts +107 -0
  629. package/src/plugins/defaults/memory-v3-shadow/__tests__/orchestrate.test.ts +436 -0
  630. package/src/plugins/defaults/memory-v3-shadow/__tests__/provider-blocks.test.ts +13 -0
  631. package/src/plugins/defaults/memory-v3-shadow/__tests__/reconcile.test.ts +274 -0
  632. package/src/plugins/defaults/memory-v3-shadow/__tests__/render-injection.test.ts +61 -0
  633. package/src/plugins/defaults/memory-v3-shadow/__tests__/router.test.ts +332 -0
  634. package/src/plugins/defaults/memory-v3-shadow/__tests__/selection-log-store.test.ts +179 -0
  635. package/src/plugins/defaults/memory-v3-shadow/__tests__/selector.test.ts +470 -0
  636. package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-plugin.test.ts +432 -0
  637. package/src/plugins/defaults/memory-v3-shadow/__tests__/snapshot.test.ts +168 -0
  638. package/src/plugins/defaults/memory-v3-shadow/__tests__/tree.test.ts +192 -0
  639. package/src/plugins/defaults/memory-v3-shadow/__tests__/types.test.ts +54 -0
  640. package/src/plugins/defaults/memory-v3-shadow/__tests__/working-set-eviction.test.ts +106 -0
  641. package/src/plugins/defaults/memory-v3-shadow/__tests__/working-set-skeleton.test.ts +44 -0
  642. package/src/plugins/defaults/memory-v3-shadow/assign.ts +268 -0
  643. package/src/plugins/defaults/memory-v3-shadow/capabilities.ts +124 -0
  644. package/src/plugins/defaults/memory-v3-shadow/core.ts +26 -0
  645. package/src/plugins/defaults/memory-v3-shadow/data/README.md +84 -0
  646. package/src/plugins/defaults/memory-v3-shadow/data/assignments.json +5 -0
  647. package/src/plugins/defaults/memory-v3-shadow/data/core.json +1 -0
  648. package/src/plugins/defaults/memory-v3-shadow/data/leaves/domain-a/topic-x.md +9 -0
  649. package/src/plugins/defaults/memory-v3-shadow/data/leaves/domain-a/topic-y.md +9 -0
  650. package/src/plugins/defaults/memory-v3-shadow/data/leaves/domain-b/topic-z.md +9 -0
  651. package/src/plugins/defaults/memory-v3-shadow/health.ts +0 -0
  652. package/src/plugins/defaults/memory-v3-shadow/hooks/post-compact.ts +14 -0
  653. package/src/plugins/defaults/memory-v3-shadow/hooks/user-prompt-submit.ts +19 -0
  654. package/src/plugins/defaults/memory-v3-shadow/injector.ts +75 -0
  655. package/src/plugins/defaults/memory-v3-shadow/llm-retry.ts +32 -0
  656. package/src/plugins/defaults/memory-v3-shadow/maintain-job.ts +314 -0
  657. package/src/plugins/defaults/memory-v3-shadow/needle.ts +115 -0
  658. package/src/plugins/defaults/memory-v3-shadow/orchestrate.ts +126 -0
  659. package/src/plugins/defaults/memory-v3-shadow/package.json +15 -0
  660. package/src/plugins/defaults/memory-v3-shadow/page-content.ts +34 -0
  661. package/src/plugins/defaults/memory-v3-shadow/provider-blocks.ts +26 -0
  662. package/src/plugins/defaults/memory-v3-shadow/reconcile.ts +523 -0
  663. package/src/plugins/defaults/memory-v3-shadow/register.ts +26 -0
  664. package/src/plugins/defaults/memory-v3-shadow/render-injection.ts +32 -0
  665. package/src/plugins/defaults/memory-v3-shadow/router.ts +190 -0
  666. package/src/plugins/defaults/memory-v3-shadow/selection-log-store.ts +84 -0
  667. package/src/plugins/defaults/memory-v3-shadow/selector.ts +226 -0
  668. package/src/plugins/defaults/memory-v3-shadow/shadow-plugin.ts +349 -0
  669. package/src/plugins/defaults/memory-v3-shadow/snapshot.ts +209 -0
  670. package/src/plugins/defaults/memory-v3-shadow/tree.ts +174 -0
  671. package/src/plugins/defaults/memory-v3-shadow/types.ts +59 -0
  672. package/src/plugins/defaults/memory-v3-shadow/working-set.ts +88 -0
  673. package/src/plugins/defaults/title-generate/hooks/stop.ts +75 -0
  674. package/src/plugins/defaults/title-generate/hooks/user-prompt-submit.ts +35 -0
  675. package/src/plugins/defaults/title-generate/package.json +15 -0
  676. package/src/plugins/defaults/title-generate/register.ts +35 -0
  677. package/src/plugins/defaults/tool-error/hooks/post-tool-use.ts +118 -0
  678. package/src/plugins/defaults/tool-error/package.json +15 -0
  679. package/src/plugins/defaults/tool-error/register.ts +23 -0
  680. package/src/plugins/defaults/tool-result-truncate/hooks/post-tool-use.ts +32 -0
  681. package/src/plugins/defaults/tool-result-truncate/package.json +15 -0
  682. package/src/plugins/defaults/tool-result-truncate/register.ts +24 -0
  683. package/src/plugins/defaults/tool-result-truncate/terminal.ts +132 -0
  684. package/src/plugins/external-plugin-loader.ts +2 -2
  685. package/src/plugins/pipeline.ts +8 -35
  686. package/src/plugins/registry.ts +8 -25
  687. package/src/plugins/types.ts +62 -721
  688. package/src/plugins/user-loader.ts +4 -3
  689. package/src/proactive-artifact/aux-message-injector.ts +4 -5
  690. package/src/proactive-artifact/job.test.ts +28 -21
  691. package/src/proactive-artifact/job.ts +3 -1
  692. package/src/prompts/__tests__/system-prompt.test.ts +42 -0
  693. package/src/prompts/sections.ts +20 -7
  694. package/src/prompts/templates/BOOTSTRAP-ACTIVATION-RAIL.md +64 -0
  695. package/src/prompts/templates/BOOTSTRAP-CONTENT-AUTOMATION.md +2 -2
  696. package/src/prompts/templates/BOOTSTRAP.md +7 -3
  697. package/src/prompts/templates/system-sections.ts +21 -0
  698. package/src/providers/__tests__/retry-callsite.test.ts +25 -25
  699. package/src/providers/__tests__/satellite-connection-routing.test.ts +7 -21
  700. package/src/providers/anthropic/client.ts +61 -34
  701. package/src/providers/call-site-routing.ts +1 -9
  702. package/src/providers/gemini/client.ts +152 -34
  703. package/src/providers/gemini/inline-media.ts +74 -0
  704. package/src/providers/openai/__tests__/chat-completions-provider-reasoning.test.ts +112 -2
  705. package/src/providers/openai/chat-completions-provider.ts +45 -4
  706. package/src/providers/openai/responses-provider.ts +1 -4
  707. package/src/providers/openrouter/client.ts +2 -6
  708. package/src/providers/placeholder-sentinels.ts +35 -0
  709. package/src/providers/provider-send-message.ts +6 -6
  710. package/src/providers/ratelimit.ts +1 -9
  711. package/src/providers/retry.ts +0 -5
  712. package/src/providers/types.ts +11 -2
  713. package/src/providers/usage-tracking.ts +1 -9
  714. package/src/runtime/__tests__/agent-wake.test.ts +141 -32
  715. package/src/runtime/__tests__/background-job-runner.test.ts +1 -3
  716. package/src/runtime/__tests__/interactive-ui.test.ts +1 -1
  717. package/src/runtime/agent-wake.ts +95 -23
  718. package/src/runtime/assistant-event-hub.ts +38 -8
  719. package/src/runtime/assistant-stream-state.ts +368 -0
  720. package/src/runtime/auth/__tests__/guard-tests.test.ts +75 -109
  721. package/src/runtime/auth/__tests__/route-policy.test.ts +153 -170
  722. package/src/runtime/auth/route-policy.ts +42 -1079
  723. package/src/runtime/background-job-runner.ts +1 -4
  724. package/src/runtime/btw-sidechain.ts +3 -1
  725. package/src/runtime/channel-approvals.ts +4 -15
  726. package/src/runtime/channel-invite-transport.ts +5 -6
  727. package/src/runtime/channel-readiness-service.ts +2 -5
  728. package/src/runtime/channel-retry-sweep.ts +12 -16
  729. package/src/runtime/http-router.ts +35 -43
  730. package/src/runtime/http-types.ts +23 -71
  731. package/src/runtime/interactive-ui.ts +1 -1
  732. package/src/runtime/invite-instruction-generator.ts +3 -3
  733. package/src/runtime/pending-interactions.ts +3 -2
  734. package/src/runtime/routes/__tests__/acp-routes.test.ts +253 -55
  735. package/src/runtime/routes/__tests__/avatar-state-routes.test.ts +565 -0
  736. package/src/runtime/routes/__tests__/consolidation-routes.test.ts +265 -2
  737. package/src/runtime/routes/__tests__/content-source-routes.test.ts +4 -4
  738. package/src/runtime/routes/__tests__/conversation-compaction-routes.test.ts +62 -32
  739. package/src/runtime/routes/__tests__/conversation-list-routes.test.ts +237 -0
  740. package/src/runtime/routes/__tests__/conversation-query-routes.test.ts +31 -1
  741. package/src/runtime/routes/__tests__/inference-provider-connection-routes.test.ts +13 -22
  742. package/src/runtime/routes/__tests__/memory-v2-routes.test.ts +6 -2
  743. package/src/runtime/routes/__tests__/memory-v2-simulate-route.test.ts +7 -2
  744. package/src/runtime/routes/__tests__/sanity-routes.test.ts +6 -6
  745. package/src/runtime/routes/__tests__/stt-routes.test.ts +3 -3
  746. package/src/runtime/routes/__tests__/suggest-trust-rule-routes.test.ts +5 -2
  747. package/src/runtime/routes/__tests__/surface-action-routes.test.ts +5 -4
  748. package/src/runtime/routes/__tests__/surface-content-routes.test.ts +4 -1
  749. package/src/runtime/routes/__tests__/tts-routes.test.ts +9 -5
  750. package/src/runtime/routes/acp-routes.test.ts +186 -100
  751. package/src/runtime/routes/acp-routes.ts +110 -35
  752. package/src/runtime/routes/app-management-routes.ts +93 -131
  753. package/src/runtime/routes/app-routes.ts +38 -20
  754. package/src/runtime/routes/approval-routes.ts +17 -5
  755. package/src/runtime/routes/attachment-routes.ts +51 -16
  756. package/src/runtime/routes/audio-routes.ts +1 -0
  757. package/src/runtime/routes/audit-routes.ts +5 -0
  758. package/src/runtime/routes/auth-routes.ts +5 -0
  759. package/src/runtime/routes/avatar-routes.ts +264 -59
  760. package/src/runtime/routes/background-tool-routes.ts +9 -0
  761. package/src/runtime/routes/background-wake-routes.ts +13 -3
  762. package/src/runtime/routes/backup-routes.ts +45 -0
  763. package/src/runtime/routes/bookmark-routes.ts +13 -0
  764. package/src/runtime/routes/brain-graph-routes.ts +9 -0
  765. package/src/runtime/routes/browser-routes.ts +6 -1
  766. package/src/runtime/routes/browser-tabs-routes.ts +11 -10
  767. package/src/runtime/routes/btw-routes.ts +34 -24
  768. package/src/runtime/routes/cache-routes.ts +13 -0
  769. package/src/runtime/routes/call-routes.ts +21 -10
  770. package/src/runtime/routes/channel-availability-routes.ts +5 -1
  771. package/src/runtime/routes/channel-readiness-routes.ts +37 -4
  772. package/src/runtime/routes/channel-route-definitions.ts +21 -0
  773. package/src/runtime/routes/channel-verification-routes.ts +21 -0
  774. package/src/runtime/routes/chatgpt-subscription-auth-routes.ts +9 -2
  775. package/src/runtime/routes/client-routes.ts +9 -0
  776. package/src/runtime/routes/consolidation-routes.ts +133 -25
  777. package/src/runtime/routes/contact-prompt-routes.ts +9 -0
  778. package/src/runtime/routes/contact-routes.ts +90 -23
  779. package/src/runtime/routes/content-source-routes.ts +5 -1
  780. package/src/runtime/routes/conversation-analysis-routes.ts +5 -1
  781. package/src/runtime/routes/conversation-attention-routes.ts +5 -0
  782. package/src/runtime/routes/conversation-cli-routes.ts +54 -7
  783. package/src/runtime/routes/conversation-compaction-routes.ts +54 -25
  784. package/src/runtime/routes/conversation-list-routes.ts +81 -12
  785. package/src/runtime/routes/conversation-management-routes.ts +57 -14
  786. package/src/runtime/routes/conversation-query-routes.ts +90 -41
  787. package/src/runtime/routes/conversation-routes.ts +446 -204
  788. package/src/runtime/routes/conversation-starter-routes.ts +35 -20
  789. package/src/runtime/routes/conversations-import-routes.ts +30 -8
  790. package/src/runtime/routes/credential-prompt-routes.ts +5 -0
  791. package/src/runtime/routes/credential-routes.ts +25 -6
  792. package/src/runtime/routes/debug-bash-routes.ts +5 -0
  793. package/src/runtime/routes/debug-routes.ts +11 -2
  794. package/src/runtime/routes/defer-routes.ts +13 -0
  795. package/src/runtime/routes/diagnostics-routes.ts +37 -46
  796. package/src/runtime/routes/disk-pressure-routes.ts +17 -31
  797. package/src/runtime/routes/document-comments-routes.ts +46 -27
  798. package/src/runtime/routes/documents-routes.ts +25 -10
  799. package/src/runtime/routes/domain-routes.ts +98 -51
  800. package/src/runtime/routes/email-routes.ts +33 -0
  801. package/src/runtime/routes/epoch-millis-range.ts +34 -0
  802. package/src/runtime/routes/events-routes.ts +107 -8
  803. package/src/runtime/routes/filing-routes.ts +9 -4
  804. package/src/runtime/routes/gateway-log-routes.ts +31 -4
  805. package/src/runtime/routes/global-search-routes.ts +53 -50
  806. package/src/runtime/routes/group-routes.ts +21 -5
  807. package/src/runtime/routes/guardian-action-routes.ts +9 -0
  808. package/src/runtime/routes/guardian-approval-interception.ts +0 -31
  809. package/src/runtime/routes/heartbeat-routes.ts +57 -21
  810. package/src/runtime/routes/home-feed-routes.ts +23 -19
  811. package/src/runtime/routes/home-state-routes.ts +8 -40
  812. package/src/runtime/routes/host-app-control-routes.ts +6 -1
  813. package/src/runtime/routes/host-bash-routes.ts +5 -0
  814. package/src/runtime/routes/host-browser-routes.ts +13 -0
  815. package/src/runtime/routes/host-cu-routes.ts +6 -1
  816. package/src/runtime/routes/host-file-routes.ts +26 -6
  817. package/src/runtime/routes/host-transfer-routes.ts +13 -2
  818. package/src/runtime/routes/http-adapter.ts +1 -2
  819. package/src/runtime/routes/identity-intro-cache.ts +28 -40
  820. package/src/runtime/routes/identity-routes.ts +236 -20
  821. package/src/runtime/routes/image-generation-routes.ts +45 -2
  822. package/src/runtime/routes/inbound-message-handler.ts +16 -12
  823. package/src/runtime/routes/inbound-stages/background-dispatch.test.ts +0 -12
  824. package/src/runtime/routes/inbound-stages/background-dispatch.ts +15 -19
  825. package/src/runtime/routes/index.ts +2 -0
  826. package/src/runtime/routes/inference-profile-session-routes.ts +13 -3
  827. package/src/runtime/routes/inference-provider-connection-routes.ts +21 -5
  828. package/src/runtime/routes/inference-send-routes.ts +11 -11
  829. package/src/runtime/routes/integrations/a2a.ts +32 -7
  830. package/src/runtime/routes/integrations/slack/__tests__/channel.test.ts +16 -0
  831. package/src/runtime/routes/integrations/slack/channel.ts +23 -3
  832. package/src/runtime/routes/integrations/slack/share.ts +36 -8
  833. package/src/runtime/routes/integrations/telegram.ts +34 -9
  834. package/src/runtime/routes/integrations/twilio.ts +77 -7
  835. package/src/runtime/routes/integrations/vercel.ts +3 -3
  836. package/src/runtime/routes/internal-oauth-routes.ts +5 -0
  837. package/src/runtime/routes/internal-telemetry-routes.ts +88 -0
  838. package/src/runtime/routes/internal-twilio-routes.ts +13 -0
  839. package/src/runtime/routes/llm-call-sites-routes.ts +39 -4
  840. package/src/runtime/routes/log-export-routes.ts +36 -10
  841. package/src/runtime/routes/mcp-auth-routes.ts +25 -0
  842. package/src/runtime/routes/memory-item-routes.ts +21 -10
  843. package/src/runtime/routes/memory-v2-routes.ts +105 -44
  844. package/src/runtime/routes/memory-v3-routes.ts +306 -408
  845. package/src/runtime/routes/migration-rollback-routes.ts +5 -1
  846. package/src/runtime/routes/migration-routes.ts +29 -0
  847. package/src/runtime/routes/notification-routes.ts +17 -1
  848. package/src/runtime/routes/oauth-apps.ts +99 -23
  849. package/src/runtime/routes/oauth-commands-routes.ts +37 -14
  850. package/src/runtime/routes/oauth-connect-routes.ts +9 -0
  851. package/src/runtime/routes/oauth-lifecycle-routes.ts +5 -1
  852. package/src/runtime/routes/oauth-providers.ts +79 -15
  853. package/src/runtime/routes/platform-routes.ts +102 -5
  854. package/src/runtime/routes/playground/__tests__/force-compact.test.ts +9 -6
  855. package/src/runtime/routes/playground/__tests__/inject-failures.test.ts +37 -16
  856. package/src/runtime/routes/playground/__tests__/reset-circuit.test.ts +7 -3
  857. package/src/runtime/routes/playground/__tests__/state.test.ts +10 -3
  858. package/src/runtime/routes/playground/force-compact.ts +2 -2
  859. package/src/runtime/routes/playground/helpers.ts +1 -2
  860. package/src/runtime/routes/playground/inject-failures.ts +13 -8
  861. package/src/runtime/routes/playground/reset-circuit.ts +14 -9
  862. package/src/runtime/routes/playground/seed-conversation.ts +1 -1
  863. package/src/runtime/routes/playground/seeded-conversations.ts +3 -3
  864. package/src/runtime/routes/playground/state.ts +4 -3
  865. package/src/runtime/routes/plugins-routes.ts +22 -19
  866. package/src/runtime/routes/profiler-routes.ts +17 -4
  867. package/src/runtime/routes/ps-routes.ts +5 -0
  868. package/src/runtime/routes/publish-routes.ts +13 -3
  869. package/src/runtime/routes/question-routes.ts +5 -0
  870. package/src/runtime/routes/recording-routes.ts +25 -12
  871. package/src/runtime/routes/rename-conversation-routes.ts +10 -0
  872. package/src/runtime/routes/sanity-routes.ts +9 -2
  873. package/src/runtime/routes/schedule-routes.ts +288 -88
  874. package/src/runtime/routes/secret-routes.ts +31 -6
  875. package/src/runtime/routes/sequence-routes.ts +33 -0
  876. package/src/runtime/routes/settings-routes.ts +65 -19
  877. package/src/runtime/routes/skills-routes.ts +166 -73
  878. package/src/runtime/routes/slack-channel-routes.ts +5 -0
  879. package/src/runtime/routes/stt-routes.ts +13 -6
  880. package/src/runtime/routes/subagents-routes.ts +24 -18
  881. package/src/runtime/routes/suggest-trust-rule-routes.ts +7 -2
  882. package/src/runtime/routes/surface-action-routes.ts +9 -0
  883. package/src/runtime/routes/surface-content-routes.ts +10 -2
  884. package/src/runtime/routes/surface-conversation-resolver.ts +4 -3
  885. package/src/runtime/routes/task-routes.ts +37 -0
  886. package/src/runtime/routes/telemetry-routes.ts +9 -0
  887. package/src/runtime/routes/tool-call-confirmation-enrichment.test.ts +161 -0
  888. package/src/runtime/routes/tool-call-confirmation-enrichment.ts +107 -0
  889. package/src/runtime/routes/trace-event-routes.ts +42 -1
  890. package/src/runtime/routes/trust-rules-routes.ts +31 -2
  891. package/src/runtime/routes/tts-routes.ts +48 -6
  892. package/src/runtime/routes/types.ts +83 -16
  893. package/src/runtime/routes/ui-request-routes.ts +5 -0
  894. package/src/runtime/routes/upgrade-broadcast-routes.ts +5 -0
  895. package/src/runtime/routes/usage-routes.ts +118 -42
  896. package/src/runtime/routes/user-routes-cli.ts +9 -0
  897. package/src/runtime/routes/user-routes.ts +5 -1
  898. package/src/runtime/routes/wake-conversation-routes.ts +5 -0
  899. package/src/runtime/routes/watcher-routes.ts +21 -0
  900. package/src/runtime/routes/webhook-routes.ts +50 -2
  901. package/src/runtime/routes/wipe-conversation-routes.ts +5 -0
  902. package/src/runtime/routes/work-items-routes.ts +49 -23
  903. package/src/runtime/routes/workspace-commit-routes.ts +5 -0
  904. package/src/runtime/routes/workspace-routes.test.ts +42 -0
  905. package/src/runtime/routes/workspace-routes.ts +124 -9
  906. package/src/runtime/services/__tests__/analyze-conversation.test.ts +8 -4
  907. package/src/runtime/services/analyze-conversation.ts +5 -8
  908. package/src/runtime/services/conversation-serializer.ts +24 -2
  909. package/src/runtime/sync/resource-sync-events.ts +16 -2
  910. package/src/runtime/sync/sync-publisher.ts +2 -2
  911. package/src/schedule/run-script.ts +28 -3
  912. package/src/schedule/schedule-store.ts +28 -1
  913. package/src/schedule/schedule-usage-store.ts +83 -0
  914. package/src/schedule/scheduler.ts +15 -6
  915. package/src/signals/cancel.ts +2 -4
  916. package/src/signals/user-message.ts +5 -8
  917. package/src/skills/catalog-files.ts +4 -1
  918. package/src/skills/catalog-install.ts +3 -0
  919. package/src/skills/categories-cache.ts +118 -0
  920. package/src/skills/clawhub-files.ts +1 -0
  921. package/src/skills/skillssh-files.ts +1 -0
  922. package/src/subagent/manager.ts +20 -11
  923. package/src/telemetry/types.ts +55 -1
  924. package/src/telemetry/usage-telemetry-reporter.test.ts +250 -4
  925. package/src/telemetry/usage-telemetry-reporter.ts +88 -2
  926. package/src/tools/acp/context.ts +20 -0
  927. package/src/tools/acp/list-agents.test.ts +7 -1
  928. package/src/tools/acp/spawn.test.ts +198 -93
  929. package/src/tools/acp/spawn.ts +32 -70
  930. package/src/tools/acp/steer.test.ts +105 -8
  931. package/src/tools/acp/steer.ts +48 -17
  932. package/src/tools/apps/definitions.ts +8 -4
  933. package/src/tools/apps/executors.ts +13 -8
  934. package/src/tools/ask-question/ask-question-tool.test.ts +120 -105
  935. package/src/tools/ask-question/ask-question-tool.ts +85 -90
  936. package/src/tools/computer-use/definitions.ts +28 -24
  937. package/src/tools/credential-execution/make-authenticated-request.ts +56 -51
  938. package/src/tools/credential-execution/manage-secure-command-tool.ts +2 -2
  939. package/src/tools/credential-execution/run-authenticated-command.ts +82 -77
  940. package/src/tools/credentials/vault.ts +112 -111
  941. package/src/tools/execution-target.ts +1 -1
  942. package/src/tools/execution-timeout.ts +3 -4
  943. package/src/tools/executor.ts +1 -53
  944. package/src/tools/filesystem/edit.ts +45 -42
  945. package/src/tools/filesystem/list.ts +33 -30
  946. package/src/tools/filesystem/read.ts +54 -35
  947. package/src/tools/filesystem/write.ts +69 -32
  948. package/src/tools/host-filesystem/edit.ts +44 -42
  949. package/src/tools/host-filesystem/read.ts +49 -35
  950. package/src/tools/host-filesystem/transfer.ts +121 -108
  951. package/src/tools/host-filesystem/write.ts +33 -31
  952. package/src/tools/host-terminal/host-shell.ts +50 -48
  953. package/src/tools/memory/register.ts +23 -24
  954. package/src/tools/network/__tests__/web-search-metadata.test.ts +7 -1
  955. package/src/tools/network/__tests__/web-search.test.ts +11 -3
  956. package/src/tools/network/web-fetch.ts +49 -46
  957. package/src/tools/network/web-search-error.test.ts +248 -0
  958. package/src/tools/network/web-search-error.ts +267 -0
  959. package/src/tools/network/web-search.ts +223 -61
  960. package/src/tools/registry.ts +39 -16
  961. package/src/tools/schedule/create.ts +13 -0
  962. package/src/tools/schedule/update.ts +16 -0
  963. package/src/tools/shared/filesystem/audio-read.ts +122 -0
  964. package/src/tools/shared/filesystem/image-read.ts +1 -1
  965. package/src/tools/skills/execute.ts +34 -31
  966. package/src/tools/skills/load.ts +29 -23
  967. package/src/tools/subagent/notify-parent.ts +35 -32
  968. package/src/tools/subagent/spawn.ts +2 -4
  969. package/src/tools/system/avatar-generator.ts +13 -22
  970. package/src/tools/system/request-permission.ts +30 -27
  971. package/src/tools/terminal/safe-env.ts +10 -1
  972. package/src/tools/terminal/shell.ts +190 -61
  973. package/src/tools/tool-defaults.ts +20 -9
  974. package/src/tools/tool-manifest.ts +4 -4
  975. package/src/tools/types.ts +74 -23
  976. package/src/tools/ui-surface/definitions.ts +99 -10
  977. package/src/tts/__tests__/provider-catalog-consistency.test.ts +85 -1
  978. package/src/tts/provider-catalog.ts +76 -1
  979. package/src/usage/types.ts +10 -0
  980. package/src/util/errors.ts +2 -2
  981. package/src/util/map-limit.ts +27 -0
  982. package/src/util/mutex.ts +47 -0
  983. package/src/util/platform.ts +15 -12
  984. package/src/work-items/work-item-runner.ts +7 -2
  985. package/src/workspace/git-service.ts +1 -42
  986. package/src/workspace/migrations/028-recover-conversations-from-disk-view.ts +7 -20
  987. package/src/workspace/migrations/092-backfill-v3-leaves.ts +169 -0
  988. package/src/workspace/migrations/093-backfill-leaf-ids.ts +144 -0
  989. package/src/workspace/migrations/094-seed-avatar-manifest.ts +155 -0
  990. package/src/workspace/migrations/095-bump-heartbeat-interval-30m-to-60m.ts +51 -0
  991. package/src/workspace/migrations/096-reduce-quality-profile-effort.ts +72 -0
  992. package/src/workspace/migrations/097-enable-adaptive-thinking-managed-profiles.ts +117 -0
  993. package/src/workspace/migrations/__tests__/094-seed-avatar-manifest.test.ts +136 -0
  994. package/src/workspace/migrations/__tests__/backfill-leaf-ids.test.ts +175 -0
  995. package/src/workspace/migrations/__tests__/backfill-v3-leaves.test.ts +124 -0
  996. package/src/workspace/migrations/registry.ts +12 -0
  997. package/src/workspace/provider-commit-message-generator.ts +15 -17
  998. package/tsconfig.json +4 -1
  999. package/src/__tests__/bootstrap-turn-cleanup.test.ts +0 -44
  1000. package/src/__tests__/circuit-breaker-pipeline.test.ts +0 -405
  1001. package/src/__tests__/compaction-pipeline.test.ts +0 -210
  1002. package/src/__tests__/compaction-timeout-recovery.test.ts +0 -262
  1003. package/src/__tests__/empty-response-pipeline.test.ts +0 -301
  1004. package/src/__tests__/history-repair-pipeline.test.ts +0 -396
  1005. package/src/__tests__/llm-call-pipeline.test.ts +0 -281
  1006. package/src/__tests__/memory-retrieval-pipeline.test.ts +0 -418
  1007. package/src/__tests__/persistence-pipeline.test.ts +0 -514
  1008. package/src/__tests__/title-generate-pipeline.test.ts +0 -211
  1009. package/src/__tests__/token-estimate-pipeline.test.ts +0 -481
  1010. package/src/__tests__/tool-error-pipeline.test.ts +0 -241
  1011. package/src/__tests__/tool-execute-pipeline.test.ts +0 -417
  1012. package/src/__tests__/tool-result-truncate-pipeline.test.ts +0 -344
  1013. package/src/cli/commands/__tests__/memory-v3-render.test.ts +0 -340
  1014. package/src/cli/commands/memory-v3-render.ts +0 -491
  1015. package/src/daemon/bootstrap-turn-cleanup.ts +0 -45
  1016. package/src/daemon/message-types/disk-pressure.ts +0 -9
  1017. package/src/email/feature-gate.ts +0 -23
  1018. package/src/gallery/default-gallery.ts +0 -1359
  1019. package/src/gallery/gallery-manifest.ts +0 -28
  1020. package/src/memory/v3/__tests__/coactivation-store.test.ts +0 -422
  1021. package/src/memory/v3/__tests__/consolidation-job.test.ts +0 -466
  1022. package/src/memory/v3/__tests__/coretrieval-seed.test.ts +0 -270
  1023. package/src/memory/v3/__tests__/edge-learning-job.test.ts +0 -324
  1024. package/src/memory/v3/__tests__/edges.test.ts +0 -706
  1025. package/src/memory/v3/__tests__/filter.test.ts +0 -560
  1026. package/src/memory/v3/__tests__/gate.test.ts +0 -637
  1027. package/src/memory/v3/__tests__/index-composition.test.ts +0 -291
  1028. package/src/memory/v3/__tests__/loop.test.ts +0 -775
  1029. package/src/memory/v3/__tests__/retriever.test.ts +0 -226
  1030. package/src/memory/v3/__tests__/scouts.test.ts +0 -489
  1031. package/src/memory/v3/__tests__/shadow-diff.test.ts +0 -225
  1032. package/src/memory/v3/__tests__/shadow-middleware.test.ts +0 -398
  1033. package/src/memory/v3/__tests__/system-prompts.test.ts +0 -154
  1034. package/src/memory/v3/__tests__/traversal.test.ts +0 -508
  1035. package/src/memory/v3/__tests__/tree-index.test.ts +0 -280
  1036. package/src/memory/v3/__tests__/tree-store.test.ts +0 -529
  1037. package/src/memory/v3/__tests__/tree-walk.test.ts +0 -784
  1038. package/src/memory/v3/__tests__/validate.test.ts +0 -277
  1039. package/src/memory/v3/auto-edges.ts +0 -223
  1040. package/src/memory/v3/coactivation-store.ts +0 -124
  1041. package/src/memory/v3/consolidation-job.ts +0 -323
  1042. package/src/memory/v3/coretrieval-seed.ts +0 -240
  1043. package/src/memory/v3/edge-learning-job.ts +0 -160
  1044. package/src/memory/v3/edges.ts +0 -286
  1045. package/src/memory/v3/filter.ts +0 -286
  1046. package/src/memory/v3/gate.ts +0 -349
  1047. package/src/memory/v3/index-composition.ts +0 -126
  1048. package/src/memory/v3/llm-capture.ts +0 -46
  1049. package/src/memory/v3/loop.ts +0 -430
  1050. package/src/memory/v3/maintenance.ts +0 -144
  1051. package/src/memory/v3/prompt-context.ts +0 -33
  1052. package/src/memory/v3/prompts/consolidation.ts +0 -458
  1053. package/src/memory/v3/prompts/system-prompts.ts +0 -196
  1054. package/src/memory/v3/retriever.ts +0 -33
  1055. package/src/memory/v3/scouts.ts +0 -431
  1056. package/src/memory/v3/shadow-diff.ts +0 -287
  1057. package/src/memory/v3/shadow-middleware.ts +0 -347
  1058. package/src/memory/v3/traversal.ts +0 -211
  1059. package/src/memory/v3/tree-index.ts +0 -237
  1060. package/src/memory/v3/tree-store.ts +0 -394
  1061. package/src/memory/v3/tree-walk.ts +0 -356
  1062. package/src/memory/v3/types.ts +0 -65
  1063. package/src/memory/v3/validate.ts +0 -323
  1064. package/src/plugins/defaults/circuit-breaker.ts +0 -141
  1065. package/src/plugins/defaults/compaction.ts +0 -141
  1066. package/src/plugins/defaults/empty-response.ts +0 -124
  1067. package/src/plugins/defaults/history-repair.ts +0 -83
  1068. package/src/plugins/defaults/llm-call.ts +0 -77
  1069. package/src/plugins/defaults/memory-retrieval.ts +0 -219
  1070. package/src/plugins/defaults/overflow-reduce.ts +0 -185
  1071. package/src/plugins/defaults/persistence.ts +0 -146
  1072. package/src/plugins/defaults/title-generate.ts +0 -90
  1073. package/src/plugins/defaults/token-estimate.ts +0 -101
  1074. package/src/plugins/defaults/tool-error.ts +0 -119
  1075. package/src/plugins/defaults/tool-execute.ts +0 -87
  1076. package/src/plugins/defaults/tool-result-truncate.ts +0 -84
  1077. package/src/runtime/routes/__tests__/memory-v3-simulate-params.test.ts +0 -35
  1078. package/src/skills/category-inference.ts +0 -111
@@ -1,14 +1,18 @@
1
1
  import { createRequire } from "node:module";
2
- import { afterAll, beforeEach, describe, expect, mock, test } from "bun:test";
3
-
4
- import type {
5
- AgentEvent,
6
- CheckpointDecision,
7
- CheckpointInfo,
8
- } from "../agent/loop.js";
2
+ import {
3
+ afterAll,
4
+ beforeEach,
5
+ describe,
6
+ expect,
7
+ mock,
8
+ spyOn,
9
+ test,
10
+ } from "bun:test";
11
+
12
+ import type { LoopToolExecutor } from "../agent/loop.js";
9
13
  import type { ServerMessage } from "../daemon/message-protocol.js";
10
14
  import { resetPluginRegistryAndRegisterDefaults } from "../plugins/defaults/index.js";
11
- import type { ContentBlock, Message } from "../providers/types.js";
15
+ import type { Message, Provider, ToolDefinition } from "../providers/types.js";
12
16
 
13
17
  const conversationCrudRealSnapshot = {
14
18
  ...(createRequire(import.meta.url)(
@@ -64,6 +68,7 @@ mock.module("../config/loader.js", () => ({
64
68
  memory: { retrieval: { scratchpadInjection: { enabled: true } } },
65
69
  ui: mockUiConfig,
66
70
  compaction: { enabled: true, autoThreshold: 0.7 },
71
+ conversations: { skipAutoRetitling: true },
67
72
  }),
68
73
  loadRawConfig: () => ({}),
69
74
  saveRawConfig: () => {},
@@ -74,17 +79,20 @@ mock.module("../config/loader.js", () => ({
74
79
 
75
80
  // Token estimator returns a small value by default (well within budget)
76
81
  // so preflight does not trigger unless the test overrides it. Both the
77
- // calibrated entry point (`estimatePromptTokens`, used in the convergence
78
- // path) and the raw entry point (`estimatePromptTokensRaw`, used by the
79
- // default `tokenEstimate` plugin pipeline for preflight/mid-loop) are
82
+ // calibrated entry point (`estimatePromptTokens`, which backs the preflight
83
+ // overflow gate and the convergence path) and the raw entry point
84
+ // (`estimatePromptTokensRaw`, used by the pre-send calibration capture) are
80
85
  // stubbed so either call site can drive the test.
81
86
  let mockEstimateTokens = 1000;
82
87
  mock.module("../context/token-estimator.js", () => ({
83
88
  estimatePromptTokens: () => mockEstimateTokens,
84
89
  estimatePromptTokensRaw: () => mockEstimateTokens,
85
- // Pass-through: the default plugin computes `toolTokenBudget` via this
86
- // helper before delegating to the raw estimator. Return 0 so the mocked
87
- // raw estimate is not perturbed.
90
+ // The preflight overflow gate calls this calibrated wrapper directly, so it
91
+ // must honor `mockEstimateTokens` too rather than fall through to the real
92
+ // implementation.
93
+ estimatePromptTokensWithTools: () => mockEstimateTokens,
94
+ // Pass-through: `estimatePromptTokensWithTools` computes `toolTokenBudget`
95
+ // via this helper. Return 0 so the mocked estimate is not perturbed.
88
96
  estimateToolsTokens: () => 0,
89
97
  }));
90
98
 
@@ -308,12 +316,14 @@ const buildUnifiedTurnContextBlockMock = mock(
308
316
  (options: Record<string, unknown>) =>
309
317
  `<turn_context>\ncurrent_time: ${String(options.timestamp)}\n</turn_context>`,
310
318
  );
311
- const applyRuntimeInjectionsMock = mock(
312
- async (msgs: Message[], _options?: unknown) => ({
313
- messages: msgs,
314
- blocks: { ...mockInjectionBlocks },
315
- }),
316
- );
319
+ const defaultApplyRuntimeInjectionsImpl = async (
320
+ msgs: Message[],
321
+ _options?: unknown,
322
+ ) => ({
323
+ messages: msgs,
324
+ blocks: { ...mockInjectionBlocks },
325
+ });
326
+ const applyRuntimeInjectionsMock = mock(defaultApplyRuntimeInjectionsImpl);
317
327
  let mockSlackChronologicalContext: {
318
328
  renderedMessages: Array<{
319
329
  message: Message;
@@ -352,15 +362,6 @@ mock.module("../daemon/conversation-runtime-assembly.js", () => ({
352
362
  applyRuntimeInjections: applyRuntimeInjectionsMock,
353
363
  buildUnifiedTurnContextBlock: buildUnifiedTurnContextBlockMock,
354
364
  stripInjectionsForCompaction: (msgs: Message[]) => msgs,
355
- findLastInjectedNowContent: () => null,
356
- readNowScratchpad: () => null,
357
- readPkbContext: () => null,
358
- getPkbAutoInjectList: () => [
359
- "INDEX.md",
360
- "essentials.md",
361
- "threads.md",
362
- "buffer.md",
363
- ],
364
365
  isSlackChannelConversation: () => false,
365
366
  getSlackCompactionWatermarkForPrefix:
366
367
  getSlackCompactionWatermarkForPrefixMock,
@@ -407,7 +408,7 @@ mock.module("../daemon/date-context.js", () => ({
407
408
  resolveTurnTimezoneContext: resolveTurnTimezoneContextMock,
408
409
  }));
409
410
 
410
- mock.module("../daemon/history-repair.js", () => ({
411
+ mock.module("../plugins/defaults/history-repair/terminal.js", () => ({
411
412
  repairHistory: (msgs: Message[]) => ({
412
413
  messages: msgs,
413
414
  stats: {
@@ -537,56 +538,78 @@ mock.module("../proactive-artifact/index.js", () => ({
537
538
 
538
539
  // ── Imports (after mocks) ────────────────────────────────────────────
539
540
 
541
+ import { AgentLoop } from "../agent/loop.js";
540
542
  import {
541
543
  type AgentLoopConversationContext,
542
544
  applyCompactionResult,
543
545
  runAgentLoopImpl,
544
546
  } from "../daemon/conversation-agent-loop.js";
547
+ import {
548
+ createMockProvider,
549
+ type ScriptedResponse,
550
+ textResponse,
551
+ toolUseResponse,
552
+ } from "./helpers/mock-provider.js";
545
553
 
546
554
  // ── Test helpers ─────────────────────────────────────────────────────
547
555
 
548
- type AgentLoopRun = (
549
- messages: Message[],
550
- onEvent: (event: AgentEvent) => void | Promise<void>,
551
- signal?: AbortSignal,
552
- requestId?: string,
553
- onCheckpoint?: (
554
- checkpoint: CheckpointInfo,
555
- ) => CheckpointDecision | Promise<CheckpointDecision>,
556
- ) => Promise<Message[]>;
557
-
558
556
  function makeCtx(
559
557
  overrides?: Partial<AgentLoopConversationContext> & {
560
- agentLoopRun?: AgentLoopRun;
558
+ providerResponses?: ScriptedResponse[];
559
+ loopProvider?: Provider;
560
+ loopTools?: ToolDefinition[];
561
+ toolExecutor?: LoopToolExecutor;
561
562
  },
562
563
  ): AgentLoopConversationContext {
563
- const agentLoopRun =
564
- overrides?.agentLoopRun ??
565
- (async (messages: Message[]) => [
566
- ...messages,
567
- {
568
- role: "assistant" as const,
569
- content: [{ type: "text" as const, text: "response" }],
570
- },
571
- ]);
564
+ const {
565
+ providerResponses,
566
+ loopProvider,
567
+ loopTools,
568
+ toolExecutor,
569
+ ...ctxOverrides
570
+ } = overrides ?? {};
571
+ const conversationId = ctxOverrides.conversationId ?? "test-conv";
572
+ let processing = true;
573
+
574
+ // Drive the real `AgentLoop` against a scripted provider, mocking only the
575
+ // provider HTTP boundary. The loop owns its mid-loop budget gate, inline
576
+ // compaction, and event emission, so these orchestrator tests exercise the
577
+ // real escalation/persistence path.
578
+ //
579
+ // Name the loop's provider after `ctx.provider` so the two stay in sync,
580
+ // mirroring production where the orchestrator hands the same provider to
581
+ // the loop. The loop stamps this name onto `usage.actualProvider` whenever
582
+ // a response omits its own, which is what the request-log fallback reads.
583
+ // Tests that need to introspect provider calls (or sequence a rejection)
584
+ // build their own `loopProvider` via `createMockProvider`.
585
+ const loopProviderName =
586
+ (ctxOverrides.provider as { name?: string } | undefined)?.name ??
587
+ "mock-provider";
588
+ const provider =
589
+ loopProvider ??
590
+ createMockProvider(
591
+ providerResponses ?? [textResponse("response")],
592
+ loopProviderName,
593
+ ).provider;
594
+ const agentLoop = new AgentLoop(provider, "system prompt", {
595
+ conversationId,
596
+ tools: loopTools ?? [],
597
+ toolExecutor,
598
+ });
572
599
 
573
600
  return {
574
601
  conversationId: "test-conv",
575
602
  messages: [
576
603
  { role: "user", content: [{ type: "text", text: "Hello" }] },
577
604
  ] as Message[],
578
- processing: true,
605
+ isProcessing: () => processing,
606
+ setProcessing: (value: boolean) => {
607
+ processing = value;
608
+ },
579
609
  abortController: new AbortController(),
580
610
  currentRequestId: "test-req",
581
611
 
582
- agentLoop: {
583
- run: agentLoopRun,
584
- getToolTokenBudget: () => 0,
585
- getResolvedTools: () => [],
586
- // Tests here don't exercise calibration; returning undefined makes
587
- // the estimator use the per-provider aggregate key.
588
- getActiveModel: () => undefined,
589
- } as unknown as AgentLoopConversationContext["agentLoop"],
612
+ agentLoop,
590
613
  provider: {
591
614
  name: "mock-provider",
592
615
  sendMessage: async () => ({
@@ -615,8 +638,6 @@ function makeCtx(
615
638
  currentTurnSurfaces: [],
616
639
 
617
640
  workingDir: "/tmp",
618
- workspaceTopLevelContext: null,
619
- workspaceTopLevelDirty: false,
620
641
  channelCapabilities: undefined,
621
642
  commandIntent: undefined,
622
643
  trustContext: undefined,
@@ -653,7 +674,6 @@ function makeCtx(
653
674
  getWorkspaceGitService: () => ({ ensureInitialized: async () => {} }),
654
675
  commitTurnChanges: async () => {},
655
676
 
656
- refreshWorkspaceTopLevelContextIfNeeded: () => {},
657
677
  markWorkspaceTopLevelDirty: () => {},
658
678
  emitActivityState: () => {},
659
679
  getQueueDepth: () => 0,
@@ -679,9 +699,10 @@ function makeCtx(
679
699
  injectedTokens: 0,
680
700
  }),
681
701
  retrackCachedNodes: () => {},
702
+ recordPkbQueryVectors: () => {},
682
703
  } as unknown as AgentLoopConversationContext["graphMemory"],
683
704
 
684
- ...overrides,
705
+ ...ctxOverrides,
685
706
  } as AgentLoopConversationContext;
686
707
  }
687
708
 
@@ -722,6 +743,9 @@ beforeEach(() => {
722
743
  setConversationHistoryStrippedAtMock.mockClear();
723
744
  setConversationHistoryStrippedAtMock.mockImplementation(() => {});
724
745
  applyRuntimeInjectionsMock.mockClear();
746
+ applyRuntimeInjectionsMock.mockImplementation(
747
+ defaultApplyRuntimeInjectionsImpl,
748
+ );
725
749
  buildUnifiedTurnContextBlockMock.mockClear();
726
750
  resolveTurnTimezoneContextMock.mockClear();
727
751
  formatTurnTimestampMock.mockClear();
@@ -735,11 +759,10 @@ beforeEach(() => {
735
759
  projectAssistantMessageMock.mockClear();
736
760
  publishSyncInvalidationMock.mockClear();
737
761
  mockMessageById = null;
738
- // Orchestrator pipelines (overflowReduce, persistence, …) run through the
739
- // plugin registry; reset and re-register every default so the pipelines
740
- // dispatch to middleware backed by the mocked collaborators these tests
741
- // install (`reduceContextOverflow`, `syncMessageToDisk`, etc.) instead of
742
- // hitting the bare terminals.
762
+ // The compaction pipeline runs through the plugin registry; reset and
763
+ // re-register every default so it dispatches to middleware backed by the
764
+ // mocked collaborators these tests install (`syncMessageToDisk`, etc.)
765
+ // instead of hitting the bare terminal.
743
766
  resetPluginRegistryAndRegisterDefaults();
744
767
  });
745
768
 
@@ -805,7 +828,7 @@ describe("session-agent-loop", () => {
805
828
  });
806
829
 
807
830
  describe("proactive artifact trigger", () => {
808
- test("suppresses proactive app build when the foreground turn used app tools", async () => {
831
+ test("does not start proactive artifact jobs after foreground user turns", async () => {
809
832
  mockConversationRow = {
810
833
  ...mockConversationRow,
811
834
  id: "test-conv",
@@ -819,63 +842,28 @@ describe("session-agent-loop", () => {
819
842
  mockHasProactiveArtifactCompleted = false;
820
843
  mockTryClaimProactiveArtifactTrigger = true;
821
844
 
822
- const agentLoopRun: AgentLoopRun = async (
823
- messages,
824
- onEvent,
825
- _signal,
826
- _requestId,
827
- onCheckpoint,
828
- ) => {
829
- // Prime the assistant row anchor for LLM call 1 — production code
830
- // emits this from `AgentLoop.run` just before `provider.sendMessage`.
831
- await onEvent({ type: "llm_call_started" });
832
- await onEvent({
833
- type: "message_complete",
834
- message: {
835
- role: "assistant",
836
- content: [{ type: "text", text: "I'll build that app." }],
837
- },
838
- });
839
- await onEvent({
840
- type: "tool_use",
841
- id: "tool-1",
842
- name: "app_create",
843
- input: { name: "Flow" },
844
- });
845
- await onEvent({
846
- type: "tool_result",
847
- toolUseId: "tool-1",
848
- content: "{}",
849
- isError: false,
850
- });
851
- await onCheckpoint?.({
852
- turnIndex: 0,
853
- toolCount: 1,
854
- hasToolUse: true,
855
- history: messages,
856
- });
857
- // Prime the anchor again for LLM call 2 — multi-call agent turns
858
- // reserve a fresh assistant row per LLM call.
859
- await onEvent({ type: "llm_call_started" });
860
- await onEvent({
861
- type: "message_complete",
862
- message: {
863
- role: "assistant",
864
- content: [{ type: "text", text: "Done." }],
865
- },
866
- });
867
- return [
868
- ...messages,
869
- {
870
- role: "assistant" as const,
871
- content: [{ type: "text" as const, text: "Done." }],
872
- },
873
- ];
874
- };
875
-
845
+ // A two-call agent turn: the model invokes `app_create`, then wraps up
846
+ // with a final text reply.
876
847
  const ctx = makeCtx({
877
848
  conversationId: "test-conv",
878
- agentLoopRun,
849
+ providerResponses: [
850
+ {
851
+ content: [
852
+ { type: "text", text: "I'll build that app." },
853
+ {
854
+ type: "tool_use",
855
+ id: "tool-1",
856
+ name: "app_create",
857
+ input: { name: "Flow" },
858
+ },
859
+ ],
860
+ model: "mock-model",
861
+ usage: { inputTokens: 10, outputTokens: 5 },
862
+ stopReason: "tool_use",
863
+ },
864
+ textResponse("Done."),
865
+ ],
866
+ toolExecutor: async () => ({ content: "{}", isError: false }),
879
867
  });
880
868
  await runAgentLoopImpl(
881
869
  ctx,
@@ -888,16 +876,28 @@ describe("session-agent-loop", () => {
888
876
  );
889
877
  await new Promise((resolve) => setTimeout(resolve, 0));
890
878
 
891
- expect(runProactiveArtifactJobMock).toHaveBeenCalledTimes(1);
892
- expect(runProactiveArtifactJobMock.mock.calls[0]?.[0]).toMatchObject({
893
- conversationId: "test-conv",
894
- suppressAppBuild: true,
895
- });
879
+ expect(runProactiveArtifactJobMock).toHaveBeenCalledTimes(0);
896
880
  });
897
881
  });
898
882
 
899
883
  describe("disk pressure injection context", () => {
900
- test("passes cleanup context into runtime injections for cleanup-mode turns", async () => {
884
+ // The loop sets `ctx.diskPressureCleanupModeActive` for the duration of the
885
+ // turn (the disk-pressure-warning injector reads it via the per-conversation
886
+ // registry) and resets it in the turn-end cleanup path. Snapshot the flag at
887
+ // each `applyRuntimeInjections` call so assertions observe its value while
888
+ // injection runs, not the post-turn reset.
889
+ function captureCleanupFlagDuringInjection(ctx: {
890
+ diskPressureCleanupModeActive?: boolean;
891
+ }): () => Array<boolean | undefined> {
892
+ const observed: Array<boolean | undefined> = [];
893
+ applyRuntimeInjectionsMock.mockImplementation(async (msgs: Message[]) => {
894
+ observed.push(ctx.diskPressureCleanupModeActive);
895
+ return { messages: msgs, blocks: { ...mockInjectionBlocks } };
896
+ });
897
+ return () => observed;
898
+ }
899
+
900
+ test("sets the cleanup-mode flag on the conversation for cleanup-mode turns", async () => {
901
901
  mockDiskPressureDecision = {
902
902
  action: "allow-cleanup-mode",
903
903
  reason: "guardian",
@@ -920,6 +920,7 @@ describe("session-agent-loop", () => {
920
920
  trustClass: "guardian",
921
921
  } as AgentLoopConversationContext["trustContext"],
922
922
  });
923
+ const cleanupFlagDuringInjection = captureCleanupFlagDuringInjection(ctx);
923
924
 
924
925
  await runAgentLoopImpl(ctx, "free up space", "msg-1", () => {});
925
926
 
@@ -938,21 +939,16 @@ describe("session-agent-loop", () => {
938
939
  },
939
940
  }),
940
941
  );
941
- const firstInjectionOptions = applyRuntimeInjectionsMock.mock
942
- .calls[0]![1] as {
943
- diskPressureContext?: { cleanupModeActive: boolean } | null;
944
- };
945
- expect(firstInjectionOptions.diskPressureContext).toEqual({
946
- cleanupModeActive: true,
947
- });
942
+ expect(cleanupFlagDuringInjection()).toEqual([true]);
948
943
  });
949
944
 
950
- test("passes cleanup context into runtime injections for local-owner turns", async () => {
945
+ test("sets the cleanup-mode flag on the conversation for local-owner turns", async () => {
951
946
  mockDiskPressureDecision = {
952
947
  action: "allow-cleanup-mode",
953
948
  reason: "local-owner",
954
949
  };
955
950
  const ctx = makeCtx();
951
+ const cleanupFlagDuringInjection = captureCleanupFlagDuringInjection(ctx);
956
952
 
957
953
  await runAgentLoopImpl(ctx, "free up space", "msg-1", () => {});
958
954
 
@@ -964,16 +960,10 @@ describe("session-agent-loop", () => {
964
960
  trustContext: null,
965
961
  }),
966
962
  );
967
- const firstInjectionOptions = applyRuntimeInjectionsMock.mock
968
- .calls[0]![1] as {
969
- diskPressureContext?: { cleanupModeActive: boolean } | null;
970
- };
971
- expect(firstInjectionOptions.diskPressureContext).toEqual({
972
- cleanupModeActive: true,
973
- });
963
+ expect(cleanupFlagDuringInjection()).toEqual([true]);
974
964
  });
975
965
 
976
- test("keeps cleanup context on overflow recovery reinjection", async () => {
966
+ test("keeps the cleanup-mode flag set across overflow recovery reinjection", async () => {
977
967
  mockDiskPressureDecision = {
978
968
  action: "allow-cleanup-mode",
979
969
  reason: "guardian",
@@ -995,18 +985,14 @@ describe("session-agent-loop", () => {
995
985
  trustClass: "guardian",
996
986
  } as AgentLoopConversationContext["trustContext"],
997
987
  });
988
+ const cleanupFlagDuringInjection = captureCleanupFlagDuringInjection(ctx);
998
989
 
999
990
  await runAgentLoopImpl(ctx, "free up space", "msg-1", () => {});
1000
991
 
1001
992
  expect(applyRuntimeInjectionsMock.mock.calls.length).toBeGreaterThan(1);
1002
- for (const call of applyRuntimeInjectionsMock.mock.calls) {
1003
- const options = call[1] as {
1004
- diskPressureContext?: { cleanupModeActive: boolean } | null;
1005
- };
1006
- expect(options.diskPressureContext).toEqual({
1007
- cleanupModeActive: true,
1008
- });
1009
- }
993
+ const flags = cleanupFlagDuringInjection();
994
+ expect(flags.length).toBeGreaterThan(1);
995
+ expect(flags.every((flag) => flag === true)).toBe(true);
1010
996
  });
1011
997
 
1012
998
  test("blocks policy-denied turns before runtime injection or model execution", async () => {
@@ -1015,9 +1001,6 @@ describe("session-agent-loop", () => {
1015
1001
  reason: "trusted-contact",
1016
1002
  };
1017
1003
  const events: ServerMessage[] = [];
1018
- const agentLoopRun = mock(async (_messages: Message[]) => {
1019
- throw new Error("agent loop should not run");
1020
- });
1021
1004
  const activityStates: unknown[][] = [];
1022
1005
  const traceEvents: unknown[][] = [];
1023
1006
  const ctx = makeCtx({
@@ -1030,17 +1013,16 @@ describe("session-agent-loop", () => {
1030
1013
  },
1031
1014
  } as unknown as AgentLoopConversationContext["traceEmitter"],
1032
1015
  });
1033
- ctx.agentLoop.run = agentLoopRun as AgentLoopRun;
1016
+ const runSpy = spyOn(ctx.agentLoop, "run");
1034
1017
 
1035
1018
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
1036
1019
 
1037
- expect(agentLoopRun).not.toHaveBeenCalled();
1020
+ expect(runSpy).not.toHaveBeenCalled();
1038
1021
  expect(applyRuntimeInjectionsMock).not.toHaveBeenCalled();
1039
1022
  expect(activityStates).toContainEqual([
1040
1023
  "idle",
1041
1024
  "error_terminal",
1042
- "global",
1043
- "test-req",
1025
+ { anchor: "global", requestId: "test-req" },
1044
1026
  ]);
1045
1027
  expect(traceEvents[0]).toEqual([
1046
1028
  "request_error",
@@ -1095,15 +1077,14 @@ describe("session-agent-loop", () => {
1095
1077
  });
1096
1078
 
1097
1079
  expect(applyRuntimeInjectionsMock).not.toHaveBeenCalled();
1098
- expect(ctx.processing).toBe(false);
1080
+ expect(ctx.isProcessing()).toBe(false);
1099
1081
  expect(ctx.abortController).toBeNull();
1100
1082
  expect(ctx.currentRequestId).toBeUndefined();
1101
1083
  expect(drainQueue).toHaveBeenCalledWith("loop_complete");
1102
1084
  expect(activityStates).toContainEqual([
1103
1085
  "idle",
1104
1086
  "error_terminal",
1105
- "global",
1106
- "test-req",
1087
+ { anchor: "global", requestId: "test-req" },
1107
1088
  ]);
1108
1089
  });
1109
1090
  });
@@ -1112,47 +1093,14 @@ describe("session-agent-loop", () => {
1112
1093
  test("error events from agent loop are classified and emitted", async () => {
1113
1094
  const events: ServerMessage[] = [];
1114
1095
 
1115
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
1116
- // Prime the assistant row anchor — production code emits this from
1117
- // `AgentLoop.run` just before `provider.sendMessage`.
1118
- await onEvent({ type: "llm_call_started" });
1119
- // Simulate tool_use + error during execution
1120
- onEvent({
1121
- type: "tool_use",
1122
- id: "tu-1",
1123
- name: "bash",
1124
- input: { cmd: "ls" },
1125
- });
1126
- onEvent({
1127
- type: "error",
1128
- error: new Error("Tool execution failed: permission denied"),
1129
- });
1130
- onEvent({
1131
- type: "message_complete",
1132
- message: {
1133
- role: "assistant",
1134
- content: [{ type: "text", text: "I encountered an error" }],
1135
- },
1136
- });
1137
- onEvent({
1138
- type: "usage",
1139
- inputTokens: 100,
1140
- outputTokens: 50,
1141
- model: "test-model",
1142
- providerDurationMs: 200,
1143
- });
1144
- return [
1145
- ...messages,
1146
- {
1147
- role: "assistant" as const,
1148
- content: [
1149
- { type: "text", text: "I encountered an error" },
1150
- ] as ContentBlock[],
1151
- },
1152
- ];
1153
- };
1154
-
1155
- const ctx = makeCtx({ agentLoopRun });
1096
+ // The model calls a tool whose executor throws, surfacing an `error`
1097
+ // event from the loop's catch handler.
1098
+ const ctx = makeCtx({
1099
+ providerResponses: [toolUseResponse("tu-1", "bash", { cmd: "ls" })],
1100
+ toolExecutor: async () => {
1101
+ throw new Error("Tool execution failed: permission denied");
1102
+ },
1103
+ });
1156
1104
  await runAgentLoopImpl(ctx, "run ls", "msg-1", (msg) => events.push(msg));
1157
1105
 
1158
1106
  const conversationError = events.find(
@@ -1164,34 +1112,9 @@ describe("session-agent-loop", () => {
1164
1112
  test("non-error agent loop completion does not emit conversation_error", async () => {
1165
1113
  const events: ServerMessage[] = [];
1166
1114
 
1167
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
1168
- // Prime the assistant row anchor — production code emits this from
1169
- // `AgentLoop.run` just before `provider.sendMessage`.
1170
- await onEvent({ type: "llm_call_started" });
1171
- onEvent({
1172
- type: "message_complete",
1173
- message: {
1174
- role: "assistant",
1175
- content: [{ type: "text", text: "All good" }],
1176
- },
1177
- });
1178
- onEvent({
1179
- type: "usage",
1180
- inputTokens: 50,
1181
- outputTokens: 25,
1182
- model: "test-model",
1183
- providerDurationMs: 100,
1184
- });
1185
- return [
1186
- ...messages,
1187
- {
1188
- role: "assistant" as const,
1189
- content: [{ type: "text", text: "All good" }] as ContentBlock[],
1190
- },
1191
- ];
1192
- };
1193
-
1194
- const ctx = makeCtx({ agentLoopRun });
1115
+ const ctx = makeCtx({
1116
+ providerResponses: [textResponse("All good")],
1117
+ });
1195
1118
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
1196
1119
 
1197
1120
  const conversationError = events.find(
@@ -1227,38 +1150,20 @@ describe("session-agent-loop", () => {
1227
1150
  },
1228
1151
  };
1229
1152
 
1230
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
1231
- // Prime the assistant row anchor production code emits this from
1232
- // `AgentLoop.run` just before `provider.sendMessage`.
1233
- await onEvent({ type: "llm_call_started" });
1234
- onEvent({
1235
- type: "message_complete",
1236
- message: {
1237
- role: "assistant",
1238
- content: [{ type: "text", text: "Hi there." }],
1239
- },
1240
- });
1241
- onEvent({
1242
- type: "usage",
1243
- inputTokens: 12,
1244
- outputTokens: 3,
1245
- model: "gpt-4.1-2026-03-01",
1246
- actualProvider: "fireworks",
1247
- providerDurationMs: 45,
1248
- rawRequest,
1249
- rawResponse,
1250
- });
1251
- return [
1252
- ...messages,
1153
+ // The provider response carries its own `actualProvider`, so the logged
1154
+ // row should record that name rather than the runtime provider.
1155
+ const ctx = makeCtx({
1156
+ providerResponses: [
1253
1157
  {
1254
- role: "assistant" as const,
1255
- content: [{ type: "text", text: "Hi there." }] as ContentBlock[],
1158
+ content: [{ type: "text", text: "Hi there." }],
1159
+ model: "gpt-4.1-2026-03-01",
1160
+ usage: { inputTokens: 12, outputTokens: 3 },
1161
+ stopReason: "end_turn",
1162
+ actualProvider: "fireworks",
1163
+ rawRequest,
1164
+ rawResponse,
1256
1165
  },
1257
- ];
1258
- };
1259
-
1260
- const ctx = makeCtx({
1261
- agentLoopRun,
1166
+ ],
1262
1167
  provider: {
1263
1168
  name: "openrouter",
1264
1169
  sendMessage: async () => ({
@@ -1295,37 +1200,19 @@ describe("session-agent-loop", () => {
1295
1200
  ],
1296
1201
  };
1297
1202
 
1298
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
1299
- // Prime the assistant row anchor production code emits this from
1300
- // `AgentLoop.run` just before `provider.sendMessage`.
1301
- await onEvent({ type: "llm_call_started" });
1302
- onEvent({
1303
- type: "message_complete",
1304
- message: {
1305
- role: "assistant",
1306
- content: [{ type: "text", text: "Hi there." }],
1307
- },
1308
- });
1309
- onEvent({
1310
- type: "usage",
1311
- inputTokens: 12,
1312
- outputTokens: 3,
1313
- model: "gpt-4.1-2026-03-01",
1314
- providerDurationMs: 45,
1315
- rawRequest,
1316
- rawResponse,
1317
- });
1318
- return [
1319
- ...messages,
1203
+ // The provider response omits `actualProvider`, so the loop stamps the
1204
+ // runtime provider name onto the usage event and the row records it.
1205
+ const ctx = makeCtx({
1206
+ providerResponses: [
1320
1207
  {
1321
- role: "assistant" as const,
1322
- content: [{ type: "text", text: "Hi there." }] as ContentBlock[],
1208
+ content: [{ type: "text", text: "Hi there." }],
1209
+ model: "gpt-4.1-2026-03-01",
1210
+ usage: { inputTokens: 12, outputTokens: 3 },
1211
+ stopReason: "end_turn",
1212
+ rawRequest,
1213
+ rawResponse,
1323
1214
  },
1324
- ];
1325
- };
1326
-
1327
- const ctx = makeCtx({
1328
- agentLoopRun,
1215
+ ],
1329
1216
  provider: {
1330
1217
  name: "openrouter",
1331
1218
  sendMessage: async () => ({
@@ -1380,38 +1267,18 @@ describe("session-agent-loop", () => {
1380
1267
  status: "completed",
1381
1268
  };
1382
1269
 
1383
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
1384
- // Prime the assistant row anchor — production code emits this from
1385
- // `AgentLoop.run` just before `provider.sendMessage`.
1386
- await onEvent({ type: "llm_call_started" });
1387
- onEvent({
1388
- type: "message_complete",
1389
- message: {
1390
- role: "assistant",
1391
- content: [{ type: "text", text: "Hi there." }],
1392
- },
1393
- });
1394
- onEvent({
1395
- type: "usage",
1396
- inputTokens: 12,
1397
- outputTokens: 3,
1398
- model: "gpt-5.4",
1399
- actualProvider: "openai",
1400
- providerDurationMs: 45,
1401
- rawRequest,
1402
- rawResponse,
1403
- });
1404
- return [
1405
- ...messages,
1270
+ const ctx = makeCtx({
1271
+ providerResponses: [
1406
1272
  {
1407
- role: "assistant" as const,
1408
- content: [{ type: "text", text: "Hi there." }] as ContentBlock[],
1273
+ content: [{ type: "text", text: "Hi there." }],
1274
+ model: "gpt-5.4",
1275
+ usage: { inputTokens: 12, outputTokens: 3 },
1276
+ stopReason: "end_turn",
1277
+ actualProvider: "openai",
1278
+ rawRequest,
1279
+ rawResponse,
1409
1280
  },
1410
- ];
1411
- };
1412
-
1413
- const ctx = makeCtx({
1414
- agentLoopRun,
1281
+ ],
1415
1282
  provider: {
1416
1283
  name: "openai",
1417
1284
  sendMessage: async () => ({
@@ -1451,37 +1318,17 @@ describe("session-agent-loop", () => {
1451
1318
  attrs: Record<string, unknown>;
1452
1319
  }> = [];
1453
1320
 
1454
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
1455
- // Prime the assistant row anchor production code emits this from
1456
- // `AgentLoop.run` just before `provider.sendMessage`.
1457
- await onEvent({ type: "llm_call_started" });
1458
- onEvent({ type: "text_delta", text: "Hi." });
1459
- onEvent({
1460
- type: "message_complete",
1461
- message: {
1462
- role: "assistant",
1463
- content: [{ type: "text", text: "Hi." }],
1464
- },
1465
- });
1466
- onEvent({
1467
- type: "usage",
1468
- inputTokens: 10,
1469
- outputTokens: 2,
1470
- model: "gpt-5.5-2026-04-23",
1471
- actualProvider: "openai",
1472
- providerDurationMs: 100,
1473
- });
1474
- return [
1475
- ...messages,
1321
+ const ctx = makeCtx({
1322
+ // The loop replays the text block as a `text_delta` before `usage`.
1323
+ providerResponses: [
1476
1324
  {
1477
- role: "assistant" as const,
1478
- content: [{ type: "text", text: "Hi." }] as ContentBlock[],
1325
+ content: [{ type: "text", text: "Hi." }],
1326
+ model: "gpt-5.5-2026-04-23",
1327
+ usage: { inputTokens: 10, outputTokens: 2 },
1328
+ stopReason: "end_turn",
1329
+ actualProvider: "openai",
1479
1330
  },
1480
- ];
1481
- };
1482
-
1483
- const ctx = makeCtx({
1484
- agentLoopRun,
1331
+ ],
1485
1332
  // Provider name matches actualProvider so both paths agree.
1486
1333
  provider: {
1487
1334
  name: "openai",
@@ -1529,31 +1376,18 @@ describe("session-agent-loop", () => {
1529
1376
  attrs: Record<string, unknown>;
1530
1377
  }> = [];
1531
1378
 
1532
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
1533
- // Prime the assistant row anchor production code emits this from
1534
- // `AgentLoop.run` just before `provider.sendMessage`.
1535
- await onEvent({ type: "llm_call_started" });
1536
- // No text_delta — pure tool-call response
1537
- onEvent({
1538
- type: "message_complete",
1539
- message: {
1540
- role: "assistant",
1379
+ const ctx = makeCtx({
1380
+ // An empty-content response: no text block fires `text_delta`, so the
1381
+ // started event falls back to the resolved usage provider name.
1382
+ providerResponses: [
1383
+ {
1541
1384
  content: [],
1385
+ model: "gpt-5.5-2026-04-23",
1386
+ usage: { inputTokens: 10, outputTokens: 2 },
1387
+ stopReason: "end_turn",
1388
+ actualProvider: "openai",
1542
1389
  },
1543
- });
1544
- onEvent({
1545
- type: "usage",
1546
- inputTokens: 10,
1547
- outputTokens: 2,
1548
- model: "gpt-5.5-2026-04-23",
1549
- actualProvider: "openai",
1550
- providerDurationMs: 100,
1551
- });
1552
- return messages;
1553
- };
1554
-
1555
- const ctx = makeCtx({
1556
- agentLoopRun,
1390
+ ],
1557
1391
  provider: {
1558
1392
  name: "anthropic",
1559
1393
  sendMessage: async () => ({
@@ -1595,52 +1429,32 @@ describe("session-agent-loop", () => {
1595
1429
  test("records the actual provider for usage accounting", async () => {
1596
1430
  const events: ServerMessage[] = [];
1597
1431
 
1598
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
1599
- // Prime the assistant row anchor — production code emits this from
1600
- // `AgentLoop.run` just before `provider.sendMessage`.
1601
- await onEvent({ type: "llm_call_started" });
1602
- onEvent({
1603
- type: "message_complete",
1604
- message: {
1605
- role: "assistant",
1432
+ const ctx = makeCtx({
1433
+ providerResponses: [
1434
+ {
1606
1435
  content: [{ type: "text", text: "Hi there." }],
1607
- },
1608
- });
1609
- onEvent({
1610
- type: "usage",
1611
- inputTokens: 12,
1612
- outputTokens: 3,
1613
- model: "gpt-4.1-2026-03-01",
1614
- actualProvider: "fireworks",
1615
- providerDurationMs: 45,
1616
- rawRequest: {
1617
- model: "gpt-4.1",
1618
- messages: [{ role: "user", content: "Hello" }],
1619
- },
1620
- rawResponse: {
1621
1436
  model: "gpt-4.1-2026-03-01",
1622
- choices: [
1623
- {
1624
- finish_reason: "stop",
1625
- message: {
1626
- role: "assistant",
1627
- content: "Hi there.",
1437
+ usage: { inputTokens: 12, outputTokens: 3 },
1438
+ stopReason: "end_turn",
1439
+ actualProvider: "fireworks",
1440
+ rawRequest: {
1441
+ model: "gpt-4.1",
1442
+ messages: [{ role: "user", content: "Hello" }],
1443
+ },
1444
+ rawResponse: {
1445
+ model: "gpt-4.1-2026-03-01",
1446
+ choices: [
1447
+ {
1448
+ finish_reason: "stop",
1449
+ message: {
1450
+ role: "assistant",
1451
+ content: "Hi there.",
1452
+ },
1628
1453
  },
1629
- },
1630
- ],
1631
- },
1632
- });
1633
- return [
1634
- ...messages,
1635
- {
1636
- role: "assistant" as const,
1637
- content: [{ type: "text", text: "Hi there." }] as ContentBlock[],
1454
+ ],
1455
+ },
1638
1456
  },
1639
- ];
1640
- };
1641
-
1642
- const ctx = makeCtx({
1643
- agentLoopRun,
1457
+ ],
1644
1458
  provider: {
1645
1459
  name: "openrouter",
1646
1460
  sendMessage: async () => ({
@@ -1710,27 +1524,9 @@ describe("session-agent-loop", () => {
1710
1524
  },
1711
1525
  });
1712
1526
 
1713
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
1714
- // Prime the assistant row anchor — production code emits this from
1715
- // `AgentLoop.run` just before `provider.sendMessage`.
1716
- await onEvent({ type: "llm_call_started" });
1717
- onEvent({
1718
- type: "message_complete",
1719
- message: {
1720
- role: "assistant",
1721
- content: [{ type: "text", text: "recovered" }],
1722
- },
1723
- });
1724
- return [
1725
- ...messages,
1726
- {
1727
- role: "assistant" as const,
1728
- content: [{ type: "text", text: "recovered" }] as ContentBlock[],
1729
- },
1730
- ];
1731
- };
1732
-
1733
- const ctx = makeCtx({ agentLoopRun });
1527
+ // After the orchestrator's preflight compaction runs, the loop completes
1528
+ // the turn normally.
1529
+ const ctx = makeCtx({ providerResponses: [textResponse("recovered")] });
1734
1530
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
1735
1531
 
1736
1532
  const compactorCall = recordUsageMock.mock.calls.find(
@@ -1769,7 +1565,6 @@ describe("session-agent-loop", () => {
1769
1565
 
1770
1566
  test("convergence loop applies reducer and retries when context-too-large is detected", async () => {
1771
1567
  const events: ServerMessage[] = [];
1772
- let callCount = 0;
1773
1568
  let reducerCalled = false;
1774
1569
 
1775
1570
  // Configure reducer to succeed on first call — return reduced messages
@@ -1803,53 +1598,15 @@ describe("session-agent-loop", () => {
1803
1598
  };
1804
1599
  };
1805
1600
 
1806
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
1807
- // Prime the assistant row anchor production code emits this from
1808
- // `AgentLoop.run` just before `provider.sendMessage`. Retry branches
1809
- // need this on every invocation: each agent-loop iteration reserves
1810
- // its own row.
1811
- await onEvent({ type: "llm_call_started" });
1812
- callCount++;
1813
- if (callCount === 1) {
1814
- onEvent({
1815
- type: "error",
1816
- error: new Error("context_length_exceeded"),
1817
- });
1818
- onEvent({
1819
- type: "usage",
1820
- inputTokens: 100,
1821
- outputTokens: 0,
1822
- model: "test-model",
1823
- providerDurationMs: 50,
1824
- });
1825
- return messages;
1826
- }
1827
- // Second call (after reducer): succeed
1828
- onEvent({
1829
- type: "message_complete",
1830
- message: {
1831
- role: "assistant",
1832
- content: [{ type: "text", text: "recovered" }],
1833
- },
1834
- });
1835
- onEvent({
1836
- type: "usage",
1837
- inputTokens: 50,
1838
- outputTokens: 25,
1839
- model: "test-model",
1840
- providerDurationMs: 100,
1841
- });
1842
- return [
1843
- ...messages,
1844
- {
1845
- role: "assistant" as const,
1846
- content: [{ type: "text", text: "recovered" }] as ContentBlock[],
1847
- },
1848
- ];
1849
- };
1601
+ // The provider rejects the first call with a context-too-large error,
1602
+ // then succeeds once the orchestrator has reduced the context.
1603
+ const { provider, calls } = createMockProvider([
1604
+ new Error("context_length_exceeded"),
1605
+ textResponse("recovered"),
1606
+ ]);
1850
1607
 
1851
1608
  const ctx = makeCtx({
1852
- agentLoopRun,
1609
+ loopProvider: provider,
1853
1610
  contextWindowManager: {
1854
1611
  shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
1855
1612
  maybeCompact: async () => ({ compacted: false }),
@@ -1859,7 +1616,7 @@ describe("session-agent-loop", () => {
1859
1616
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
1860
1617
 
1861
1618
  expect(reducerCalled).toBe(true);
1862
- expect(callCount).toBe(2);
1619
+ expect(calls.length).toBe(2);
1863
1620
  const compactEvent = events.find((e) => e.type === "context_compacted");
1864
1621
  expect(compactEvent).toBeDefined();
1865
1622
  });
@@ -1867,23 +1624,10 @@ describe("session-agent-loop", () => {
1867
1624
  test("emits conversation_error when context stays too large after all recovery attempts", async () => {
1868
1625
  const events: ServerMessage[] = [];
1869
1626
 
1870
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
1871
- onEvent({
1872
- type: "error",
1873
- error: new Error("context_length_exceeded"),
1874
- });
1875
- onEvent({
1876
- type: "usage",
1877
- inputTokens: 100,
1878
- outputTokens: 0,
1879
- model: "test-model",
1880
- providerDurationMs: 50,
1881
- });
1882
- return messages;
1883
- };
1884
-
1627
+ // The provider rejects every call with a context-too-large error, so the
1628
+ // orchestrator exhausts its recovery attempts.
1885
1629
  const ctx = makeCtx({
1886
- agentLoopRun,
1630
+ providerResponses: [new Error("context_length_exceeded")],
1887
1631
  contextWindowManager: {
1888
1632
  shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
1889
1633
  // Compaction succeeds but context is still too large
@@ -1917,7 +1661,6 @@ describe("session-agent-loop", () => {
1917
1661
 
1918
1662
  test("bounded convergence loop applies reducer tiers and recovers", async () => {
1919
1663
  const events: ServerMessage[] = [];
1920
- let callCount = 0;
1921
1664
  let reducerCalls = 0;
1922
1665
 
1923
1666
  // Reducer: succeed on first call, returning reduced messages
@@ -1935,55 +1678,15 @@ describe("session-agent-loop", () => {
1935
1678
  };
1936
1679
  };
1937
1680
 
1938
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
1939
- // Prime the assistant row anchor production code emits this from
1940
- // `AgentLoop.run` just before `provider.sendMessage`. Retry branches
1941
- // need this on every invocation: each agent-loop iteration reserves
1942
- // its own row.
1943
- await onEvent({ type: "llm_call_started" });
1944
- callCount++;
1945
- if (callCount === 1) {
1946
- onEvent({
1947
- type: "error",
1948
- error: new Error("context_length_exceeded"),
1949
- });
1950
- onEvent({
1951
- type: "usage",
1952
- inputTokens: 100,
1953
- outputTokens: 0,
1954
- model: "test-model",
1955
- providerDurationMs: 50,
1956
- });
1957
- return messages;
1958
- }
1959
- // After reducer runs, succeed
1960
- onEvent({
1961
- type: "message_complete",
1962
- message: {
1963
- role: "assistant",
1964
- content: [{ type: "text", text: "recovered via convergence" }],
1965
- },
1966
- });
1967
- onEvent({
1968
- type: "usage",
1969
- inputTokens: 50,
1970
- outputTokens: 25,
1971
- model: "test-model",
1972
- providerDurationMs: 100,
1973
- });
1974
- return [
1975
- ...messages,
1976
- {
1977
- role: "assistant" as const,
1978
- content: [
1979
- { type: "text", text: "recovered via convergence" },
1980
- ] as ContentBlock[],
1981
- },
1982
- ];
1983
- };
1681
+ // The provider rejects the first call with a context-too-large error,
1682
+ // then succeeds once the orchestrator has reduced the context.
1683
+ const { provider, calls } = createMockProvider([
1684
+ new Error("context_length_exceeded"),
1685
+ textResponse("recovered via convergence"),
1686
+ ]);
1984
1687
 
1985
1688
  const ctx = makeCtx({
1986
- agentLoopRun,
1689
+ loopProvider: provider,
1987
1690
  contextWindowManager: {
1988
1691
  shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
1989
1692
  maybeCompact: async () => ({ compacted: false }),
@@ -1993,7 +1696,7 @@ describe("session-agent-loop", () => {
1993
1696
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
1994
1697
 
1995
1698
  expect(reducerCalls).toBeGreaterThanOrEqual(1);
1996
- expect(callCount).toBe(2);
1699
+ expect(calls.length).toBe(2);
1997
1700
  const conversationError = events.find(
1998
1701
  (e) => e.type === "conversation_error",
1999
1702
  );
@@ -2004,7 +1707,6 @@ describe("session-agent-loop", () => {
2004
1707
 
2005
1708
  test("non-interactive auto-compress continues without approval prompt", async () => {
2006
1709
  const events: ServerMessage[] = [];
2007
- let callCount = 0;
2008
1710
 
2009
1711
  // Reducer exhausts all tiers
2010
1712
  mockReducerStepFn = (msgs: Message[]) => ({
@@ -2025,54 +1727,14 @@ describe("session-agent-loop", () => {
2025
1727
 
2026
1728
  mockOverflowAction = "auto_compress_latest_turn";
2027
1729
 
2028
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
2029
- // Prime the assistant row anchor production code emits this from
2030
- // `AgentLoop.run` just before `provider.sendMessage`. Retry branches
2031
- // need this on every invocation: each agent-loop iteration reserves
2032
- // its own row.
2033
- await onEvent({ type: "llm_call_started" });
2034
- callCount++;
2035
- if (callCount <= 2) {
2036
- onEvent({
2037
- type: "error",
2038
- error: new Error("context_length_exceeded"),
2039
- });
2040
- onEvent({
2041
- type: "usage",
2042
- inputTokens: 100,
2043
- outputTokens: 0,
2044
- model: "test-model",
2045
- providerDurationMs: 50,
2046
- });
2047
- return messages;
2048
- }
2049
- onEvent({
2050
- type: "message_complete",
2051
- message: {
2052
- role: "assistant",
2053
- content: [{ type: "text", text: "auto-recovered" }],
2054
- },
2055
- });
2056
- onEvent({
2057
- type: "usage",
2058
- inputTokens: 50,
2059
- outputTokens: 25,
2060
- model: "test-model",
2061
- providerDurationMs: 100,
2062
- });
2063
- return [
2064
- ...messages,
2065
- {
2066
- role: "assistant" as const,
2067
- content: [
2068
- { type: "text", text: "auto-recovered" },
2069
- ] as ContentBlock[],
2070
- },
2071
- ];
2072
- };
2073
-
1730
+ // The provider rejects the first two calls with context-too-large errors,
1731
+ // then succeeds after the emergency auto-compress runs.
2074
1732
  const ctx = makeCtx({
2075
- agentLoopRun,
1733
+ providerResponses: [
1734
+ new Error("context_length_exceeded"),
1735
+ new Error("context_length_exceeded"),
1736
+ textResponse("auto-recovered"),
1737
+ ],
2076
1738
  hasNoClient: true,
2077
1739
  contextWindowManager: {
2078
1740
  shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
@@ -2119,7 +1781,6 @@ describe("session-agent-loop", () => {
2119
1781
  // `budget_yield_unrecovered` so the inspector and dashboards can
2120
1782
  // attribute the silent stall.
2121
1783
  const events: ServerMessage[] = [];
2122
- let callCount = 0;
2123
1784
 
2124
1785
  // Reducer exhausts all 4 tiers on first call so the convergence
2125
1786
  // loop runs exactly one iteration before falling through to
@@ -2150,49 +1811,30 @@ describe("session-agent-loop", () => {
2150
1811
  // call). 90k satisfies both so the path reaches call 3.
2151
1812
  mockEstimateTokens = 90_000;
2152
1813
 
2153
- const agentLoopRun: AgentLoopRun = async (
2154
- messages,
2155
- onEvent,
2156
- _signal,
2157
- _reqId,
2158
- onCheckpoint,
2159
- ) => {
2160
- callCount++;
2161
- if (callCount <= 2) {
2162
- // Calls 1 (initial) and 2 (convergence rerun): error so
2163
- // `state.contextTooLargeDetected` stays true through
2164
- // convergence exit and we enter the auto_compress branch.
2165
- onEvent({
2166
- type: "error",
2167
- error: new Error("context_length_exceeded"),
2168
- });
2169
- onEvent({
2170
- type: "usage",
2171
- inputTokens: 100,
2172
- outputTokens: 0,
2173
- model: "test-model",
2174
- providerDurationMs: 50,
2175
- });
2176
- return messages;
2177
- }
2178
- // Call 3: the auto_compress_latest_turn rerun. Invoke
2179
- // onCheckpoint so the orchestrator's mid-loop budget check
2180
- // flips `yieldedForBudget` to true, then return without
2181
- // finishing — mirroring what AgentLoop.run does when its
2182
- // checkpoint returns "yield".
2183
- if (onCheckpoint) {
2184
- await onCheckpoint({
2185
- turnIndex: 0,
2186
- toolCount: 1,
2187
- hasToolUse: true,
2188
- history: messages,
2189
- });
2190
- }
2191
- return messages;
2192
- };
2193
-
1814
+ // Calls 1 (initial) and 2 (convergence rerun) reject with
1815
+ // context-too-large so `contextTooLargeDetected` stays true through the
1816
+ // convergence exit and the orchestrator enters the auto_compress branch.
1817
+ // Call 3 (the auto_compress rerun) is a tool turn: the loop runs it
1818
+ // without a compaction hook, so when its mid-loop budget gate trips on
1819
+ // the still-oversized estimate it yields `exitReason = "budget"` rather
1820
+ // than recovering — the silent-stall path under test.
2194
1821
  const ctx = makeCtx({
2195
- agentLoopRun,
1822
+ providerResponses: [
1823
+ new Error("context_length_exceeded"),
1824
+ new Error("context_length_exceeded"),
1825
+ toolUseResponse("t1", "read_file", { path: "/a.txt" }),
1826
+ ],
1827
+ loopTools: [
1828
+ {
1829
+ name: "read_file",
1830
+ description: "Read a file",
1831
+ input_schema: {
1832
+ type: "object",
1833
+ properties: { path: { type: "string" } },
1834
+ },
1835
+ },
1836
+ ],
1837
+ toolExecutor: async () => ({ content: "data", isError: false }),
2196
1838
  hasNoClient: true,
2197
1839
  contextWindowManager: {
2198
1840
  shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
@@ -2275,23 +1917,10 @@ describe("session-agent-loop", () => {
2275
1917
  };
2276
1918
  };
2277
1919
 
2278
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
2279
- onEvent({
2280
- type: "error",
2281
- error: new Error("context_length_exceeded"),
2282
- });
2283
- onEvent({
2284
- type: "usage",
2285
- inputTokens: 100,
2286
- outputTokens: 0,
2287
- model: "test-model",
2288
- providerDurationMs: 50,
2289
- });
2290
- return messages;
2291
- };
2292
-
1920
+ // The provider rejects every call with a context-too-large error, so the
1921
+ // orchestrator keeps retrying until it hits the attempt ceiling.
2293
1922
  const ctx = makeCtx({
2294
- agentLoopRun,
1923
+ providerResponses: [new Error("context_length_exceeded")],
2295
1924
  contextWindowManager: {
2296
1925
  shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
2297
1926
  maybeCompact: async () => ({ compacted: false }),
@@ -2307,7 +1936,6 @@ describe("session-agent-loop", () => {
2307
1936
  test("preflight budget evaluation invokes reducer before provider call", async () => {
2308
1937
  const events: ServerMessage[] = [];
2309
1938
  let reducerCalls = 0;
2310
- let agentLoopCalls = 0;
2311
1939
 
2312
1940
  // Set token estimate above budget (100000 * 0.95 = 95000)
2313
1941
  mockEstimateTokens = 96000;
@@ -2326,36 +1954,11 @@ describe("session-agent-loop", () => {
2326
1954
  };
2327
1955
  };
2328
1956
 
2329
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
2330
- agentLoopCalls++;
2331
- // Prime the assistant row anchor — production code emits this from
2332
- // `AgentLoop.run` just before `provider.sendMessage`.
2333
- await onEvent({ type: "llm_call_started" });
2334
- onEvent({
2335
- type: "message_complete",
2336
- message: {
2337
- role: "assistant",
2338
- content: [{ type: "text", text: "ok" }],
2339
- },
2340
- });
2341
- onEvent({
2342
- type: "usage",
2343
- inputTokens: 50,
2344
- outputTokens: 25,
2345
- model: "test-model",
2346
- providerDurationMs: 100,
2347
- });
2348
- return [
2349
- ...messages,
2350
- {
2351
- role: "assistant" as const,
2352
- content: [{ type: "text", text: "ok" }] as ContentBlock[],
2353
- },
2354
- ];
2355
- };
2356
-
1957
+ // After the preflight reducer brings the estimate under budget, the loop
1958
+ // completes the turn in a single provider call.
1959
+ const { provider, calls } = createMockProvider([textResponse("ok")]);
2357
1960
  const ctx = makeCtx({
2358
- agentLoopRun,
1961
+ loopProvider: provider,
2359
1962
  contextWindowManager: {
2360
1963
  shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
2361
1964
  maybeCompact: async () => ({ compacted: false }),
@@ -2366,8 +1969,8 @@ describe("session-agent-loop", () => {
2366
1969
 
2367
1970
  // Reducer should have been called during preflight
2368
1971
  expect(reducerCalls).toBeGreaterThanOrEqual(1);
2369
- // Agent loop should still succeed
2370
- expect(agentLoopCalls).toBe(1);
1972
+ // Agent loop should still succeed in a single provider call
1973
+ expect(calls.length).toBe(1);
2371
1974
  const complete = events.find((e) => e.type === "message_complete");
2372
1975
  expect(complete).toBeDefined();
2373
1976
  });
@@ -2376,78 +1979,28 @@ describe("session-agent-loop", () => {
2376
1979
  describe("provider ordering error retry", () => {
2377
1980
  test("retries with deep repair when ordering error is detected", async () => {
2378
1981
  const events: ServerMessage[] = [];
2379
- let callCount = 0;
2380
-
2381
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
2382
- // Prime the assistant row anchor — production code emits this from
2383
- // `AgentLoop.run` just before `provider.sendMessage`. Retry branches
2384
- // need this on every invocation: each agent-loop iteration reserves
2385
- // its own row.
2386
- await onEvent({ type: "llm_call_started" });
2387
- callCount++;
2388
- if (callCount === 1) {
2389
- onEvent({
2390
- type: "error",
2391
- error: new Error("messages ordering error"),
2392
- });
2393
- onEvent({
2394
- type: "usage",
2395
- inputTokens: 100,
2396
- outputTokens: 0,
2397
- model: "test-model",
2398
- providerDurationMs: 50,
2399
- });
2400
- return messages;
2401
- }
2402
- // Retry succeeds
2403
- onEvent({
2404
- type: "message_complete",
2405
- message: {
2406
- role: "assistant",
2407
- content: [{ type: "text", text: "fixed" }],
2408
- },
2409
- });
2410
- onEvent({
2411
- type: "usage",
2412
- inputTokens: 50,
2413
- outputTokens: 25,
2414
- model: "test-model",
2415
- providerDurationMs: 100,
2416
- });
2417
- return [
2418
- ...messages,
2419
- {
2420
- role: "assistant" as const,
2421
- content: [{ type: "text", text: "fixed" }] as ContentBlock[],
2422
- },
2423
- ];
2424
- };
2425
1982
 
2426
- const ctx = makeCtx({ agentLoopRun });
1983
+ // The provider rejects the first call with an ordering error, then
1984
+ // succeeds once the orchestrator's deep repair re-sends the turn.
1985
+ const { provider, calls } = createMockProvider([
1986
+ new Error("messages ordering error"),
1987
+ textResponse("fixed"),
1988
+ ]);
1989
+
1990
+ const ctx = makeCtx({ loopProvider: provider });
2427
1991
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
2428
1992
 
2429
- expect(callCount).toBe(2);
1993
+ expect(calls.length).toBe(2);
2430
1994
  });
2431
1995
 
2432
1996
  test("emits deferred ordering error when retry also fails", async () => {
2433
1997
  const events: ServerMessage[] = [];
2434
1998
 
2435
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
2436
- onEvent({
2437
- type: "error",
2438
- error: new Error("messages ordering error"),
2439
- });
2440
- onEvent({
2441
- type: "usage",
2442
- inputTokens: 100,
2443
- outputTokens: 0,
2444
- model: "test-model",
2445
- providerDurationMs: 50,
2446
- });
2447
- return messages;
2448
- };
2449
-
2450
- const ctx = makeCtx({ agentLoopRun });
1999
+ // The provider rejects every call with an ordering error, so even the
2000
+ // deep-repair retry fails and the orchestrator surfaces the error.
2001
+ const ctx = makeCtx({
2002
+ providerResponses: [new Error("messages ordering error")],
2003
+ });
2451
2004
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
2452
2005
 
2453
2006
  const conversationError = events.find(
@@ -2461,68 +2014,18 @@ describe("session-agent-loop", () => {
2461
2014
  test("yields at checkpoint when canHandoffAtCheckpoint returns true", async () => {
2462
2015
  const events: ServerMessage[] = [];
2463
2016
 
2464
- const agentLoopRun: AgentLoopRun = async (
2465
- messages,
2466
- onEvent,
2467
- _signal,
2468
- _reqId,
2469
- onCheckpoint,
2470
- ) => {
2471
- // Prime the assistant row anchor — production code emits this from
2472
- // `AgentLoop.run` just before `provider.sendMessage`. Retry branches
2473
- // need this on every invocation: each agent-loop iteration reserves
2474
- // its own row.
2475
- await onEvent({ type: "llm_call_started" });
2476
- // Simulate tool use followed by checkpoint
2477
- onEvent({ type: "tool_use", id: "tu-1", name: "file_read", input: {} });
2478
- onEvent({
2479
- type: "tool_result",
2480
- toolUseId: "tu-1",
2481
- content: "file content",
2482
- isError: false,
2483
- });
2484
- onEvent({
2485
- type: "message_complete",
2486
- message: {
2487
- role: "assistant",
2488
- content: [{ type: "text", text: "partial" }],
2489
- },
2490
- });
2491
- onEvent({
2492
- type: "usage",
2493
- inputTokens: 100,
2494
- outputTokens: 50,
2495
- model: "test-model",
2496
- providerDurationMs: 100,
2497
- });
2498
- if (onCheckpoint) {
2499
- const decision = await onCheckpoint({
2500
- turnIndex: 0,
2501
- toolCount: 1,
2502
- hasToolUse: true,
2503
- history: messages,
2504
- });
2505
- if (decision === "yield") {
2506
- return [
2507
- ...messages,
2508
- {
2509
- role: "assistant" as const,
2510
- content: [{ type: "text", text: "partial" }] as ContentBlock[],
2511
- },
2512
- ];
2513
- }
2514
- }
2515
- return [
2516
- ...messages,
2017
+ // A tool turn drives the loop to its first mid-loop checkpoint, where the
2018
+ // orchestrator yields for a queued handoff.
2019
+ const ctx = makeCtx({
2020
+ providerResponses: [toolUseResponse("tu-1", "file_read", {})],
2021
+ loopTools: [
2517
2022
  {
2518
- role: "assistant" as const,
2519
- content: [{ type: "text", text: "partial" }] as ContentBlock[],
2023
+ name: "file_read",
2024
+ description: "Read a file",
2025
+ input_schema: { type: "object", properties: {} },
2520
2026
  },
2521
- ];
2522
- };
2523
-
2524
- const ctx = makeCtx({
2525
- agentLoopRun,
2027
+ ],
2028
+ toolExecutor: async () => ({ content: "file content", isError: false }),
2526
2029
  canHandoffAtCheckpoint: () => true,
2527
2030
  } as unknown as Partial<AgentLoopConversationContext>);
2528
2031
 
@@ -2539,58 +2042,21 @@ describe("session-agent-loop", () => {
2539
2042
  test("continues when canHandoffAtCheckpoint returns false", async () => {
2540
2043
  const events: ServerMessage[] = [];
2541
2044
 
2542
- const agentLoopRun: AgentLoopRun = async (
2543
- messages,
2544
- onEvent,
2545
- _signal,
2546
- _reqId,
2547
- onCheckpoint,
2548
- ) => {
2549
- // Prime the assistant row anchor — production code emits this from
2550
- // `AgentLoop.run` just before `provider.sendMessage`. Retry branches
2551
- // need this on every invocation: each agent-loop iteration reserves
2552
- // its own row.
2553
- await onEvent({ type: "llm_call_started" });
2554
- onEvent({ type: "tool_use", id: "tu-1", name: "file_read", input: {} });
2555
- onEvent({
2556
- type: "tool_result",
2557
- toolUseId: "tu-1",
2558
- content: "content",
2559
- isError: false,
2560
- });
2561
- onEvent({
2562
- type: "message_complete",
2563
- message: {
2564
- role: "assistant",
2565
- content: [{ type: "text", text: "done" }],
2566
- },
2567
- });
2568
- onEvent({
2569
- type: "usage",
2570
- inputTokens: 100,
2571
- outputTokens: 50,
2572
- model: "test-model",
2573
- providerDurationMs: 100,
2574
- });
2575
- if (onCheckpoint) {
2576
- await onCheckpoint({
2577
- turnIndex: 0,
2578
- toolCount: 1,
2579
- hasToolUse: true,
2580
- history: messages,
2581
- });
2582
- }
2583
- return [
2584
- ...messages,
2045
+ // The tool turn reaches a checkpoint, but with handoff disabled the loop
2046
+ // continues to the next turn and completes normally.
2047
+ const ctx = makeCtx({
2048
+ providerResponses: [
2049
+ toolUseResponse("tu-1", "file_read", {}),
2050
+ textResponse("done"),
2051
+ ],
2052
+ loopTools: [
2585
2053
  {
2586
- role: "assistant" as const,
2587
- content: [{ type: "text", text: "done" }] as ContentBlock[],
2054
+ name: "file_read",
2055
+ description: "Read a file",
2056
+ input_schema: { type: "object", properties: {} },
2588
2057
  },
2589
- ];
2590
- };
2591
-
2592
- const ctx = makeCtx({
2593
- agentLoopRun,
2058
+ ],
2059
+ toolExecutor: async () => ({ content: "content", isError: false }),
2594
2060
  canHandoffAtCheckpoint: () => false,
2595
2061
  } as unknown as Partial<AgentLoopConversationContext>);
2596
2062
 
@@ -2612,36 +2078,18 @@ describe("session-agent-loop", () => {
2612
2078
  const events: ServerMessage[] = [];
2613
2079
  const abortController = new AbortController();
2614
2080
 
2615
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
2616
- // Prime the assistant row anchor production code emits this from
2617
- // `AgentLoop.run` just before `provider.sendMessage`.
2618
- await onEvent({ type: "llm_call_started" });
2619
- onEvent({
2620
- type: "message_complete",
2621
- message: {
2622
- role: "assistant",
2623
- content: [{ type: "text", text: "partial" }],
2624
- },
2625
- });
2626
- onEvent({
2627
- type: "usage",
2628
- inputTokens: 100,
2629
- outputTokens: 50,
2630
- model: "test-model",
2631
- providerDurationMs: 100,
2632
- });
2633
- // Simulate abort after processing
2634
- abortController.abort();
2635
- return [
2636
- ...messages,
2637
- {
2638
- role: "assistant" as const,
2639
- content: [{ type: "text", text: "partial" }] as ContentBlock[],
2640
- },
2641
- ];
2081
+ // The provider completes its response but the user cancels mid-turn, so
2082
+ // the orchestrator observes the aborted signal once the loop returns.
2083
+ const provider: Provider = {
2084
+ name: "mock",
2085
+ async sendMessage(_messages, options) {
2086
+ options?.onEvent?.({ type: "text_delta", text: "partial" });
2087
+ abortController.abort();
2088
+ return textResponse("partial");
2089
+ },
2642
2090
  };
2643
2091
 
2644
- const ctx = makeCtx({ agentLoopRun, abortController });
2092
+ const ctx = makeCtx({ loopProvider: provider, abortController });
2645
2093
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
2646
2094
 
2647
2095
  const cancelled = events.find((e) => e.type === "generation_cancelled");
@@ -2652,13 +2100,16 @@ describe("session-agent-loop", () => {
2652
2100
  const events: ServerMessage[] = [];
2653
2101
  const abortController = new AbortController();
2654
2102
 
2655
- const agentLoopRun: AgentLoopRun = async () => {
2656
- abortController.abort();
2657
- const err = new DOMException("The operation was aborted", "AbortError");
2658
- throw err;
2103
+ // The provider rejects with an AbortError after the user cancels.
2104
+ const provider: Provider = {
2105
+ name: "mock",
2106
+ async sendMessage() {
2107
+ abortController.abort();
2108
+ throw new DOMException("The operation was aborted", "AbortError");
2109
+ },
2659
2110
  };
2660
2111
 
2661
- const ctx = makeCtx({ agentLoopRun, abortController });
2112
+ const ctx = makeCtx({ loopProvider: provider, abortController });
2662
2113
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
2663
2114
 
2664
2115
  const cancelled = events.find((e) => e.type === "generation_cancelled");
@@ -2675,36 +2126,17 @@ describe("session-agent-loop", () => {
2675
2126
  const abortController = new AbortController();
2676
2127
  resolveAssistantAttachmentsMock.mockClear();
2677
2128
 
2678
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
2679
- // Prime the assistant row anchor — production code emits this from
2680
- // `AgentLoop.run` just before `provider.sendMessage`.
2681
- await onEvent({ type: "llm_call_started" });
2682
- onEvent({
2683
- type: "message_complete",
2684
- message: {
2685
- role: "assistant",
2686
- content: [{ type: "text", text: "partial" }],
2687
- },
2688
- });
2689
- onEvent({
2690
- type: "usage",
2691
- inputTokens: 100,
2692
- outputTokens: 50,
2693
- model: "test-model",
2694
- providerDurationMs: 100,
2695
- });
2696
- // Simulate abort after processing
2697
- abortController.abort();
2698
- return [
2699
- ...messages,
2700
- {
2701
- role: "assistant" as const,
2702
- content: [{ type: "text", text: "partial" }] as ContentBlock[],
2703
- },
2704
- ];
2129
+ // The provider completes its response but the user cancels mid-turn.
2130
+ const provider: Provider = {
2131
+ name: "mock",
2132
+ async sendMessage(_messages, options) {
2133
+ options?.onEvent?.({ type: "text_delta", text: "partial" });
2134
+ abortController.abort();
2135
+ return textResponse("partial");
2136
+ },
2705
2137
  };
2706
2138
 
2707
- const ctx = makeCtx({ agentLoopRun, abortController });
2139
+ const ctx = makeCtx({ loopProvider: provider, abortController });
2708
2140
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
2709
2141
 
2710
2142
  const cancelled = events.find((e) => e.type === "generation_cancelled");
@@ -2716,96 +2148,50 @@ describe("session-agent-loop", () => {
2716
2148
 
2717
2149
  describe("finally block cleanup", () => {
2718
2150
  test("increments turnCount after successful run", async () => {
2719
- const ctx = makeCtx({
2720
- agentLoopRun: async (messages, onEvent) => {
2721
- // Prime the assistant row anchor — production code emits this from
2722
- // `AgentLoop.run` just before `provider.sendMessage`.
2723
- await onEvent({ type: "llm_call_started" });
2724
- onEvent({
2725
- type: "message_complete",
2726
- message: {
2727
- role: "assistant",
2728
- content: [{ type: "text", text: "hi" }],
2729
- },
2730
- });
2731
- onEvent({
2732
- type: "usage",
2733
- inputTokens: 10,
2734
- outputTokens: 5,
2735
- model: "test",
2736
- providerDurationMs: 50,
2737
- });
2738
- return [
2739
- ...messages,
2740
- {
2741
- role: "assistant" as const,
2742
- content: [{ type: "text", text: "hi" }] as ContentBlock[],
2743
- },
2744
- ];
2745
- },
2746
- });
2151
+ // GIVEN a real loop that answers in a single text turn
2152
+ const ctx = makeCtx({ providerResponses: [textResponse("hi")] });
2747
2153
  expect(ctx.turnCount).toBe(0);
2748
2154
 
2155
+ // WHEN the orchestrator runs the turn to completion
2749
2156
  await runAgentLoopImpl(ctx, "hi", "msg-1", () => {});
2750
2157
 
2158
+ // THEN the finally block increments the turn count
2751
2159
  expect(ctx.turnCount).toBe(1);
2752
2160
  });
2753
2161
 
2754
2162
  test("clears processing state and abort controller", async () => {
2755
- const ctx = makeCtx({
2756
- agentLoopRun: async (messages, onEvent) => {
2757
- // Prime the assistant row anchor — production code emits this from
2758
- // `AgentLoop.run` just before `provider.sendMessage`.
2759
- await onEvent({ type: "llm_call_started" });
2760
- onEvent({
2761
- type: "message_complete",
2762
- message: {
2763
- role: "assistant",
2764
- content: [{ type: "text", text: "hi" }],
2765
- },
2766
- });
2767
- onEvent({
2768
- type: "usage",
2769
- inputTokens: 10,
2770
- outputTokens: 5,
2771
- model: "test",
2772
- providerDurationMs: 50,
2773
- });
2774
- return [
2775
- ...messages,
2776
- {
2777
- role: "assistant" as const,
2778
- content: [{ type: "text", text: "hi" }] as ContentBlock[],
2779
- },
2780
- ];
2781
- },
2782
- });
2163
+ // GIVEN a real loop that answers in a single text turn
2164
+ const ctx = makeCtx({ providerResponses: [textResponse("hi")] });
2783
2165
 
2166
+ // WHEN the orchestrator runs the turn to completion
2784
2167
  await runAgentLoopImpl(ctx, "hi", "msg-1", () => {});
2785
2168
 
2786
- expect(ctx.processing).toBe(false);
2169
+ // THEN the finally block clears all per-turn processing state
2170
+ expect(ctx.isProcessing()).toBe(false);
2787
2171
  expect(ctx.abortController).toBeNull();
2788
2172
  expect(ctx.currentRequestId).toBeUndefined();
2789
2173
  expect(ctx.commandIntent).toBeUndefined();
2790
2174
  });
2791
2175
 
2792
- test("clears state even when agent loop throws", async () => {
2176
+ test("clears state and surfaces a processing error when the provider call fails", async () => {
2177
+ // GIVEN a real loop whose provider rejects with an unexpected error
2793
2178
  const events: ServerMessage[] = [];
2794
2179
  const ctx = makeCtx({
2795
- agentLoopRun: async () => {
2796
- throw new Error("unexpected crash");
2797
- },
2180
+ loopProvider: {
2181
+ name: "mock-provider",
2182
+ async sendMessage() {
2183
+ throw new Error("unexpected crash");
2184
+ },
2185
+ } as unknown as Provider,
2798
2186
  });
2799
2187
 
2188
+ // WHEN the orchestrator runs the turn
2800
2189
  await runAgentLoopImpl(ctx, "hi", "msg-1", (msg) => events.push(msg));
2801
2190
 
2802
- expect(ctx.processing).toBe(false);
2191
+ // THEN the finally block clears per-turn state and the failure is
2192
+ // surfaced as a processing-failed conversation error
2193
+ expect(ctx.isProcessing()).toBe(false);
2803
2194
  expect(ctx.abortController).toBeNull();
2804
- expect(events.find((event) => event.type === "error")).toMatchObject({
2805
- type: "error",
2806
- code: "CONVERSATION_PROCESSING_FAILED",
2807
- errorCategory: "processing_failed",
2808
- });
2809
2195
  expect(
2810
2196
  events.find((event) => event.type === "conversation_error"),
2811
2197
  ).toMatchObject({
@@ -2816,46 +2202,19 @@ describe("session-agent-loop", () => {
2816
2202
  });
2817
2203
 
2818
2204
  test("drains queue after completion", async () => {
2205
+ // GIVEN a real loop that answers in a single text turn
2819
2206
  let drainReason: string | undefined;
2820
2207
  const ctx = makeCtx({
2821
- agentLoopRun: async (
2822
- messages: Message[],
2823
- onEvent: (event: AgentEvent) => void | Promise<void>,
2824
- ) => {
2825
- // Prime the assistant row anchor — production code emits this from
2826
- // `AgentLoop.run` just before `provider.sendMessage`. Must be
2827
- // awaited so the assistant row is reserved before message_complete
2828
- // tries to write into it.
2829
- await onEvent({ type: "llm_call_started" });
2830
- onEvent({
2831
- type: "message_complete",
2832
- message: {
2833
- role: "assistant",
2834
- content: [{ type: "text", text: "ok" }],
2835
- },
2836
- });
2837
- onEvent({
2838
- type: "usage",
2839
- inputTokens: 10,
2840
- outputTokens: 5,
2841
- model: "test",
2842
- providerDurationMs: 50,
2843
- });
2844
- return [
2845
- ...messages,
2846
- {
2847
- role: "assistant" as const,
2848
- content: [{ type: "text", text: "ok" }] as ContentBlock[],
2849
- },
2850
- ];
2851
- },
2208
+ providerResponses: [textResponse("ok")],
2852
2209
  drainQueue: (reason: string) => {
2853
2210
  drainReason = reason;
2854
2211
  },
2855
2212
  } as unknown as Partial<AgentLoopConversationContext>);
2856
2213
 
2214
+ // WHEN the orchestrator runs the turn to completion
2857
2215
  await runAgentLoopImpl(ctx, "hi", "msg-1", () => {});
2858
2216
 
2217
+ // THEN the queue is drained with the loop-complete reason
2859
2218
  expect(drainReason).toBe("loop_complete");
2860
2219
  });
2861
2220
  });
@@ -2974,7 +2333,7 @@ describe("session-agent-loop", () => {
2974
2333
  isUserMessage: true,
2975
2334
  });
2976
2335
 
2977
- expect(ctx.processing).toBe(false);
2336
+ expect(ctx.isProcessing()).toBe(false);
2978
2337
  expect(ctx.abortController).toBeNull();
2979
2338
  expect(ctx.currentRequestId).toBeUndefined();
2980
2339
  });
@@ -3084,24 +2443,17 @@ describe("session-agent-loop", () => {
3084
2443
  test("synthesizes error assistant message when provider returns no response", async () => {
3085
2444
  const events: ServerMessage[] = [];
3086
2445
 
3087
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
3088
- // Emit a non-ordering, non-context-too-large error that sets providerErrorUserMessage
3089
- onEvent({
3090
- type: "error",
3091
- error: new Error("Internal processing failure"),
3092
- });
3093
- onEvent({
3094
- type: "usage",
3095
- inputTokens: 100,
3096
- outputTokens: 0,
3097
- model: "test-model",
3098
- providerDurationMs: 50,
3099
- });
3100
- // Return same messages (no assistant message appended)
3101
- return messages;
3102
- };
3103
-
3104
- const ctx = makeCtx({ agentLoopRun });
2446
+ // GIVEN a real loop whose provider rejects with a generic error
2447
+ // (non-ordering, non-context-too-large) so the loop emits `error` and
2448
+ // the orchestrator sets `providerErrorUserMessage`.
2449
+ const ctx = makeCtx({
2450
+ loopProvider: {
2451
+ name: "mock-provider",
2452
+ async sendMessage() {
2453
+ throw new Error("Internal processing failure");
2454
+ },
2455
+ } as unknown as Provider,
2456
+ });
3105
2457
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
3106
2458
 
3107
2459
  // The error should be sent as a conversation_error (not as an
@@ -3125,26 +2477,19 @@ describe("session-agent-loop", () => {
3125
2477
  // sweep would wrong-attach this row to the wrong assistant message.
3126
2478
  const events: ServerMessage[] = [];
3127
2479
 
3128
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
3129
- // 1) handleProviderError -> writes an `llm_request_logs` row with
3130
- // messageId=null (the orphan we are trying to link).
3131
- onEvent({
3132
- type: "provider_error",
3133
- error: new Error("upstream 500"),
3134
- rawRequest: { model: "gpt-4.1", messages: [] },
3135
- actualProvider: "openai",
3136
- });
3137
- // 2) handleError -> sets `state.providerErrorUserMessage`, which
3138
- // activates the synthetic-message branch below the loop.
3139
- onEvent({
3140
- type: "error",
3141
- error: new Error("upstream 500"),
3142
- });
3143
- // Provider returned no assistant content — same messages back.
3144
- return messages;
3145
- };
3146
-
3147
- const ctx = makeCtx({ agentLoopRun });
2480
+ // GIVEN a real loop whose provider rejects: the loop emits
2481
+ // `provider_error` (writing an `llm_request_logs` row with
2482
+ // messageId=null the orphan we link) then `error` (which sets
2483
+ // `state.providerErrorUserMessage`, activating the synthetic-message
2484
+ // branch below the loop).
2485
+ const ctx = makeCtx({
2486
+ loopProvider: {
2487
+ name: "mock-provider",
2488
+ async sendMessage() {
2489
+ throw new Error("upstream 500");
2490
+ },
2491
+ } as unknown as Provider,
2492
+ });
3148
2493
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
3149
2494
 
3150
2495
  // The orphan was written with messageId=undefined.
@@ -3191,39 +2536,10 @@ describe("session-agent-loop", () => {
3191
2536
  // observe the sync-invalidation publish path on the same turn.
3192
2537
  projectAssistantMessageMock.mockImplementationOnce(() => true);
3193
2538
 
3194
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
3195
- await onEvent({ type: "llm_call_started" });
3196
- // `message_complete` is awaited so `handleMessageComplete` (and its
3197
- // async indexer + projector chain) completes before the next event
3198
- // or before the loop returns. Without the await the projector's
3199
- // synchronous call still races against the test's assertion phase
3200
- // because the indexer's `await` yields microtasks.
3201
- await onEvent({
3202
- type: "message_complete",
3203
- message: {
3204
- role: "assistant",
3205
- content: [{ type: "text", text: "indexed reply" }],
3206
- },
3207
- });
3208
- onEvent({
3209
- type: "usage",
3210
- inputTokens: 10,
3211
- outputTokens: 5,
3212
- model: "test",
3213
- providerDurationMs: 50,
3214
- });
3215
- return [
3216
- ...messages,
3217
- {
3218
- role: "assistant" as const,
3219
- content: [
3220
- { type: "text", text: "indexed reply" },
3221
- ] as ContentBlock[],
3222
- },
3223
- ];
3224
- };
3225
-
3226
- const ctx = makeCtx({ agentLoopRun });
2539
+ // GIVEN a real loop that answers with a single finalized assistant turn
2540
+ const ctx = makeCtx({
2541
+ providerResponses: [textResponse("indexed reply")],
2542
+ });
3227
2543
  await runAgentLoopImpl(ctx, "hi", "msg-1", () => {});
3228
2544
 
3229
2545
  // Indexer fired with the reserved row's id + the finalized content.
@@ -3286,34 +2602,8 @@ describe("session-agent-loop", () => {
3286
2602
  metadata: null,
3287
2603
  };
3288
2604
 
3289
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
3290
- await onEvent({ type: "llm_call_started" });
3291
- // See sibling test — `message_complete` must be awaited so the
3292
- // projector call lands before the assertion phase.
3293
- await onEvent({
3294
- type: "message_complete",
3295
- message: {
3296
- role: "assistant",
3297
- content: [{ type: "text", text: "quiet" }],
3298
- },
3299
- });
3300
- onEvent({
3301
- type: "usage",
3302
- inputTokens: 1,
3303
- outputTokens: 1,
3304
- model: "test",
3305
- providerDurationMs: 1,
3306
- });
3307
- return [
3308
- ...messages,
3309
- {
3310
- role: "assistant" as const,
3311
- content: [{ type: "text", text: "quiet" }] as ContentBlock[],
3312
- },
3313
- ];
3314
- };
3315
-
3316
- const ctx = makeCtx({ agentLoopRun });
2605
+ // GIVEN a real loop that answers with a single finalized assistant turn
2606
+ const ctx = makeCtx({ providerResponses: [textResponse("quiet")] });
3317
2607
  await runAgentLoopImpl(ctx, "hi", "msg-1", () => {});
3318
2608
 
3319
2609
  expect(projectAssistantMessageMock).toHaveBeenCalledTimes(1);
@@ -3338,40 +2628,33 @@ describe("session-agent-loop", () => {
3338
2628
  // Indexer/projector mocks default to no-op; no finalized row in this
3339
2629
  // test, so `mockMessageById` stays null.
3340
2630
 
3341
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
3342
- // First LLM call: reserve msg-strand-A, never finalize.
3343
- await onEvent({ type: "llm_call_started" });
3344
- // Second LLM call: should delete msg-strand-A before reserving
3345
- // msg-strand-B.
3346
- await onEvent({ type: "llm_call_started" });
3347
- // Finalize the second one so the loop has a valid assistant message
3348
- // and exits cleanly.
3349
- onEvent({
3350
- type: "message_complete",
3351
- message: {
3352
- role: "assistant",
3353
- content: [{ type: "text", text: "retry succeeded" }],
3354
- },
3355
- });
3356
- onEvent({
3357
- type: "usage",
3358
- inputTokens: 5,
3359
- outputTokens: 3,
3360
- model: "test",
3361
- providerDurationMs: 25,
3362
- });
3363
- return [
3364
- ...messages,
3365
- {
3366
- role: "assistant" as const,
3367
- content: [
3368
- { type: "text", text: "retry succeeded" },
3369
- ] as ContentBlock[],
3370
- },
3371
- ];
3372
- };
2631
+ // A single reducer tier converges the oversized context so the
2632
+ // orchestrator re-enters the loop after the first call fails.
2633
+ mockReducerStepFn = (msgs: Message[]) => ({
2634
+ messages: msgs,
2635
+ tier: "forced_compaction",
2636
+ state: {
2637
+ appliedTiers: ["forced_compaction"],
2638
+ injectionMode: "full",
2639
+ exhausted: false,
2640
+ },
2641
+ estimatedTokens: 5000,
2642
+ });
3373
2643
 
3374
- const ctx = makeCtx({ agentLoopRun });
2644
+ // GIVEN a real loop whose first call rejects with context-too-large
2645
+ // (reserving msg-strand-A but never finalizing it), then recovers via
2646
+ // convergence on re-entry. The re-entry's `llm_call_started` must
2647
+ // delete the stranded msg-strand-A before reserving msg-strand-B.
2648
+ const ctx = makeCtx({
2649
+ providerResponses: [
2650
+ new Error("context_length_exceeded"),
2651
+ textResponse("retry succeeded"),
2652
+ ],
2653
+ contextWindowManager: {
2654
+ shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
2655
+ maybeCompact: async () => ({ compacted: false }),
2656
+ } as unknown as AgentLoopConversationContext["contextWindowManager"],
2657
+ });
3375
2658
  await runAgentLoopImpl(ctx, "hi", "msg-1", () => {});
3376
2659
 
3377
2660
  // Exactly one delete fires — for msg-strand-A, before the second
@@ -3399,27 +2682,20 @@ describe("session-agent-loop", () => {
3399
2682
  id: "msg-orphaned-reservation",
3400
2683
  }));
3401
2684
 
3402
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
3403
- // Reserve the orphan.
3404
- await onEvent({ type: "llm_call_started" });
3405
- // Provider rejects writes the llm_request_log row and arms
3406
- // `state.providerErrorUserMessage` via `handleError`.
3407
- onEvent({
3408
- type: "provider_error",
3409
- error: new Error("upstream 500"),
3410
- rawRequest: { model: "gpt-4.1", messages: [] },
3411
- actualProvider: "openai",
3412
- });
3413
- onEvent({
3414
- type: "error",
3415
- error: new Error("upstream 500"),
3416
- });
3417
- // No assistant message in the result — the synthetic-error branch
3418
- // below the agent loop fires.
3419
- return messages;
3420
- };
3421
-
3422
- const ctx = makeCtx({ agentLoopRun });
2685
+ // GIVEN a real loop that reserves an assistant row at
2686
+ // `llm_call_started`, then whose provider rejects: the loop emits
2687
+ // `provider_error` (writing the llm_request_log row) and `error`
2688
+ // (arming `state.providerErrorUserMessage`), exiting with no
2689
+ // `message_complete` so the synthetic-error branch below the loop
2690
+ // fires.
2691
+ const ctx = makeCtx({
2692
+ loopProvider: {
2693
+ name: "mock-provider",
2694
+ async sendMessage() {
2695
+ throw new Error("upstream 500");
2696
+ },
2697
+ } as unknown as Provider,
2698
+ });
3423
2699
  await runAgentLoopImpl(ctx, "hi", "msg-1", () => {});
3424
2700
 
3425
2701
  // The orphan was deleted exactly once, before the synthetic error
@@ -3444,6 +2720,315 @@ describe("session-agent-loop", () => {
3444
2720
  });
3445
2721
  });
3446
2722
 
2723
+ describe("partial persistence", () => {
2724
+ // The legacy flow reserves an empty assistant row at `llm_call_started`
2725
+ // (`content: "[]"`) and never touches it again until
2726
+ // `handleMessageComplete` fires the single authoritative
2727
+ // `updateContent`. Between those events the row is empty for the full
2728
+ // duration of a turn — a browser refresh mid-turn sees nothing where
2729
+ // the in-progress assistant reply should be.
2730
+ //
2731
+ // Partial persistence closes that durability gap with a debounced
2732
+ // flush from `handleTextDelta` (250ms timer). `handleToolUse`
2733
+ // intentionally does NOT flush — `AgentLoop.run` emits `tool_use`
2734
+ // strictly AFTER `message_complete`, so any flush from that handler
2735
+ // would land after the authoritative finalize and overwrite the
2736
+ // finalized row. The indexer + projector still fire ONLY at
2737
+ // `message_complete` — partial rows are never indexed.
2738
+ //
2739
+ // These tests pin down the wire-level contract by counting
2740
+ // `updateMessageContent` calls and inspecting the JSON payload of the
2741
+ // partial-flush writes. The indexing / sync-invalidation paths are
2742
+ // covered by the pre-allocation block above.
2743
+
2744
+ test("debounced time gate flushes one partial write after PARTIAL_PERSIST_DEBOUNCE_MS", async () => {
2745
+ mockMessageById = {
2746
+ id: "msg-reserve",
2747
+ conversationId: "test-conv",
2748
+ createdAt: 1234567,
2749
+ role: "assistant",
2750
+ content: "[]",
2751
+ metadata: null,
2752
+ };
2753
+
2754
+ // GIVEN a real loop whose provider streams two small deltas (each under
2755
+ // the 1024-char size gate) then holds the turn open past the 250ms
2756
+ // debounce window before completing, so a single debounced partial
2757
+ // flush lands before `message_complete`.
2758
+ const ctx = makeCtx({
2759
+ loopProvider: {
2760
+ name: "mock-provider",
2761
+ async sendMessage(_messages, options) {
2762
+ options?.onEvent?.({ type: "text_delta", text: "Hello, " });
2763
+ options?.onEvent?.({ type: "text_delta", text: "world." });
2764
+ await new Promise((resolve) => setTimeout(resolve, 1100));
2765
+ return textResponse("Hello, world.");
2766
+ },
2767
+ },
2768
+ });
2769
+
2770
+ // WHEN the orchestrator runs the turn to completion
2771
+ await runAgentLoopImpl(ctx, "hi", "msg-1", () => {});
2772
+
2773
+ // Exactly two `updateContent` calls land:
2774
+ // 1. the debounced partial flush after both deltas accumulated, and
2775
+ // 2. the final authoritative flush in `handleMessageComplete`.
2776
+ // Without the debounce gate this would be one-per-delta + one final
2777
+ // (3). Without the partial flush at all it would be just 1.
2778
+ expect(updateMessageContentMock).toHaveBeenCalledTimes(2);
2779
+ const calls = updateMessageContentMock.mock.calls as unknown as Array<
2780
+ [string, string]
2781
+ >;
2782
+ const partialFlush = calls[0];
2783
+ expect(partialFlush?.[0]).toBe("msg-reserve");
2784
+ const partialBlocks = JSON.parse(partialFlush?.[1] ?? "[]") as Array<{
2785
+ type: string;
2786
+ text?: string;
2787
+ }>;
2788
+ expect(partialBlocks).toEqual([{ type: "text", text: "Hello, world." }]);
2789
+ });
2790
+
2791
+ test("handleToolUse does NOT trigger a partial flush of its own", async () => {
2792
+ // `AgentLoop.run` emits `tool_use` strictly AFTER `message_complete`,
2793
+ // so a flush from the tool_use handler would land after the
2794
+ // authoritative final `updateContent` and overwrite the finalized
2795
+ // row (Codex P1 / Vargas review feedback). The handler must be a
2796
+ // no-op for the partial-persist accumulator.
2797
+ mockMessageById = {
2798
+ id: "msg-reserve",
2799
+ conversationId: "test-conv",
2800
+ createdAt: 1234567,
2801
+ role: "assistant",
2802
+ content: "[]",
2803
+ metadata: null,
2804
+ };
2805
+
2806
+ // GIVEN a real loop that runs one tool turn — the loop emits `tool_use`
2807
+ // strictly AFTER `message_complete` — and then answers with a final
2808
+ // text turn. The tool executor returns immediately.
2809
+ const ctx = makeCtx({
2810
+ providerResponses: [
2811
+ toolUseResponse("tu-no-flush", "file_read", { path: "/foo" }),
2812
+ textResponse("done"),
2813
+ ],
2814
+ loopTools: [
2815
+ {
2816
+ name: "file_read",
2817
+ description: "Read a file",
2818
+ input_schema: {
2819
+ type: "object",
2820
+ properties: { path: { type: "string" } },
2821
+ },
2822
+ },
2823
+ ],
2824
+ toolExecutor: async () => ({ content: "ok", isError: false }),
2825
+ });
2826
+
2827
+ // WHEN the orchestrator runs the turn to completion
2828
+ await runAgentLoopImpl(ctx, "hi", "msg-1", () => {});
2829
+
2830
+ // Four authoritative writes land and no stray partial flush:
2831
+ // - one final flush per `message_complete` (the tool turn and the final
2832
+ // text turn), plus
2833
+ // - two grouped tool-result user-row writes (persist-on-arrival and the
2834
+ // turn-boundary finalize).
2835
+ // `handleToolUse` contributes no partial flush of its own; one would make
2836
+ // this 5. That stray flush is the regression this test guards against.
2837
+ expect(updateMessageContentMock).toHaveBeenCalledTimes(4);
2838
+ });
2839
+
2840
+ test("handleMessageComplete clears any pending debounce timer before the final flush", async () => {
2841
+ mockMessageById = {
2842
+ id: "msg-reserve",
2843
+ conversationId: "test-conv",
2844
+ createdAt: 1234567,
2845
+ role: "assistant",
2846
+ content: "[]",
2847
+ metadata: null,
2848
+ };
2849
+
2850
+ // GIVEN a real loop whose first turn streams a short delta (scheduling a
2851
+ // debounce timer) and completes as a tool turn — so `message_complete`
2852
+ // arrives before the 250ms timer and clears it. The tool executor then
2853
+ // holds the loop open well past the original debounce window, proving a
2854
+ // late timer does NOT fire a stray partial flush, before a final text
2855
+ // turn ends the run.
2856
+ const ctx = makeCtx({
2857
+ providerResponses: [
2858
+ {
2859
+ content: [
2860
+ { type: "text", text: "Quick reply." },
2861
+ {
2862
+ type: "tool_use",
2863
+ id: "tu-keep-alive",
2864
+ name: "file_read",
2865
+ input: {},
2866
+ },
2867
+ ],
2868
+ model: "mock-model",
2869
+ usage: { inputTokens: 10, outputTokens: 5 },
2870
+ stopReason: "tool_use",
2871
+ },
2872
+ textResponse("done"),
2873
+ ],
2874
+ loopTools: [
2875
+ {
2876
+ name: "file_read",
2877
+ description: "Read a file",
2878
+ input_schema: { type: "object", properties: {} },
2879
+ },
2880
+ ],
2881
+ toolExecutor: async () => {
2882
+ await new Promise((resolve) => setTimeout(resolve, 1100));
2883
+ return { content: "ok", isError: false };
2884
+ },
2885
+ });
2886
+
2887
+ // WHEN the orchestrator runs the turn to completion
2888
+ await runAgentLoopImpl(ctx, "hi", "msg-1", () => {});
2889
+
2890
+ // Four authoritative writes land: one final flush per `message_complete`
2891
+ // (the tool turn and the final text turn) plus two grouped tool-result
2892
+ // user-row writes (persist-on-arrival and the turn-boundary finalize).
2893
+ // The debounced partial would have fired around T+250ms — during the tool
2894
+ // executor's hold — but the timer-clear at the top of
2895
+ // `handleMessageComplete` cancels it, so no stray fifth flush appears.
2896
+ expect(updateMessageContentMock).toHaveBeenCalledTimes(4);
2897
+ });
2898
+
2899
+ test("partial flushes never trigger the indexer or attention projector", async () => {
2900
+ mockMessageById = {
2901
+ id: "msg-reserve",
2902
+ conversationId: "test-conv",
2903
+ createdAt: 1234567,
2904
+ role: "assistant",
2905
+ content: "[]",
2906
+ metadata: null,
2907
+ };
2908
+
2909
+ // GIVEN a real loop whose provider streams a delta then holds the turn
2910
+ // open past the 250ms debounce window so the partial flush lands BEFORE
2911
+ // `message_complete`. The indexer/projector counts are snapshotted at
2912
+ // that mid-turn point (after the partial flush, before completion).
2913
+ let snapshot: [number, number] | undefined;
2914
+ const ctx = makeCtx({
2915
+ loopProvider: {
2916
+ name: "mock-provider",
2917
+ async sendMessage(_messages, options) {
2918
+ options?.onEvent?.({ type: "text_delta", text: "hello world" });
2919
+ await new Promise((resolve) => setTimeout(resolve, 1100));
2920
+ snapshot = [
2921
+ indexMessageNowMock.mock.calls.length,
2922
+ projectAssistantMessageMock.mock.calls.length,
2923
+ ];
2924
+ return textResponse("hello world");
2925
+ },
2926
+ },
2927
+ });
2928
+
2929
+ // WHEN the orchestrator runs the turn to completion
2930
+ await runAgentLoopImpl(ctx, "hi", "msg-1", () => {});
2931
+
2932
+ expect(snapshot).toBeDefined();
2933
+ // Indexer + projector were both ZERO during the mid-turn partial
2934
+ // flush — they only fire from `handleMessageComplete` after the
2935
+ // authoritative `updateContent`.
2936
+ expect(snapshot![0]).toBe(0);
2937
+ expect(snapshot![1]).toBe(0);
2938
+ // After the loop completes the indexer + projector each ran exactly
2939
+ // once (the pre-allocation finalize path).
2940
+ expect(indexMessageNowMock).toHaveBeenCalledTimes(1);
2941
+ expect(projectAssistantMessageMock).toHaveBeenCalledTimes(1);
2942
+ });
2943
+
2944
+ test("partial flushes redact secrets from text blocks before writing", async () => {
2945
+ mockMessageById = {
2946
+ id: "msg-reserve",
2947
+ conversationId: "test-conv",
2948
+ createdAt: 1234567,
2949
+ role: "assistant",
2950
+ content: "[]",
2951
+ metadata: null,
2952
+ };
2953
+ // A GitHub PAT-shaped token mid-stream — the redaction discipline
2954
+ // mirrors `handleMessageComplete`'s final flush so a refresh mid-turn
2955
+ // never sees plaintext credentials in the persisted row.
2956
+ const ghToken = "ghp_" + "a".repeat(36);
2957
+ const payload = "Here's the key: " + ghToken + " enjoy.";
2958
+
2959
+ // GIVEN a real loop whose provider streams the PAT-bearing payload as a
2960
+ // delta then holds the turn open past the 250ms debounce window so the
2961
+ // partial flush lands before `message_complete`.
2962
+ const ctx = makeCtx({
2963
+ loopProvider: {
2964
+ name: "mock-provider",
2965
+ async sendMessage(_messages, options) {
2966
+ options?.onEvent?.({ type: "text_delta", text: payload });
2967
+ await new Promise((resolve) => setTimeout(resolve, 1100));
2968
+ return textResponse(payload);
2969
+ },
2970
+ },
2971
+ });
2972
+
2973
+ // WHEN the orchestrator runs the turn to completion
2974
+ await runAgentLoopImpl(ctx, "hi", "msg-1", () => {});
2975
+
2976
+ expect(updateMessageContentMock).toHaveBeenCalledTimes(2);
2977
+ const partialPayload = (
2978
+ updateMessageContentMock.mock.calls[0] as unknown as [string, string]
2979
+ )[1];
2980
+ // The raw PAT must never appear in the persisted snapshot. The
2981
+ // redaction substitute is implementation-defined; the contract here
2982
+ // is "the literal token string is gone".
2983
+ expect(partialPayload).not.toContain(ghToken);
2984
+ });
2985
+
2986
+ test("provider-error cleanup deletes a row that has accumulated partial content", async () => {
2987
+ // Regression check: the pre-allocation orphan-cleanup branch
2988
+ // already deletes the reserved row when the LLM call exits via
2989
+ // `provider_error`. Partial-persist writes content to that row
2990
+ // mid-turn; the cleanup must still fire and the row (along with
2991
+ // its partial content) must still be deleted before the synthetic
2992
+ // error message lands.
2993
+ reserveMessageMock.mockImplementationOnce(async () => ({
2994
+ id: "msg-orphan-with-partial",
2995
+ }));
2996
+
2997
+ // GIVEN a real loop whose provider streams a delta — landing a debounced
2998
+ // partial flush on the reserved row — then rejects, so the loop emits
2999
+ // `provider_error` and `error` and exits with no `message_complete`.
3000
+ const ctx = makeCtx({
3001
+ loopProvider: {
3002
+ name: "mock-provider",
3003
+ async sendMessage(_messages, options) {
3004
+ options?.onEvent?.({ type: "text_delta", text: "hello world" });
3005
+ await new Promise((resolve) => setTimeout(resolve, 1100));
3006
+ throw new Error("upstream 500");
3007
+ },
3008
+ },
3009
+ });
3010
+
3011
+ // WHEN the orchestrator runs the turn
3012
+ await runAgentLoopImpl(ctx, "hi", "msg-1", () => {});
3013
+
3014
+ // Partial flush fired exactly once (before the provider error).
3015
+ // The orphan row was then deleted; the synthetic error message is
3016
+ // inserted separately via `addMessage` (`mock-msg-id`) and never
3017
+ // touched by `updateContent`.
3018
+ const partialFlushes = (
3019
+ updateMessageContentMock.mock.calls as unknown as Array<
3020
+ [string, string]
3021
+ >
3022
+ ).filter(([id]) => id === "msg-orphan-with-partial");
3023
+ expect(partialFlushes).toHaveLength(1);
3024
+ expect(deleteMessageByIdMock).toHaveBeenCalledTimes(1);
3025
+ const deleteCall = deleteMessageByIdMock.mock.calls[0] as unknown as [
3026
+ string,
3027
+ ];
3028
+ expect(deleteCall[0]).toBe("msg-orphan-with-partial");
3029
+ });
3030
+ });
3031
+
3447
3032
  describe("pkbSystemReminderBlock metadata persistence", () => {
3448
3033
  test("persists pkbSystemReminderBlock in full mode with PKB active", async () => {
3449
3034
  const reminder = "<system_reminder>\npkb content\n</system_reminder>";
@@ -3924,50 +3509,32 @@ describe("session-agent-loop", () => {
3924
3509
  compactableStartIndex: 0,
3925
3510
  };
3926
3511
 
3927
- const rawMidLoopBasis: Message[] = [
3928
- {
3929
- role: "user",
3930
- content: [{ type: "text", text: "fresh DB basis user row" }],
3931
- },
3932
- {
3933
- role: "assistant",
3934
- content: [{ type: "text", text: "partial assistant response" }],
3935
- },
3936
- ];
3937
3512
  const maybeCompactInputs: Message[][] = [];
3938
- let runCount = 0;
3939
- const agentLoopRun: AgentLoopRun = async (
3940
- messages,
3941
- _onEvent,
3942
- _signal,
3943
- _reqId,
3944
- onCheckpoint,
3945
- ) => {
3946
- runCount++;
3947
- if (runCount === 1) {
3948
- mockEstimateTokens = 90_000;
3949
- const decision = await onCheckpoint?.({
3950
- turnIndex: 0,
3951
- toolCount: 1,
3952
- hasToolUse: true,
3953
- history: messages,
3954
- });
3955
- mockEstimateTokens = 1000;
3956
- if (decision === "yield") {
3957
- return rawMidLoopBasis;
3958
- }
3959
- }
3960
- return [
3961
- ...messages,
3962
- {
3963
- role: "assistant" as const,
3964
- content: [{ type: "text" as const, text: "final response" }],
3965
- },
3966
- ];
3967
- };
3968
3513
 
3514
+ // AND a real loop that runs one tool turn and then a final text turn.
3515
+ // The tool executor raises the token estimate above the mid-loop budget
3516
+ // threshold so the loop compacts in place at the post-tool checkpoint —
3517
+ // over its own in-loop history, which does not match the loaded Slack
3518
+ // rows.
3969
3519
  const ctx = makeCtx({
3970
- agentLoopRun,
3520
+ providerResponses: [
3521
+ toolUseResponse("tu-mid-loop", "file_read", { path: "/foo" }),
3522
+ textResponse("final response"),
3523
+ ],
3524
+ loopTools: [
3525
+ {
3526
+ name: "file_read",
3527
+ description: "Read a file",
3528
+ input_schema: {
3529
+ type: "object",
3530
+ properties: { path: { type: "string" } },
3531
+ },
3532
+ },
3533
+ ],
3534
+ toolExecutor: async () => {
3535
+ mockEstimateTokens = 90_000;
3536
+ return { content: "ok", isError: false };
3537
+ },
3971
3538
  channelCapabilities: {
3972
3539
  channel: "slack",
3973
3540
  dashboardCapable: false,
@@ -4004,6 +3571,9 @@ describe("session-agent-loop", () => {
4004
3571
  summaryText: "",
4005
3572
  };
4006
3573
  }
3574
+ // The mid-loop gate compacted its in-loop basis; drop the estimate
3575
+ // back under budget so the post-compaction provider call proceeds.
3576
+ mockEstimateTokens = 1000;
4007
3577
  return {
4008
3578
  compacted: true,
4009
3579
  messages: [
@@ -4032,7 +3602,9 @@ describe("session-agent-loop", () => {
4032
3602
  await runAgentLoopImpl(ctx, "next reply", "user-msg-mid-loop", () => {});
4033
3603
 
4034
3604
  expect(maybeCompactInputs[0]).toBe(renderedSlackMessages);
4035
- expect(maybeCompactInputs[1]).toBe(rawMidLoopBasis);
3605
+ // The mid-loop gate compacts the loop's own in-loop history, never the
3606
+ // loaded Slack rows — the mismatch this test guards against.
3607
+ expect(maybeCompactInputs[1]).not.toBe(renderedSlackMessages);
4036
3608
  expect(getSlackCompactionWatermarkForPrefixMock).toHaveBeenCalledWith(
4037
3609
  null,
4038
3610
  2,
@@ -4305,67 +3877,32 @@ describe("session-agent-loop", () => {
4305
3877
  estimatedTokens: 5000,
4306
3878
  });
4307
3879
 
4308
- let callCount = 0;
4309
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
4310
- callCount++;
4311
- // Prime the assistant row anchor production code emits this from
4312
- // `AgentLoop.run` just before `provider.sendMessage`. Retry branches
4313
- // need this on every invocation: each agent-loop iteration reserves
4314
- // its own row.
4315
- await onEvent({ type: "llm_call_started" });
4316
- if (callCount === 1) {
4317
- // Trigger convergence path: error + appended assistant message so
4318
- // updatedHistory.length > preRunHistoryLength at the strip site.
4319
- onEvent({
4320
- type: "error",
4321
- error: new Error("context_length_exceeded"),
4322
- });
4323
- onEvent({
4324
- type: "usage",
4325
- inputTokens: 100,
4326
- outputTokens: 0,
4327
- model: "test-model",
4328
- providerDurationMs: 50,
4329
- });
4330
- return [
4331
- ...messages,
4332
- {
4333
- role: "assistant" as const,
4334
- content: [{ type: "text", text: "partial" }] as ContentBlock[],
4335
- },
4336
- ];
4337
- }
4338
- onEvent({
4339
- type: "message_complete",
4340
- message: {
4341
- role: "assistant",
4342
- content: [{ type: "text", text: "recovered" }],
4343
- },
4344
- });
4345
- onEvent({
4346
- type: "usage",
4347
- inputTokens: 50,
4348
- outputTokens: 25,
4349
- model: "test-model",
4350
- providerDurationMs: 100,
4351
- });
4352
- return [
4353
- ...messages,
3880
+ // GIVEN a real loop that appends a tool turn (so the run reports
3881
+ // `appendedNewMessages`) and then rejects with a context-too-large
3882
+ // error on the following call — the orchestrator strips that appended
3883
+ // history during its bounded convergence path before a final call
3884
+ // recovers.
3885
+ const ctx = makeCtx({
3886
+ providerResponses: [
3887
+ toolUseResponse("t1", "file_read", {}),
3888
+ new Error("context_length_exceeded"),
3889
+ textResponse("recovered"),
3890
+ ],
3891
+ loopTools: [
4354
3892
  {
4355
- role: "assistant" as const,
4356
- content: [{ type: "text", text: "recovered" }] as ContentBlock[],
3893
+ name: "file_read",
3894
+ description: "Read a file",
3895
+ input_schema: { type: "object", properties: {} },
4357
3896
  },
4358
- ];
4359
- };
4360
-
4361
- const ctx = makeCtx({
4362
- agentLoopRun,
3897
+ ],
3898
+ toolExecutor: async () => ({ content: "ok", isError: false }),
4363
3899
  contextWindowManager: {
4364
3900
  shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
4365
3901
  maybeCompact: async () => ({ compacted: false }),
4366
3902
  } as unknown as AgentLoopConversationContext["contextWindowManager"],
4367
3903
  });
4368
3904
 
3905
+ // WHEN the orchestrator runs the turn to completion
4369
3906
  await runAgentLoopImpl(ctx, "hello", "msg-1", () => {});
4370
3907
 
4371
3908
  const stripCalls = setConversationHistoryStrippedAtMock.mock.calls.filter(
@@ -4390,59 +3927,24 @@ describe("session-agent-loop", () => {
4390
3927
  estimatedTokens: 5000,
4391
3928
  });
4392
3929
 
4393
- let callCount = 0;
4394
- const agentLoopRun: AgentLoopRun = async (messages, onEvent) => {
4395
- callCount++;
4396
- // Prime the assistant row anchor — production code emits this from
4397
- // `AgentLoop.run` just before `provider.sendMessage`. Retry branches
4398
- // need this on every invocation: each agent-loop iteration reserves
4399
- // its own row.
4400
- await onEvent({ type: "llm_call_started" });
4401
- if (callCount === 1) {
4402
- onEvent({
4403
- type: "error",
4404
- error: new Error("context_length_exceeded"),
4405
- });
4406
- onEvent({
4407
- type: "usage",
4408
- inputTokens: 100,
4409
- outputTokens: 0,
4410
- model: "test-model",
4411
- providerDurationMs: 50,
4412
- });
4413
- return [
4414
- ...messages,
4415
- {
4416
- role: "assistant" as const,
4417
- content: [{ type: "text", text: "partial" }] as ContentBlock[],
4418
- },
4419
- ];
4420
- }
4421
- onEvent({
4422
- type: "message_complete",
4423
- message: {
4424
- role: "assistant",
4425
- content: [{ type: "text", text: "recovered" }],
4426
- },
4427
- });
4428
- onEvent({
4429
- type: "usage",
4430
- inputTokens: 50,
4431
- outputTokens: 25,
4432
- model: "test-model",
4433
- providerDurationMs: 100,
4434
- });
4435
- return [
4436
- ...messages,
3930
+ // GIVEN a real loop that appends a tool turn and then rejects with a
3931
+ // context-too-large error on the following call, driving the
3932
+ // convergence strip whose marker-write helper is stubbed to throw,
3933
+ // before a final call recovers.
3934
+ const ctx = makeCtx({
3935
+ providerResponses: [
3936
+ toolUseResponse("t1", "file_read", {}),
3937
+ new Error("context_length_exceeded"),
3938
+ textResponse("recovered"),
3939
+ ],
3940
+ loopTools: [
4437
3941
  {
4438
- role: "assistant" as const,
4439
- content: [{ type: "text", text: "recovered" }] as ContentBlock[],
3942
+ name: "file_read",
3943
+ description: "Read a file",
3944
+ input_schema: { type: "object", properties: {} },
4440
3945
  },
4441
- ];
4442
- };
4443
-
4444
- const ctx = makeCtx({
4445
- agentLoopRun,
3946
+ ],
3947
+ toolExecutor: async () => ({ content: "ok", isError: false }),
4446
3948
  contextWindowManager: {
4447
3949
  shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
4448
3950
  maybeCompact: async () => ({ compacted: false }),