@vellumai/assistant 0.8.10 → 0.8.11-staging.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (400) hide show
  1. package/bun.lock +62 -1
  2. package/docs/workspace-tools.md +196 -0
  3. package/examples/plugins/echo/README.md +3 -3
  4. package/knip.json +1 -0
  5. package/openapi.yaml +460 -128
  6. package/package.json +2 -1
  7. package/scripts/build-plugin-api.ts +299 -0
  8. package/src/__tests__/agent-loop-callsite-precedence.test.ts +7 -0
  9. package/src/__tests__/agent-loop-compaction-events.test.ts +197 -0
  10. package/src/__tests__/agent-loop-exit-reason.test.ts +93 -96
  11. package/src/__tests__/agent-loop-mutable-latest-user-message.test.ts +2 -0
  12. package/src/__tests__/agent-loop-output-hooks.test.ts +274 -1
  13. package/src/__tests__/agent-loop-override-profile.test.ts +3 -0
  14. package/src/__tests__/agent-loop-provider-error-recording.test.ts +4 -0
  15. package/src/__tests__/agent-loop-thinking.test.ts +4 -0
  16. package/src/__tests__/agent-loop.test.ts +578 -5
  17. package/src/__tests__/agent-wake-disk-pressure-callsite.test.ts +0 -1
  18. package/src/__tests__/approval-cascade.test.ts +1 -0
  19. package/src/__tests__/background-workers-disk-pressure.test.ts +0 -2
  20. package/src/__tests__/btw-routes.test.ts +0 -1
  21. package/src/__tests__/build-persisted-content.test.ts +75 -1
  22. package/src/__tests__/catalog-install-normalize.test.ts +141 -0
  23. package/src/__tests__/ces-startup-timeout.test.ts +60 -0
  24. package/src/__tests__/compaction-events.test.ts +1 -0
  25. package/src/__tests__/config-managed-gemini-defaults.test.ts +2 -46
  26. package/src/__tests__/context-overflow-reducer.test.ts +264 -124
  27. package/src/__tests__/context-window-manager-overflow-rung.test.ts +351 -0
  28. package/src/__tests__/conversation-abort-tool-results.test.ts +1 -1
  29. package/src/__tests__/conversation-agent-loop-inference-profile.test.ts +13 -5
  30. package/src/__tests__/conversation-agent-loop-overflow.test.ts +284 -455
  31. package/src/__tests__/conversation-agent-loop.test.ts +131 -551
  32. package/src/__tests__/conversation-app-control-instantiation.test.ts +13 -0
  33. package/src/__tests__/conversation-confirmation-signals.test.ts +1 -0
  34. package/src/__tests__/conversation-fork-crud.test.ts +259 -0
  35. package/src/__tests__/conversation-history-web-search.test.ts +1 -1
  36. package/src/__tests__/conversation-lifecycle.test.ts +257 -1
  37. package/src/__tests__/conversation-process-callsite.test.ts +1 -0
  38. package/src/__tests__/conversation-provider-retry-repair.test.ts +38 -377
  39. package/src/__tests__/conversation-queue.test.ts +1 -39
  40. package/src/__tests__/conversation-runtime-assembly.test.ts +119 -8
  41. package/src/__tests__/conversation-skill-tools.test.ts +491 -5
  42. package/src/__tests__/conversation-slash-queue.test.ts +1 -1
  43. package/src/__tests__/conversation-slash-unknown.test.ts +1 -0
  44. package/src/__tests__/conversation-speed-override.test.ts +1 -0
  45. package/src/__tests__/conversation-store.test.ts +74 -0
  46. package/src/__tests__/conversation-surfaces-app-control.test.ts +4 -1
  47. package/src/__tests__/conversation-tool-setup-attribution.test.ts +323 -0
  48. package/src/__tests__/conversation-tool-setup-tools-disabled.test.ts +34 -0
  49. package/src/__tests__/conversation-workspace-cache-state.test.ts +1 -0
  50. package/src/__tests__/conversation-workspace-injection.test.ts +1 -1
  51. package/src/__tests__/conversation-workspace-tool-tracking.test.ts +1 -0
  52. package/src/__tests__/corrected-target.test.ts +93 -0
  53. package/src/__tests__/credential-execution-feature-gates.test.ts +3 -5
  54. package/src/__tests__/credential-execution-tools.test.ts +23 -11
  55. package/src/__tests__/credential-security-invariants.test.ts +6 -1
  56. package/src/__tests__/db-schedule-syntax-migration.test.ts +80 -0
  57. package/src/__tests__/device-id.test.ts +70 -1
  58. package/src/__tests__/embedding-managed-proxy-selection.test.ts +6 -40
  59. package/src/__tests__/empty-response-hook.test.ts +242 -66
  60. package/src/__tests__/external-plugin-loader.test.ts +0 -31
  61. package/src/__tests__/get-skill-detail-audit.test.ts +43 -1
  62. package/src/__tests__/guardian-routing-invariants.test.ts +91 -0
  63. package/src/__tests__/history-repair-hook.test.ts +228 -3
  64. package/src/__tests__/host-app-control-proxy.test.ts +45 -0
  65. package/src/__tests__/host-browser-proxy.test.ts +254 -9
  66. package/src/__tests__/identity-routes.test.ts +1 -0
  67. package/src/__tests__/image-recovery-hook.test.ts +387 -0
  68. package/src/__tests__/injector-chain.test.ts +5 -4
  69. package/src/__tests__/injector-v3-suppression.test.ts +373 -47
  70. package/src/__tests__/intent-routing.test.ts +7 -0
  71. package/src/__tests__/memory-retrieval-hook.test.ts +117 -15
  72. package/src/__tests__/notification-decision-strategy.test.ts +3 -3
  73. package/src/__tests__/oauth-store.test.ts +0 -85
  74. package/src/__tests__/{context-overflow-policy.test.ts → overflow-policy.test.ts} +1 -1
  75. package/src/__tests__/parallel-tool.benchmark.test.ts +4 -0
  76. package/src/__tests__/persist-unsendable-image-downscale.test.ts +29 -9
  77. package/src/__tests__/persist-unsendable-image.test.ts +4 -4
  78. package/src/__tests__/persistence-secret-redaction.test.ts +78 -0
  79. package/src/__tests__/plugin-bootstrap.test.ts +82 -73
  80. package/src/__tests__/plugin-tool-contribution.test.ts +7 -4
  81. package/src/__tests__/plugin-types.test.ts +0 -8
  82. package/src/__tests__/provider-catalog-visibility.test.ts +1 -9
  83. package/src/__tests__/prune-old-conversations-job.test.ts +99 -0
  84. package/src/__tests__/registry.test.ts +240 -1
  85. package/src/__tests__/require-fresh-approval.test.ts +3 -0
  86. package/src/__tests__/schedule-routes.test.ts +116 -1
  87. package/src/__tests__/schedule-store.test.ts +28 -0
  88. package/src/__tests__/schedule-tools.test.ts +94 -1
  89. package/src/__tests__/server-history-render.test.ts +39 -0
  90. package/src/__tests__/skill-projection-feature-flag.test.ts +13 -0
  91. package/src/__tests__/skill-projection.benchmark.test.ts +25 -7
  92. package/src/__tests__/skills.test.ts +202 -0
  93. package/src/__tests__/slim-skill-category.test.ts +195 -0
  94. package/src/__tests__/strip-memory-injections.test.ts +33 -39
  95. package/src/__tests__/test-support/tool-invocation-seed.ts +79 -0
  96. package/src/__tests__/title-generate-hook.test.ts +9 -7
  97. package/src/__tests__/tool-audit-listener.test.ts +264 -1
  98. package/src/__tests__/tool-error-hook.test.ts +4 -3
  99. package/src/__tests__/tool-execution-pipeline.benchmark.test.ts +1 -0
  100. package/src/__tests__/tool-executor-lifecycle-events.test.ts +273 -0
  101. package/src/__tests__/tool-result-truncate-hook.test.ts +1 -0
  102. package/src/__tests__/tool-start-timestamp.test.ts +218 -0
  103. package/src/__tests__/tools-get-route.test.ts +202 -0
  104. package/src/__tests__/workspace-tool-loader.test.ts +319 -0
  105. package/src/__tests__/workspace-tools-watcher-flag.test.ts +70 -0
  106. package/src/agent/loop.ts +569 -319
  107. package/src/api/events/tool-result.ts +9 -0
  108. package/src/api/events/tool-use-start.ts +7 -0
  109. package/src/api/index.ts +10 -0
  110. package/src/api/responses/conversation-message.ts +135 -27
  111. package/src/api/responses/memory-v3-selection-log.ts +4 -4
  112. package/src/approvals/guardian-request-resolvers.ts +26 -0
  113. package/src/browser-session/backends/host-bridge.ts +29 -0
  114. package/src/browser-session/index.ts +1 -0
  115. package/src/browser-session/types.ts +5 -1
  116. package/src/cli/commands/__tests__/schedules.test.ts +62 -4
  117. package/src/cli/commands/__tests__/skills.test.ts +53 -0
  118. package/src/cli/commands/channel-verification-sessions.ts +6 -6
  119. package/src/cli/commands/inference-providers.ts +0 -8
  120. package/src/cli/commands/plugins.ts +2 -2
  121. package/src/cli/commands/schedules.ts +27 -4
  122. package/src/cli/commands/skills.ts +187 -146
  123. package/src/cli/commands/tools.ts +106 -0
  124. package/src/cli/lib/__tests__/install-from-github.test.ts +256 -328
  125. package/src/cli/lib/__tests__/plugin-catalog-cache.test.ts +6 -2
  126. package/src/cli/lib/__tests__/plugin-details.test.ts +10 -16
  127. package/src/cli/lib/__tests__/plugin-marketplace.test.ts +2 -2
  128. package/src/cli/lib/__tests__/search-plugins.test.ts +145 -240
  129. package/src/cli/lib/install-from-github.ts +187 -117
  130. package/src/cli/lib/plugin-catalog-cache.ts +9 -9
  131. package/src/cli/lib/plugin-details.ts +38 -68
  132. package/src/cli/lib/plugin-marketplace.ts +42 -14
  133. package/src/cli/lib/search-plugins.ts +29 -129
  134. package/src/cli/program.ts +2 -0
  135. package/src/config/bundled-skills/acp/SKILL.md +1 -0
  136. package/src/config/bundled-skills/app-builder/SKILL.md +1 -0
  137. package/src/config/bundled-skills/app-control/SKILL.md +1 -0
  138. package/src/config/bundled-skills/computer-use/SKILL.md +1 -0
  139. package/src/config/bundled-skills/contacts/SKILL.md +1 -0
  140. package/src/config/bundled-skills/document-editor/SKILL.md +1 -0
  141. package/src/config/bundled-skills/followups/SKILL.md +1 -0
  142. package/src/config/bundled-skills/image-studio/SKILL.md +1 -0
  143. package/src/config/bundled-skills/media-processing/SKILL.md +1 -0
  144. package/src/config/bundled-skills/messaging/SKILL.md +1 -0
  145. package/src/config/bundled-skills/phone-calls/SKILL.md +1 -0
  146. package/src/config/bundled-skills/playbooks/SKILL.md +1 -0
  147. package/src/config/bundled-skills/schedule/SKILL.md +1 -0
  148. package/src/config/bundled-skills/schedule/TOOLS.json +11 -3
  149. package/src/config/bundled-skills/sequences/SKILL.md +1 -0
  150. package/src/config/bundled-skills/settings/SKILL.md +1 -0
  151. package/src/config/bundled-skills/skill-management/SKILL.md +101 -1
  152. package/src/config/bundled-skills/subagent/SKILL.md +1 -0
  153. package/src/config/bundled-skills/transcribe/SKILL.md +1 -0
  154. package/src/config/env-registry.ts +23 -0
  155. package/src/config/feature-flag-registry.json +17 -81
  156. package/src/config/loader.ts +5 -22
  157. package/src/config/schema.ts +2 -0
  158. package/src/config/schemas/__tests__/compaction-logs.test.ts +56 -0
  159. package/src/config/schemas/__tests__/memory-v2.test.ts +0 -1
  160. package/src/config/schemas/__tests__/memory-v3.test.ts +61 -1
  161. package/src/config/schemas/compaction-logs.ts +79 -0
  162. package/src/config/schemas/memory-v2.ts +0 -8
  163. package/src/config/schemas/memory-v3.ts +104 -33
  164. package/src/config/seed-inference-profiles.ts +1 -1
  165. package/src/config/skills.ts +117 -47
  166. package/src/context/compactor.ts +11 -0
  167. package/src/context/strip-injections.ts +38 -4
  168. package/src/credential-execution/feature-gates.ts +0 -21
  169. package/src/credential-execution/startup-timeout.ts +32 -4
  170. package/src/daemon/__tests__/conversation-tool-setup-exclude.test.ts +18 -0
  171. package/src/daemon/conversation-agent-loop-handlers.ts +140 -91
  172. package/src/daemon/conversation-agent-loop.ts +123 -658
  173. package/src/daemon/conversation-error.ts +6 -33
  174. package/src/daemon/conversation-lifecycle.ts +1 -1
  175. package/src/daemon/conversation-runtime-assembly.ts +183 -22
  176. package/src/daemon/conversation-skill-tools.ts +137 -8
  177. package/src/daemon/conversation-slash.ts +0 -14
  178. package/src/daemon/conversation-store.ts +2 -19
  179. package/src/daemon/conversation-tool-setup.ts +87 -1
  180. package/src/daemon/conversation.ts +119 -50
  181. package/src/daemon/external-plugins-bootstrap.ts +36 -95
  182. package/src/daemon/handlers/config-channels.ts +11 -2
  183. package/src/daemon/handlers/shared.ts +9 -1
  184. package/src/daemon/handlers/skills.ts +10 -4
  185. package/src/daemon/host-app-control-proxy.ts +72 -57
  186. package/src/daemon/host-browser-proxy.ts +117 -22
  187. package/src/daemon/lifecycle.ts +34 -5
  188. package/src/daemon/message-protocol.ts +0 -7
  189. package/src/daemon/message-types/schedules.ts +1 -0
  190. package/src/daemon/message-types/skills.ts +17 -0
  191. package/src/daemon/providers-setup.ts +3 -0
  192. package/src/daemon/server.ts +3 -3
  193. package/src/daemon/tool-setup-types.ts +9 -3
  194. package/src/daemon/trust-context.ts +23 -0
  195. package/src/daemon/workspace-tools-watcher.ts +324 -0
  196. package/src/events/tool-audit-listener.ts +78 -16
  197. package/src/events/tool-metrics-listener.ts +2 -5
  198. package/src/memory/__tests__/compaction-log-writer-clickhouse.test.ts +227 -0
  199. package/src/memory/__tests__/conversation-queries.test.ts +176 -0
  200. package/src/memory/__tests__/jobs-worker-v2-schedule.test.ts +20 -32
  201. package/src/memory/compaction-log-writer-clickhouse.ts +418 -0
  202. package/src/memory/conversation-crud.ts +134 -3
  203. package/src/memory/conversation-queries.ts +64 -7
  204. package/src/memory/db-init.ts +14 -0
  205. package/src/memory/embedding-backend.test.ts +130 -1
  206. package/src/memory/embedding-backend.ts +79 -106
  207. package/src/memory/embedding-gemini.ts +5 -0
  208. package/src/memory/graph/__tests__/conversation-graph-memory-v2-routing.test.ts +12 -0
  209. package/src/memory/graph/__tests__/handle-remember-v2.test.ts +19 -0
  210. package/src/memory/graph/conversation-graph-memory.ts +36 -25
  211. package/src/memory/graph/tool-handlers.ts +3 -0
  212. package/src/memory/job-handlers/cleanup.ts +3 -1
  213. package/src/memory/jobs-store.ts +0 -28
  214. package/src/memory/jobs-worker.ts +10 -22
  215. package/src/memory/memory-marker.ts +29 -0
  216. package/src/memory/memory-retrospective-startup-cleanup.ts +1 -1
  217. package/src/memory/migrations/268-add-memory-v3-selections.ts +6 -0
  218. package/src/memory/migrations/270-schedule-description.ts +36 -0
  219. package/src/memory/migrations/275-tool-invocations-add-skill-id.test.ts +81 -0
  220. package/src/memory/migrations/275-tool-invocations-add-skill-id.ts +20 -0
  221. package/src/memory/migrations/276-tool-invocations-created-at-id-index.test.ts +68 -0
  222. package/src/memory/migrations/276-tool-invocations-created-at-id-index.ts +20 -0
  223. package/src/memory/migrations/277-add-memory-v3-ever-injected.ts +29 -0
  224. package/src/memory/migrations/278-tool-invocations-telemetry-columns.test.ts +96 -0
  225. package/src/memory/migrations/278-tool-invocations-telemetry-columns.ts +39 -0
  226. package/src/memory/migrations/279-create-skill-loaded-events.test.ts +84 -0
  227. package/src/memory/migrations/279-create-skill-loaded-events.ts +26 -0
  228. package/src/memory/migrations/280-conversations-surfaced-at.test.ts +88 -0
  229. package/src/memory/migrations/280-conversations-surfaced-at.ts +24 -0
  230. package/src/memory/migrations/index.ts +10 -0
  231. package/src/memory/migrations/registry.ts +8 -0
  232. package/src/memory/schema/conversations.ts +16 -0
  233. package/src/memory/schema/infrastructure.ts +26 -0
  234. package/src/memory/skill-loaded-events-store.test.ts +160 -0
  235. package/src/memory/skill-loaded-events-store.ts +95 -0
  236. package/src/memory/tool-executed-events-store.test.ts +219 -0
  237. package/src/memory/tool-executed-events-store.ts +102 -0
  238. package/src/memory/tool-usage-store.ts +15 -3
  239. package/src/memory/v2/__tests__/consolidation-job.test.ts +117 -12
  240. package/src/memory/v2/__tests__/consolidation-prompt-flag-gating-guard.test.ts +189 -0
  241. package/src/memory/v2/__tests__/injected-block-slugs.test.ts +90 -0
  242. package/src/memory/v2/__tests__/page-store.test.ts +33 -0
  243. package/src/memory/v2/__tests__/prompts-consolidation.test.ts +88 -15
  244. package/src/memory/v2/activation-store.ts +50 -1
  245. package/src/memory/v2/consolidation-job.ts +113 -29
  246. package/src/memory/v2/injected-block-slugs.ts +79 -0
  247. package/src/memory/v2/injection.ts +6 -1
  248. package/src/memory/v2/prompts/consolidation.ts +414 -13
  249. package/src/memory/v2/router.ts +2 -28
  250. package/src/memory/v2/static-context.ts +1 -1
  251. package/src/memory/v2/types.ts +16 -0
  252. package/src/notifications/__tests__/copy-composer.test.ts +244 -0
  253. package/src/notifications/access-request-copy.ts +298 -0
  254. package/src/notifications/adapters/slack.ts +3 -3
  255. package/src/notifications/adapters/telegram.ts +2 -1
  256. package/src/notifications/copy-composer.ts +49 -267
  257. package/src/notifications/decision-engine.ts +16 -35
  258. package/src/notifications/home-feed-side-effect.ts +1 -6
  259. package/src/oauth/oauth-store.ts +0 -9
  260. package/src/permissions/checker.test.ts +83 -1
  261. package/src/permissions/checker.ts +25 -2
  262. package/src/permissions/gateway-threshold-reader.test.ts +182 -0
  263. package/src/permissions/gateway-threshold-reader.ts +80 -0
  264. package/src/platform/client.ts +1 -3
  265. package/src/platform/feature-gate.ts +3 -12
  266. package/src/plugin-api/constants.ts +4 -2
  267. package/src/plugin-api/index.ts +63 -11
  268. package/src/plugin-api/types.ts +236 -71
  269. package/src/plugins/defaults/compaction/compact.ts +66 -2
  270. package/src/plugins/defaults/compaction/context-overflow-reducer.ts +240 -32
  271. package/src/plugins/defaults/compaction/corrected-target.ts +53 -0
  272. package/src/{daemon/context-overflow-policy.ts → plugins/defaults/compaction/overflow-policy.ts} +1 -1
  273. package/src/plugins/defaults/compaction/window-manager.ts +303 -1
  274. package/src/plugins/defaults/empty-response/hooks/post-model-call.ts +173 -0
  275. package/src/plugins/defaults/empty-response/hooks/stop.ts +11 -115
  276. package/src/plugins/defaults/empty-response/nudge-state-store.ts +46 -0
  277. package/src/plugins/defaults/history-repair/hooks/post-model-call.ts +50 -0
  278. package/src/plugins/defaults/history-repair/hooks/stop.ts +22 -0
  279. package/src/plugins/defaults/history-repair/repair-state-store.ts +51 -0
  280. package/src/plugins/defaults/history-repair/terminal.ts +39 -2
  281. package/src/plugins/defaults/image-recovery/detect.ts +25 -0
  282. package/src/plugins/defaults/image-recovery/hooks/post-model-call.ts +73 -0
  283. package/src/plugins/defaults/image-recovery/hooks/stop.ts +22 -0
  284. package/src/plugins/defaults/image-recovery/image-recovery-state-store.ts +48 -0
  285. package/src/plugins/defaults/image-recovery/package.json +14 -0
  286. package/src/{daemon/persist-unsendable-image.ts → plugins/defaults/image-recovery/recover.ts} +67 -14
  287. package/src/plugins/defaults/index.ts +71 -5
  288. package/src/plugins/defaults/memory-retrieval/hooks/post-compact.ts +76 -112
  289. package/src/plugins/defaults/memory-retrieval/hooks/{user-prompt-submit-temp.ts → user-prompt-submit.ts} +83 -74
  290. package/src/plugins/defaults/memory-retrieval/injector-chain.ts +14 -8
  291. package/src/plugins/defaults/memory-retrieval/injectors.ts +2 -18
  292. package/src/plugins/defaults/memory-retrieval/package.json +14 -0
  293. package/src/plugins/defaults/memory-v3-shadow/__tests__/carry-integration.test.ts +1157 -0
  294. package/src/plugins/defaults/memory-v3-shadow/__tests__/injection.test.ts +683 -0
  295. package/src/plugins/defaults/memory-v3-shadow/__tests__/live-integration.test.ts +161 -140
  296. package/src/plugins/defaults/memory-v3-shadow/__tests__/maintain-job.test.ts +160 -0
  297. package/src/plugins/defaults/memory-v3-shadow/__tests__/orchestrate.test.ts +335 -316
  298. package/src/plugins/defaults/memory-v3-shadow/__tests__/pool-select.test.ts +145 -53
  299. package/src/plugins/defaults/memory-v3-shadow/__tests__/render-injection.test.ts +38 -1
  300. package/src/plugins/defaults/memory-v3-shadow/__tests__/selection-log-store.test.ts +18 -8
  301. package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-integration.test.ts +112 -71
  302. package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-plugin.test.ts +192 -51
  303. package/src/plugins/defaults/memory-v3-shadow/__tests__/types.test.ts +4 -16
  304. package/src/plugins/defaults/memory-v3-shadow/card.test.ts +173 -0
  305. package/src/plugins/defaults/memory-v3-shadow/card.ts +116 -0
  306. package/src/plugins/defaults/memory-v3-shadow/core-set.test.ts +104 -0
  307. package/src/plugins/defaults/memory-v3-shadow/core-set.ts +59 -0
  308. package/src/plugins/defaults/memory-v3-shadow/ever-injected-store.test.ts +305 -0
  309. package/src/plugins/defaults/memory-v3-shadow/ever-injected-store.ts +278 -0
  310. package/src/plugins/defaults/memory-v3-shadow/hot-set.test.ts +138 -0
  311. package/src/plugins/defaults/memory-v3-shadow/hot-set.ts +85 -0
  312. package/src/plugins/defaults/memory-v3-shadow/injector.ts +331 -24
  313. package/src/plugins/defaults/memory-v3-shadow/maintain-job.ts +119 -13
  314. package/src/plugins/defaults/memory-v3-shadow/orchestrate.ts +169 -114
  315. package/src/plugins/defaults/memory-v3-shadow/page-content.ts +47 -16
  316. package/src/plugins/defaults/memory-v3-shadow/pool-select.ts +144 -66
  317. package/src/plugins/defaults/memory-v3-shadow/prune.test.ts +758 -0
  318. package/src/plugins/defaults/memory-v3-shadow/prune.ts +471 -0
  319. package/src/plugins/defaults/memory-v3-shadow/render-injection.ts +68 -16
  320. package/src/plugins/defaults/memory-v3-shadow/selection-log-store.ts +19 -12
  321. package/src/plugins/defaults/memory-v3-shadow/shadow-plugin.ts +96 -43
  322. package/src/plugins/defaults/memory-v3-shadow/types.ts +34 -17
  323. package/src/plugins/defaults/title-generate/hooks/stop.ts +9 -11
  324. package/src/plugins/pipeline.ts +8 -5
  325. package/src/plugins/types.ts +5 -51
  326. package/src/providers/cache-control.ts +26 -0
  327. package/src/providers/inference/__tests__/base-url-route-validation.test.ts +1 -2
  328. package/src/providers/model-catalog.ts +13 -1
  329. package/src/providers/openai/__tests__/tool-choice-mapping.test.ts +147 -0
  330. package/src/providers/openai/chat-completions-provider.ts +46 -0
  331. package/src/providers/openai/responses-provider.ts +45 -0
  332. package/src/providers/registry.ts +0 -8
  333. package/src/runtime/__tests__/agent-wake.test.ts +0 -1
  334. package/src/runtime/agent-wake.ts +13 -0
  335. package/src/runtime/routes/__tests__/acp-routes.test.ts +151 -0
  336. package/src/runtime/routes/__tests__/consolidation-routes.test.ts +12 -50
  337. package/src/runtime/routes/__tests__/conversation-surface-routes.test.ts +322 -0
  338. package/src/runtime/routes/__tests__/inference-provider-connection-routes.test.ts +0 -62
  339. package/src/runtime/routes/__tests__/plugins-routes.test.ts +31 -11
  340. package/src/runtime/routes/acp-routes.test.ts +106 -0
  341. package/src/runtime/routes/acp-routes.ts +248 -2
  342. package/src/runtime/routes/browser-tabs-routes.ts +1 -1
  343. package/src/runtime/routes/channel-verification-routes.ts +14 -5
  344. package/src/runtime/routes/chatgpt-subscription-auth-routes.ts +0 -14
  345. package/src/runtime/routes/consolidation-routes.ts +6 -82
  346. package/src/runtime/routes/conversation-list-routes.ts +6 -0
  347. package/src/runtime/routes/conversation-management-routes.ts +71 -0
  348. package/src/runtime/routes/identity-routes.ts +8 -0
  349. package/src/runtime/routes/inbound-stages/acl-enforcement.ts +33 -34
  350. package/src/runtime/routes/inference-provider-connection-routes.ts +0 -45
  351. package/src/runtime/routes/plugins-routes.ts +34 -45
  352. package/src/runtime/routes/schedule-routes.ts +43 -5
  353. package/src/runtime/routes/settings-routes.ts +140 -15
  354. package/src/runtime/routes/skills-routes.ts +18 -6
  355. package/src/runtime/services/__tests__/conversation-serializer.test.ts +140 -0
  356. package/src/runtime/services/conversation-serializer.ts +38 -1
  357. package/src/runtime/verification-outbound-actions.ts +147 -2
  358. package/src/runtime/verification-templates.ts +29 -3
  359. package/src/schedule/schedule-store.ts +19 -0
  360. package/src/skills/catalog-install.ts +77 -13
  361. package/src/tasks/task-scheduler.ts +1 -0
  362. package/src/telemetry/types.ts +66 -1
  363. package/src/telemetry/usage-telemetry-reporter.test.ts +542 -13
  364. package/src/telemetry/usage-telemetry-reporter.ts +213 -20
  365. package/src/tools/browser/__tests__/browser-execution-acquire.test.ts +49 -2
  366. package/src/tools/browser/__tests__/browser-status.test.ts +29 -5
  367. package/src/tools/browser/browser-execution.ts +27 -9
  368. package/src/tools/browser/cdp-client/__tests__/factory.test.ts +380 -4
  369. package/src/tools/browser/cdp-client/__tests__/host-bridge-cdp-client.test.ts +107 -0
  370. package/src/tools/browser/cdp-client/__tests__/types.test.ts +6 -1
  371. package/src/tools/browser/cdp-client/factory.ts +217 -17
  372. package/src/tools/browser/cdp-client/host-bridge-cdp-client.ts +67 -0
  373. package/src/tools/browser/cdp-client/types.ts +22 -2
  374. package/src/tools/credential-execution/make-authenticated-request.ts +2 -1
  375. package/src/tools/credential-execution/manage-secure-command-tool.ts +169 -164
  376. package/src/tools/credential-execution/run-authenticated-command.ts +2 -1
  377. package/src/tools/executor.ts +39 -7
  378. package/src/tools/registry.ts +387 -5
  379. package/src/tools/schedule/create.ts +16 -0
  380. package/src/tools/schedule/list.ts +12 -4
  381. package/src/tools/schedule/update.ts +12 -0
  382. package/src/tools/skills/load.ts +11 -6
  383. package/src/tools/terminal/safe-env.ts +2 -0
  384. package/src/tools/types.ts +65 -9
  385. package/src/tools/workspace-tools/loader.ts +673 -0
  386. package/src/usage/attribution.ts +28 -0
  387. package/src/util/device-id.ts +17 -3
  388. package/src/util/platform.ts +16 -0
  389. package/tsconfig.plugin-api.json +13 -0
  390. package/src/__tests__/plugin-external-api.test.ts +0 -68
  391. package/src/__tests__/plugin-skill-contribution.test.ts +0 -355
  392. package/src/daemon/message-types/browser.ts +0 -10
  393. package/src/notifications/__tests__/emit-signal-home-feed.test.ts +0 -187
  394. package/src/plugins/defaults/memory-v3-shadow/__tests__/fixtures/eval-turns.json +0 -36
  395. package/src/plugins/defaults/memory-v3-shadow/__tests__/fixtures/live-turns.json +0 -37
  396. package/src/plugins/defaults/memory-v3-shadow/__tests__/working-set-eviction.test.ts +0 -106
  397. package/src/plugins/defaults/memory-v3-shadow/__tests__/working-set-skeleton.test.ts +0 -44
  398. package/src/plugins/defaults/memory-v3-shadow/working-set.ts +0 -91
  399. package/src/plugins/external-api.ts +0 -114
  400. package/src/plugins/plugin-skill-contributions.ts +0 -292
package/src/agent/loop.ts CHANGED
@@ -3,6 +3,7 @@ import * as Sentry from "@sentry/node";
3
3
  import { isAssistantFeatureFlagEnabled } from "../config/assistant-feature-flags.js";
4
4
  import { getConfig } from "../config/loader.js";
5
5
  import type { LLMCallSite } from "../config/schemas/llm.js";
6
+ import { recordEstimate } from "../context/estimator-calibration.js";
6
7
  import { stripInjectionsForCompaction } from "../context/strip-injections.js";
7
8
  import {
8
9
  estimatePromptTokensRaw,
@@ -10,20 +11,22 @@ import {
10
11
  estimateToolsTokens,
11
12
  getCalibrationProviderKey,
12
13
  } from "../context/token-estimator.js";
13
- import type { InboundActorContext } from "../daemon/conversation-runtime-assembly.js";
14
14
  import type { ToolActivityMetadata } from "../daemon/message-types/web-activity.js";
15
+ import { parseActualTokensFromError } from "../daemon/parse-actual-tokens-from-error.js";
15
16
  import type { TrustContext } from "../daemon/trust-context.js";
16
17
  import { stripHistoricalWebSearchResults } from "../daemon/web-search-history.js";
17
18
  import { HOOKS } from "../plugin-api/constants.js";
18
19
  import type {
20
+ AgentLoopExitReason,
21
+ PostCompactContext,
19
22
  PostModelCallContext,
23
+ PostModelCallDecision,
20
24
  PostToolUseContext,
21
25
  PreModelCallContext,
22
26
  StopContext,
23
27
  } from "../plugin-api/types.js";
24
28
  import { defaultCompact } from "../plugins/defaults/compaction/compact.js";
25
29
  import type { ContextWindowResult } from "../plugins/defaults/compaction/window-manager.js";
26
- import postCompact from "../plugins/defaults/memory-retrieval/hooks/post-compact.js";
27
30
  import { runHook } from "../plugins/pipeline.js";
28
31
  import type { CompactionCircuitEvent } from "../plugins/types.js";
29
32
  import { normalizeThinkingConfigForWire } from "../providers/thinking-config.js";
@@ -36,6 +39,7 @@ import type {
36
39
  ToolDefinition,
37
40
  ToolResultContent,
38
41
  } from "../providers/types.js";
42
+ import { isContextOverflowError } from "../providers/types.js";
39
43
  import type { SensitiveOutputBinding } from "../tools/sensitive-output-placeholders.js";
40
44
  import {
41
45
  applyStreamingSubstitution,
@@ -80,11 +84,11 @@ export interface CheckpointInfo {
80
84
 
81
85
  /**
82
86
  * Why a checkpoint paused the loop. Surfaced back to the caller via
83
- * {@link AgentLoopRunResult.exitReason} so the orchestrator reacts to
84
- * the loop's own signal (hand off to a queued message vs. compact and
85
- * re-enter) instead of the checkpoint callback mutating orchestrator state.
87
+ * {@link AgentLoopRunResult.exitReason} so the wrapper reacts to the loop's
88
+ * own signal (hand off to a queued message) instead of the checkpoint callback
89
+ * mutating wrapper state.
86
90
  */
87
- export type ExitReason = "handoff" | "budget";
91
+ export type ExitReason = "handoff";
88
92
 
89
93
  export type CheckpointDecision = "continue" | ExitReason;
90
94
 
@@ -97,13 +101,6 @@ export interface AgentLoopRunResult {
97
101
  * (completion, error, abort, or a tool-requested yield-to-user).
98
102
  */
99
103
  exitReason: ExitReason | null;
100
- /**
101
- * Whether the loop produced at least one new assistant message this run —
102
- * the forward-progress signal for the ordering-error retry gate and the
103
- * overflow convergence fold (immune to in-loop compaction shrinking history
104
- * below a pre-run length).
105
- */
106
- appendedNewMessages: boolean;
107
104
  /**
108
105
  * Slice of `history` appended this run, measured from the loop's input or
109
106
  * from the compacted base when it compacts in place. The loop owns this
@@ -113,54 +110,28 @@ export interface AgentLoopRunResult {
113
110
  }
114
111
 
115
112
  /**
116
- * Why an agent turn reached a terminal state.
117
- *
118
- * Emitted as part of an {@link AgentEvent} of type `agent_loop_exit`, then
119
- * persisted onto the **final** `llm_request_logs` row of the turn. Rows from
120
- * intermediate turns keep a NULL `agent_loop_exit_reason`, which is how
121
- * downstream tooling (and the LLM Context Inspector) distinguishes "loop kept
122
- * going" from "loop is done".
123
- *
124
- * Values are stable wire/DB strings — they are written to SQLite and
125
- * surfaced over the inspector wire format, so renaming any of them is a
126
- * breaking change.
127
- *
128
- * Keep in sync with `emitExit` call sites in {@link AgentLoop.run} and the
129
- * outer conversation orchestrator paths that terminate after a checkpoint
130
- * yield. A checkpoint yield used for budget compaction is intentionally not
131
- * a terminal reason — it is a control transfer before re-entering the loop.
113
+ * Outcome of an in-loop {@link AgentLoop.compact} call.
132
114
  */
133
- export type AgentLoopExitReason =
134
- /** `if (signal?.aborted) break;` at the top of the loop. */
135
- | "aborted_pre_call"
136
- /** Assistant message has no tool-use blocks (or no tool executor). */
137
- | "no_tool_calls"
138
- /** Signal aborted while building the user-side tool-results message. */
139
- | "aborted_post_response"
140
- /** Signal aborted mid-tool-execution; completed results were pushed. */
141
- | "aborted_during_tools"
142
- /** A tool result requested handing back to the user. */
143
- | "yield_to_user"
144
- /** The orchestrator yielded at checkpoint to process a queued message. */
145
- | "checkpoint_handoff"
146
- /** Context-window recovery exhausted and the turn ended with an error. */
147
- | "context_too_large"
115
+ interface CompactionAttempt {
148
116
  /**
149
- * Auto-compress rerun (post-emergency-compaction, post-tier reducer)
150
- * still yielded at the mid-loop budget checkpoint — the turn silently
151
- * terminated with no further recovery layer to re-enter. Pure
152
- * observability signal so the silent stall is attributable instead of
153
- * leaving `agent_loop_exit_reason` NULL.
117
+ * Re-injected history to continue from, or `null` when an ordinary forced
118
+ * compaction exhausted with nothing reduced worth continuing from. The
119
+ * overflow-recovery path always returns the reduction rung's history.
154
120
  */
155
- | "budget_yield_unrecovered"
156
- /** Provider stopped because the configured output-token limit was reached. */
157
- | "max_tokens_reached"
158
- /** User cancellation landed after a non-terminal checkpoint yield. */
159
- | "aborted_after_checkpoint"
160
- /** Signal aborted while the catch handler was synthesizing an error turn. */
161
- | "aborted_via_error"
162
- /** Catch-block fallback: an unhandled error broke the loop. */
163
- | "error";
121
+ history: Message[] | null;
122
+ /** Whether the overflow reduction ladder reported it is spent. */
123
+ exhausted: boolean;
124
+ /** Whether the ladder applied its terminal auto-compress-latest-turn rung. */
125
+ autoCompressApplied: boolean;
126
+ }
127
+
128
+ export type { AgentLoopExitReason };
129
+
130
+ /**
131
+ * Why a mid-loop compaction ran: `"budget"` for the proactive estimate gate,
132
+ * `"overflow"` for recovery from a provider context-overflow rejection.
133
+ */
134
+ export type CompactionTrigger = "budget" | "overflow";
164
135
 
165
136
  export type AgentEvent =
166
137
  /**
@@ -302,16 +273,38 @@ export type AgentEvent =
302
273
  * the mid-loop budget gate tripped. The daemon's event dispatcher
303
274
  * translates it into a "compacting context" activity state so clients
304
275
  * surface that the turn paused to summarize context.
276
+ *
277
+ * Carries the start-side half of the compaction record: everything the
278
+ * loop knows before handing the history to the compaction pipeline.
279
+ * The pipeline between this event and `compaction_completed` is
280
+ * plugin-owned, so consumers must treat the start/end pair (correlated
281
+ * by `compactionId`) as the complete picture of an attempt. A start
282
+ * event with no matching end means the pipeline threw or the turn
283
+ * aborted mid-compaction.
305
284
  */
306
285
  type: "context_compacting";
286
+ /** Correlates this start event with its `compaction_completed` pair. */
287
+ compactionId: string;
288
+ /** The turn's request id, linking the attempt to the triggering turn. */
289
+ requestId: string;
290
+ /**
291
+ * Why the loop compacted: `"budget"` when the proactive mid-loop
292
+ * estimate gate tripped, `"overflow"` when recovering from a provider
293
+ * context-overflow rejection via the reduction ladder.
294
+ */
295
+ trigger: CompactionTrigger;
296
+ /** Epoch ms when the loop began the compaction ceremony. */
297
+ startedAt: number;
298
+ /** The running history before injection stripping and compaction. */
299
+ messages: Message[];
307
300
  }
308
301
  | {
309
302
  /**
310
303
  * Emitted after the loop's inline mid-loop compaction pipeline runs,
311
304
  * immediately before re-injection — whether or not the pipeline actually
312
- * compacted. The daemon's event dispatcher always commits `basis` (the
305
+ * compacted. The daemon's event dispatcher always commits `messages` (the
313
306
  * stripped pre-compaction history) as the conversation's durable message
314
- * state, so re-injection ({@link postCompact}) re-applies
307
+ * state, so re-injection (the post-compaction hook) re-applies
315
308
  * injections onto the stripped base rather than stacking on top of the
316
309
  * still-injected messages. When `result.compacted` is set it
317
310
  * additionally commits the durable compaction result (DB-record fields,
@@ -321,13 +314,24 @@ export type AgentEvent =
321
314
  * Treated as a critical event: a failed durable commit re-throws so the
322
315
  * turn aborts rather than re-injecting against half-applied state.
323
316
  *
324
- * `basis` is the stripped pre-compaction history the summary was built
325
- * from; the dispatcher uses it to project Slack provenance onto the
326
- * compacted result.
317
+ * `messages` is the stripped pre-compaction history the summary was
318
+ * built from; the dispatcher uses it to project Slack provenance onto
319
+ * the compacted result.
327
320
  */
328
321
  type: "compaction_completed";
322
+ /** Correlates this end event with its `context_compacting` pair. */
323
+ compactionId: string;
324
+ /** The turn's request id, linking the attempt to the triggering turn. */
325
+ requestId: string;
326
+ /** Same trigger as the paired start event, duplicated so the end
327
+ * event is self-sufficient for consumers that only buffer ends. */
328
+ trigger: CompactionTrigger;
329
+ /** Epoch ms when the loop began the compaction ceremony. */
330
+ startedAt: number;
331
+ /** Epoch ms when the compaction pipeline returned. */
332
+ finishedAt: number;
329
333
  result: ContextWindowResult;
330
- basis: Message[];
334
+ messages: Message[];
331
335
  }
332
336
  | {
333
337
  /**
@@ -370,7 +374,19 @@ const DEFAULT_CONFIG: AgentLoopConfig = {
370
374
  minTurnIntervalMs: 150,
371
375
  };
372
376
 
373
- const MAX_STOP_CONTINUE_RETRIES = 1;
377
+ /**
378
+ * Per-run backstop on `post-model-call`-driven retries. A recovery hook that
379
+ * sets `decision: "continue"` re-issues the provider call; this bounds the
380
+ * total such re-issues across a run so a misbehaving hook can't spin forever.
381
+ *
382
+ * It is a backstop, not the primary guard: each recovery class owns a
383
+ * one-shot per-conversation bound that stops it repeating within a turn, so
384
+ * the legitimate ceiling is one continue per class (empty-response nudge,
385
+ * ordering repair, image downscale). This sits above that sum to leave
386
+ * headroom while still catching pathological alternation between classes.
387
+ */
388
+ const MAX_POST_MODEL_CALL_CONTINUES = 5;
389
+
374
390
  const MAX_TOKENS_STOP_REASONS = new Set([
375
391
  "length",
376
392
  "max_output_tokens",
@@ -398,6 +414,13 @@ function assistantTextOf(content: ReadonlyArray<ContentBlock>): string {
398
414
  return text;
399
415
  }
400
416
 
417
+ /** Whether `content` carries at least one non-empty `text` block. */
418
+ function hasVisibleText(content: ReadonlyArray<ContentBlock>): boolean {
419
+ return content.some(
420
+ (block) => block.type === "text" && block.text.trim().length > 0,
421
+ );
422
+ }
423
+
401
424
  /**
402
425
  * User-config HTTP status codes that should never page the on-call: billing
403
426
  * exhaustion (402), invalid credentials (401), and forbidden/plan-gated (403).
@@ -446,7 +469,7 @@ export interface AgentLoopRunOptions {
446
469
  /** Sink the loop streams its {@link AgentEvent}s through as the turn runs. */
447
470
  onEvent: (event: AgentEvent) => void | Promise<void>;
448
471
  signal?: AbortSignal;
449
- requestId?: string;
472
+ requestId: string;
450
473
  onCheckpoint?: (
451
474
  checkpoint: CheckpointInfo,
452
475
  ) => CheckpointDecision | Promise<CheckpointDecision>;
@@ -455,8 +478,8 @@ export interface AgentLoopRunOptions {
455
478
  * Trust classification and channel identity for the turn's inbound actor,
456
479
  * supplied by the caller as the turn-start snapshot. Read only on the
457
480
  * mid-loop in-place compaction path — to scope the compactor's image
458
- * manifest (guardian-only attachments are excluded for untrusted actors) and
459
- * forwarded to {@link postCompact}. Callers without a meaningful actor (agent
481
+ * manifest (guardian-only attachments are excluded for untrusted actors).
482
+ * Callers without a meaningful actor (agent
460
483
  * wakes, standalone unit tests) pass an `unknown`-class snapshot so the
461
484
  * compactor fail-closes to excluding guardian-only attachments.
462
485
  */
@@ -485,49 +508,37 @@ export interface AgentLoopRunOptions {
485
508
  /**
486
509
  * When `true`, the loop owns turn-start and mid-loop compaction. The pre-call
487
510
  * budget gate runs before the very first provider call — subsuming the
488
- * proactive turn-start compaction the orchestrator would otherwise perform
489
- * inline before `run()` — as well as before each tool-use re-entry. When the
490
- * gate trips it compacts the running history in place, re-applying runtime
491
- * injections via the default post-compaction hook ({@link postCompact}), and
492
- * continues instead of yielding `exitReason = "budget"`.
511
+ * proactive turn-start compaction the wrapper would otherwise perform inline
512
+ * before `run()` — as well as before each tool-use re-entry. When the gate
513
+ * trips it compacts the running history in place, re-applying runtime
514
+ * injections via the default post-compaction hook ({@link HOOKS.POST_COMPACT}),
515
+ * and continues with the call.
493
516
  *
494
517
  * The first-call pass honors the compaction circuit breaker and proceeds with
495
- * the call whether or not it compacted (preflight-overflow recovery and the
496
- * convergence loop remain the escalation path), so it never yields on the
497
- * first call. Reruns without an inline compaction path (agent wakes,
498
- * convergence/auto-compress reruns) leave it `false`: they skip the
499
- * first-call gate and keep yielding for budget on mid-loop re-entries.
500
- * Defaults to `false` when omitted.
518
+ * the call whether or not it compacted, so it never yields on the first call;
519
+ * a provider context-too-large rejection then drives the reactive recovery
520
+ * ladder from the catch. Reruns that carry no inline compaction path (the
521
+ * deep-repair and image-recovery retries) leave it `false` and skip the
522
+ * first-call gate. Defaults to `false` when omitted.
501
523
  */
502
524
  compactInPlace?: boolean;
503
525
  /**
504
526
  * Whether the in-flight turn has no human present to answer clarification
505
527
  * questions. Resolved once by the orchestrator at turn start and forwarded to
506
- * {@link postCompact} so post-compaction
528
+ * the post-compaction hook so post-compaction
507
529
  * re-injection uses the turn-start snapshot rather than re-reading mutable
508
530
  * client/headless state mid-turn. Defaults to `false` when omitted.
509
531
  */
510
532
  isNonInteractive?: boolean;
511
533
  /**
512
- * The `model_profile:` turn-context label resolved once by the orchestrator
513
- * at turn start, or `null` when the active inference profile is unchanged
514
- * since the last notified one. Forwarded to
515
- * {@link postCompact} so post-compaction re-injection
516
- * re-emits the turn-start value rather than re-deriving the change-detected
517
- * label (which flips once the notification is persisted mid-turn). Defaults to
518
- * `null` when omitted.
534
+ * The turn's resolved inference-profile key, or `null` when the active
535
+ * profile is unchanged since the last notified one. Forwarded to
536
+ * the post-compaction hook, which renders the `model_profile:` label from it so
537
+ * post-compaction re-injection re-emits the turn-start profile rather than
538
+ * re-deriving the change-detected value (which flips once the notification is
539
+ * persisted mid-turn). Defaults to `null` when omitted.
519
540
  */
520
- modelProfile?: string | null;
521
- /**
522
- * Inbound actor identity and trust fields for the unified `<turn_context>`
523
- * block, or `null` on guardian turns. Resolved once by the orchestrator at
524
- * turn start via the actor-trust resolver, whose contact/member registry
525
- * inputs can be mutated mid-turn by contact tools, and forwarded to
526
- * {@link postCompact} so post-compaction
527
- * re-injection re-emits the turn-start value rather than re-resolving it.
528
- * Defaults to `null` when omitted.
529
- */
530
- actorContext?: InboundActorContext | null;
541
+ modelProfileKey?: string | null;
531
542
  }
532
543
 
533
544
  /**
@@ -685,7 +696,7 @@ export class AgentLoop {
685
696
  * compaction outcome into a user-visible turn failure.
686
697
  */
687
698
  private async recordCompactionOutcome(
688
- requestId: string | undefined,
699
+ requestId: string,
689
700
  summaryFailed: boolean,
690
701
  onEvent: (event: AgentEvent) => void | Promise<void>,
691
702
  ): Promise<void> {
@@ -700,25 +711,42 @@ export class AgentLoop {
700
711
  }
701
712
 
702
713
  /**
703
- * Compact the running history in place when the mid-loop budget gate trips.
714
+ * Compact the running history in place when the budget gate trips.
704
715
  *
705
716
  * Calls the default compaction plugin on the stripped history, then
706
- * re-applies injections via the supplied hooks. Returns the history to
707
- * continue from, or `null` when the compactor exhausted its retry budget so
708
- * the caller yields `exitReason = "budget"` and the orchestrator escalates.
717
+ * re-applies injections via the supplied hooks. When `overflowSignal` is
718
+ * supplied the plugin routes through the manager's reduction ladder (which
719
+ * advances one rung per call and reports `exhausted` / `autoCompressApplied`
720
+ * / `injectionMode`); otherwise it runs ordinary forced compaction. Returns
721
+ * the re-injected history to continue from alongside the ladder's terminal
722
+ * state. On the ordinary path an exhausted compactor yields a `null` history
723
+ * (nothing reduced worth continuing from, so the caller proceeds with the
724
+ * call); the overflow path always returns the rung's reduced history so the
725
+ * call is retried once at maximum reduction before the turn ends.
709
726
  */
710
727
  private async compact(
711
728
  history: Message[],
712
- requestId: string | undefined,
729
+ requestId: string,
713
730
  trust: TrustContext,
714
731
  signal: AbortSignal | undefined,
715
732
  onEvent: (event: AgentEvent) => void | Promise<void>,
716
733
  overrideProfile: string | null,
717
734
  isNonInteractive: boolean,
718
- modelProfile: string | null,
719
- actorContext: InboundActorContext | null,
720
- ): Promise<Message[] | null> {
721
- await onEvent({ type: "context_compacting" });
735
+ modelProfileKey: string | null,
736
+ overflowSignal?: { actualTokens: number | null; isInteractive: boolean },
737
+ ): Promise<CompactionAttempt> {
738
+ const compactionId = crypto.randomUUID();
739
+ const startedAt = Date.now();
740
+ const trigger: CompactionTrigger =
741
+ overflowSignal != null ? "overflow" : "budget";
742
+ await onEvent({
743
+ type: "context_compacting",
744
+ compactionId,
745
+ requestId,
746
+ trigger,
747
+ startedAt,
748
+ messages: history,
749
+ });
722
750
  // Strip runtime injections so the compactor summarizes the raw persistent
723
751
  // messages.
724
752
  const rawHistory = stripInjectionsForCompaction(history);
@@ -727,12 +755,13 @@ export class AgentLoop {
727
755
  await onEvent({ type: "history_stripped" });
728
756
  // The compaction module owns the per-conversation manager; pass the
729
757
  // conversation id and let `defaultCompact` resolve it from the store.
730
- // The mid-loop budget gate is reached only when this turn decides to
731
- // compact in place, so `force` past the auto-threshold check.
732
- // `actorTrustClass` comes from the turn's trust snapshot (the actor whose
733
- // turn triggered compaction) so the compactor's image manifest excludes
734
- // guardian-only attachments for untrusted actors. `overrideProfile` is the
735
- // turn's resolved inference-profile override for the summary call.
758
+ // The budget gate is reached only when this turn decides to compact in
759
+ // place, so `force` past the auto-threshold check. `actorTrustClass` comes
760
+ // from the turn's trust snapshot (the actor whose turn triggered
761
+ // compaction) so the compactor's image manifest excludes guardian-only
762
+ // attachments for untrusted actors. `overrideProfile` is the turn's
763
+ // resolved inference-profile override for the summary call. `overflowSignal`
764
+ // routes the request through the reduction ladder when present.
736
765
  const compactResult = await defaultCompact({
737
766
  conversationId: this.conversationId,
738
767
  messages: rawHistory,
@@ -740,6 +769,7 @@ export class AgentLoop {
740
769
  force: true,
741
770
  actorTrustClass: trust.trustClass,
742
771
  overrideProfile,
772
+ overflowSignal,
743
773
  });
744
774
  // `force: true` bypasses the auto-threshold gate, but early returns
745
775
  // for "no eligible messages" / "insufficient messages" still leave
@@ -752,33 +782,53 @@ export class AgentLoop {
752
782
  onEvent,
753
783
  );
754
784
  }
755
- // Emit unconditionally: the dispatcher commits the stripped `basis` as the
785
+ // Emit unconditionally: the dispatcher commits the stripped `messages` as the
756
786
  // durable message base whether or not the pipeline compacted (re-injection
757
787
  // reads it), and runs the durable compaction commit only when
758
788
  // `result.compacted`.
759
789
  await onEvent({
760
790
  type: "compaction_completed",
791
+ compactionId,
792
+ requestId,
793
+ trigger,
794
+ startedAt,
795
+ finishedAt: Date.now(),
761
796
  result: compactResult,
762
- basis: rawHistory,
797
+ messages: rawHistory,
763
798
  });
764
- if (compactResult.exhausted ?? false) {
765
- return null;
799
+ const exhausted = compactResult.exhausted ?? false;
800
+ const autoCompressApplied = compactResult.autoCompressApplied ?? false;
801
+ if (overflowSignal == null && exhausted) {
802
+ return { history: null, exhausted, autoCompressApplied };
766
803
  }
767
- // Re-inject onto the same base the `compaction_completed` dispatch commits:
768
- // the compacted messages when the pipeline compacted, the stripped
769
- // pre-compaction history otherwise.
770
- const injection = await postCompact({
771
- history: compactResult.compacted ? compactResult.messages : rawHistory,
804
+ // Re-inject onto the same base the `compaction_completed` dispatch commits.
805
+ // The overflow ladder transforms the history on every rung (truncation /
806
+ // media stubbing / injection downgrade) regardless of whether the summary
807
+ // ran, so continue from its reduced messages; the ordinary path continues
808
+ // from the compacted messages only when the pipeline actually compacted.
809
+ const base =
810
+ overflowSignal != null || compactResult.compacted
811
+ ? compactResult.messages
812
+ : rawHistory;
813
+ const postCompactCtx: PostCompactContext = {
814
+ history: base,
772
815
  requestId,
773
816
  conversationId: this.conversationId,
774
- trust,
775
817
  isNonInteractive,
776
- // Mid-loop re-injection always runs at full injection volume.
777
- mode: "full",
778
- modelProfile,
779
- actorContext,
780
- });
781
- return injection.messages;
818
+ modelProfileKey,
819
+ injectionMode: compactResult.injectionMode,
820
+ };
821
+ // The hook chain writes the re-injected history back onto the context;
822
+ // read it from there once the chain settles.
823
+ const finalPostCompactCtx = await runHook(
824
+ HOOKS.POST_COMPACT,
825
+ postCompactCtx,
826
+ );
827
+ return {
828
+ history: finalPostCompactCtx.history,
829
+ exhausted,
830
+ autoCompressApplied,
831
+ };
782
832
  }
783
833
 
784
834
  async run(options: AgentLoopRunOptions): Promise<AgentLoopRunResult> {
@@ -795,21 +845,19 @@ export class AgentLoop {
795
845
  resolveContextWindow,
796
846
  compactInPlace = false,
797
847
  isNonInteractive = false,
798
- modelProfile = null,
799
- actorContext = null,
848
+ modelProfileKey = null,
800
849
  } = options;
801
850
  let history = [...messages];
802
851
  // Index into `history` where this run's appended output begins. It starts
803
- // after the input and resets to the compacted base whenever the loop
804
- // compacts in place, so `history.slice(newMessagesStart)` is always exactly
805
- // what the loop produced since the last (re-injected) base.
852
+ // after the input and resets to the new base whenever the loop rewrites the
853
+ // history in place (compaction re-injection, ordering deep-repair), so
854
+ // `history.slice(newMessagesStart)` is always exactly what the loop produced
855
+ // since the last base.
806
856
  let newMessagesStart = history.length;
807
- let producedVisibleTextThisRun = false;
808
857
  let toolUseTurns = 0;
809
- let stopContinueRetries = 0;
858
+ let postModelCallContinues = 0;
810
859
  let lastLlmCallTime = 0;
811
860
  let exitReason: ExitReason | null = null;
812
- let appendedNewMessages = false;
813
861
  // Armed at the end of a tool-use iteration so the budget gate runs at the
814
862
  // top of the NEXT iteration — before that iteration's provider call —
815
863
  // instead of after the current one. Stop-hook re-query continues re-enter
@@ -817,7 +865,26 @@ export class AgentLoop {
817
865
  // prior post-call placement, plus the first call when
818
866
  // `compactInPlace` is set (the primary run's turn-start compaction).
819
867
  let budgetGateArmed = compactInPlace;
820
- const rlog = requestId ? log.child({ requestId }) : log;
868
+ // Raw pre-send estimate for the most recent provider call, captured so the
869
+ // overflow catch can calibrate the estimator against the provider's actual
870
+ // token count. Reset to the success path's value on every call.
871
+ let lastPreSendEstimatedTokens = 0;
872
+ // Overflow signal stashed by the reactive catch when the provider rejects a
873
+ // call as context-too-large. The next budget gate forwards it into
874
+ // `compact()`, which routes through the manager's reduction ladder, then
875
+ // clears it (one rung consumed per recovery pass).
876
+ let pendingOverflowSignal: {
877
+ actualTokens: number | null;
878
+ isInteractive: boolean;
879
+ } | null = null;
880
+ // Mirror of the reduction ladder's terminal state from the most recent
881
+ // overflow-recovery compaction. When the ladder is spent and the provider
882
+ // still rejects, the catch ends the turn with the reason the final rung
883
+ // implies (auto-compress applied → `budget_yield_unrecovered`, otherwise
884
+ // `context_too_large`) instead of looping.
885
+ let overflowLadderExhausted = false;
886
+ let overflowAutoCompressApplied = false;
887
+ const rlog = log.child({ requestId });
821
888
 
822
889
  // Resolve the inference-profile override that applies right now. The
823
890
  // optional resolver lets a turn observe a confirmed mid-turn profile switch
@@ -831,21 +898,61 @@ export class AgentLoop {
831
898
  const substitutionMap = new Map<string, string>();
832
899
  let streamingPending = "";
833
900
 
834
- // Idempotency guard for `emitExit`: the first reason stamped wins. A break
835
- // site that stamps a specific reason before unwinding into the catch
836
- // handler keeps that reason instead of the generic "error", and the guard
837
- // also defends against accidental double-emits if a new break site is
838
- // added without checking this.
839
- let exitReasonEmitted = false;
840
- const emitExit = async (reason: AgentLoopExitReason): Promise<void> => {
841
- if (exitReasonEmitted) return;
842
- exitReasonEmitted = true;
843
- await onEvent({ type: "agent_loop_exit", reason });
901
+ // Single chokepoint for ending the turn. Runs the definitive terminal
902
+ // `stop` hook chain exactly once — by the time it fires the loop has
903
+ // committed to ending, so teardown hooks can clear per-turn state with the
904
+ // guarantee that nothing will re-enter the loop this turn. The first reason
905
+ // stamped wins: a break site that stamps a specific reason before unwinding
906
+ // into the catch handler keeps that reason instead of the generic "error",
907
+ // and the guard also defends against accidental double-invocation if a new
908
+ // terminal break site is added.
909
+ //
910
+ // `emitExit` controls whether the matching `agent_loop_exit` observability
911
+ // event fires. Real terminal exits emit it; a `checkpoint_handoff` runs the
912
+ // teardown chain (so per-turn state like recovery bounds is cleared before
913
+ // the queued message drains) but does not emit, because the orchestrator
914
+ // owns the handoff signal and the conversation resumes in a fresh run.
915
+ //
916
+ // A throwing `stop` hook must not suppress the terminal exit: the chain is
917
+ // isolated so a failing teardown hook (e.g. a third-party plugin) is logged
918
+ // but `agent_loop_exit` still fires with the real exit reason. Otherwise the
919
+ // throw would unwind into the outer catch, which re-enters here as a no-op
920
+ // (the guard is already set) and the turn's terminal observability event
921
+ // would be dropped.
922
+ let turnStopped = false;
923
+ const runTerminalStop = async (
924
+ reason: AgentLoopExitReason,
925
+ { emitExit, error }: { emitExit: boolean; error?: Error },
926
+ ): Promise<void> => {
927
+ if (turnStopped) return;
928
+ turnStopped = true;
929
+ const stopCtx: StopContext = {
930
+ conversationId: this.conversationId,
931
+ messages: [...history],
932
+ error,
933
+ exitReason: reason,
934
+ logger: rlog,
935
+ };
936
+ try {
937
+ await runHook(HOOKS.STOP, stopCtx);
938
+ } catch (stopHookError) {
939
+ rlog.error(
940
+ { err: stopHookError, exitReason: reason },
941
+ "stop hook threw during terminal teardown; continuing",
942
+ );
943
+ }
944
+ if (emitExit) {
945
+ await onEvent({ type: "agent_loop_exit", reason });
946
+ }
844
947
  };
948
+ const stopTurn = (
949
+ reason: AgentLoopExitReason,
950
+ error?: Error,
951
+ ): Promise<void> => runTerminalStop(reason, { emitExit: true, error });
845
952
 
846
953
  while (true) {
847
954
  if (signal?.aborted) {
848
- await emitExit("aborted_pre_call");
955
+ await stopTurn("aborted_pre_call");
849
956
  break;
850
957
  }
851
958
 
@@ -855,32 +962,40 @@ export class AgentLoop {
855
962
  );
856
963
 
857
964
  let toolUseBlocks: Extract<ContentBlock, { type: "tool_use" }>[] = [];
965
+ // The provider rejection thrown by this iteration's call, if any. Set in
966
+ // the inner provider catch and read by the outer catch to confine
967
+ // error-stop recovery to genuine provider rejections — a throw from
968
+ // elsewhere in the turn body (tool execution, the success-path stop
969
+ // chain, post-model-call hooks) must not re-enter the stop chain.
970
+ let providerCallError: unknown;
858
971
 
859
972
  try {
860
973
  // ── Pre-call budget gate ─────────────────────────────────────
861
- // When overflow recovery is enabled, estimate the running context
862
- // size as it approaches the preflight budget before issuing the
863
- // provider call. With `compactInPlace` the loop compacts in place and
864
- // proceeds with the call; otherwise it yields (`exitReason =
865
- // "budget"`) so the orchestrator can recover before the call risks a
866
- // hard context-too-large rejection. Keyed off the loop's own
867
- // `history.length` (the messages actually in context this turn,
868
- // including tool iterations) rather than the durable conversation
869
- // count.
974
+ // Compact the running history before issuing the provider call when
975
+ // either the running estimate approaches the preflight budget
976
+ // (proactive) or a prior call was rejected as context-too-large
977
+ // (reactive, signalled via `pendingOverflowSignal`). The reactive case
978
+ // forwards the overflow signal into `compact()`, which routes through
979
+ // the manager's reduction ladder; the proactive case runs ordinary
980
+ // forced compaction. Either way the loop proceeds with the call —
981
+ // recovery is driven by compaction, never by yielding out of the loop.
870
982
  //
871
- // Armed after each tool-use iteration; stop-hook re-query continues
872
- // skip it. The first call runs it only when `compactInPlace` is set,
873
- // where it stands in for the orchestrator's turn-start compaction: it
874
- // honors the compaction circuit breaker and proceeds with the call
875
- // rather than yielding, since there is no prior turn output to
876
- // escalate.
983
+ // Keyed off the loop's own `history.length` (the messages actually in
984
+ // context this turn, including tool iterations) rather than the durable
985
+ // conversation count. Armed after each tool-use iteration and by the
986
+ // reactive catch; stop-hook re-query continues skip it. The first call
987
+ // runs it only when `compactInPlace` is set (standing in for turn-start
988
+ // compaction) or when recovering an overflow.
877
989
  if (budgetGateArmed) {
878
990
  budgetGateArmed = false;
991
+ const overflowSignal = pendingOverflowSignal;
992
+ pendingOverflowSignal = null;
879
993
  // The gate only re-arms after a completed tool-use iteration
880
994
  // (`toolUseTurns` is incremented first), so reaching it with
881
- // `toolUseTurns === 0` uniquely identifies the first-call pass: it
882
- // compacts-or-proceeds (never yields) and honors the compaction
883
- // circuit breaker, matching the orchestrator's turn-start compaction.
995
+ // `toolUseTurns === 0` and no overflow signal uniquely identifies the
996
+ // first-call turn-start pass, which honors the compaction circuit
997
+ // breaker. Overflow recovery ignores the breaker — the provider has
998
+ // already rejected the call, so it must reduce regardless.
884
999
  const isFirstCallGate = toolUseTurns === 0;
885
1000
  const contextWindow = resolveContextWindow?.();
886
1001
  if (contextWindow?.overflowRecovery.enabled) {
@@ -898,58 +1013,50 @@ export class AgentLoop {
898
1013
  const midLoopThreshold =
899
1014
  preflightBudget * MID_LOOP_YIELD_THRESHOLD_RATIO;
900
1015
  const estimated = this.estimateTokens(history);
901
- if (estimated > midLoopThreshold) {
902
- let compactedInPlace = false;
903
- // The turn-start pass skips compaction while the circuit breaker
904
- // is open so a run of failed summaries doesn't keep hammering the
905
- // summary LLM; mid-loop compaction is force-driven and proceeds
906
- // regardless (it has already committed to compacting in place).
907
- const compactionAllowed =
908
- !isFirstCallGate || !(await this.compactionCircuit.isOpen());
909
- if (compactInPlace && compactionAllowed) {
910
- rlog.info(
911
- {
912
- turn: toolUseTurns,
913
- estimated,
914
- threshold: midLoopThreshold,
915
- },
916
- "Token estimate approaching budget — compacting in place",
917
- );
918
- const compacted = await this.compact(
919
- history,
920
- requestId,
921
- trust,
922
- signal,
923
- onEvent,
924
- resolveEffectiveOverrideProfile() ?? null,
925
- isNonInteractive,
926
- modelProfile,
927
- actorContext,
928
- );
929
- if (compacted) {
930
- history = compacted;
931
- // The compacted, re-injected array is the new base; output
932
- // produced after this point is what the orchestrator
933
- // persists.
934
- newMessagesStart = history.length;
935
- compactedInPlace = true;
936
- }
1016
+ const overflowDriven = overflowSignal !== null;
1017
+ // Proactive compaction fires when the primary run's turn-start
1018
+ // signal (`compactInPlace`) crosses the estimate threshold;
1019
+ // overflow recovery always compacts.
1020
+ const shouldCompact =
1021
+ overflowDriven ||
1022
+ (compactInPlace && estimated > midLoopThreshold);
1023
+ const compactionAllowed =
1024
+ overflowDriven ||
1025
+ !isFirstCallGate ||
1026
+ !(await this.compactionCircuit.isOpen());
1027
+ if (shouldCompact && compactionAllowed) {
1028
+ rlog.info(
1029
+ {
1030
+ turn: toolUseTurns,
1031
+ estimated,
1032
+ threshold: midLoopThreshold,
1033
+ overflowDriven,
1034
+ },
1035
+ "Compacting in place before provider call",
1036
+ );
1037
+ const attempt = await this.compact(
1038
+ history,
1039
+ requestId,
1040
+ trust,
1041
+ signal,
1042
+ onEvent,
1043
+ resolveEffectiveOverrideProfile() ?? null,
1044
+ isNonInteractive,
1045
+ modelProfileKey,
1046
+ overflowSignal ?? undefined,
1047
+ );
1048
+ if (attempt.history) {
1049
+ history = attempt.history;
1050
+ // The compacted, re-injected array is the new base; output
1051
+ // produced after this point is what the wrapper persists.
1052
+ newMessagesStart = history.length;
937
1053
  }
938
- // The turn-start gate proceeds with the call whether or not it
939
- // compacted (preflight-overflow recovery and the convergence loop
940
- // remain the escalation path); only mid-loop re-entries yield to
941
- // the orchestrator before the call.
942
- if (!compactedInPlace && !isFirstCallGate) {
943
- rlog.warn(
944
- {
945
- turn: toolUseTurns,
946
- estimated,
947
- threshold: midLoopThreshold,
948
- },
949
- "Token estimate approaching budget — yielding for compaction",
950
- );
951
- exitReason = "budget";
952
- break;
1054
+ if (overflowDriven) {
1055
+ // Carry the ladder's terminal state to the catch: if the
1056
+ // provider rejects again after the ladder is spent, the turn
1057
+ // ends instead of looping.
1058
+ overflowLadderExhausted = attempt.exhausted;
1059
+ overflowAutoCompressApplied = attempt.autoCompressApplied;
953
1060
  }
954
1061
  }
955
1062
  }
@@ -1091,6 +1198,7 @@ export class AgentLoop {
1091
1198
  toolTokenBudget,
1092
1199
  },
1093
1200
  );
1201
+ lastPreSendEstimatedTokens = preSendEstimatedTokens;
1094
1202
  rlog.info({ turn: toolUseTurns }, "LLM call start");
1095
1203
 
1096
1204
  // Sanitize the outbound history right before sending: drop accumulated
@@ -1104,6 +1212,13 @@ export class AgentLoop {
1104
1212
  // instead. Reset per model call.
1105
1213
  let deferAssistantOutput = false;
1106
1214
 
1215
+ // Set once any visible assistant text reaches the client live this
1216
+ // model call. A deferred turn holds the live stream and a turn the
1217
+ // model leaves visibly empty streams nothing, so this stays false for
1218
+ // both — letting the loop surface the finalized text exactly once when
1219
+ // the client would otherwise see nothing.
1220
+ let streamedVisibleText = false;
1221
+
1107
1222
  // The `onEvent` wrapping below applies sensitive-output placeholder
1108
1223
  // substitution to streamed text while forwarding every other event
1109
1224
  // type through unchanged.
@@ -1125,9 +1240,11 @@ export class AgentLoop {
1125
1240
  );
1126
1241
  streamingPending = pending;
1127
1242
  if (emit.length > 0) {
1243
+ streamedVisibleText = true;
1128
1244
  onEvent({ type: "text_delta", text: emit });
1129
1245
  }
1130
1246
  } else {
1247
+ if (event.text.length > 0) streamedVisibleText = true;
1131
1248
  onEvent({ type: "text_delta", text: event.text });
1132
1249
  }
1133
1250
  } else if (event.type === "thinking_delta") {
@@ -1178,8 +1295,8 @@ export class AgentLoop {
1178
1295
  try {
1179
1296
  const preModelCtx: PreModelCallContext = {
1180
1297
  conversationId: this.conversationId,
1181
- callSite,
1182
- systemPrompt: providerOptions.systemPrompt,
1298
+ callSite: callSite ?? null,
1299
+ systemPrompt: providerOptions.systemPrompt ?? null,
1183
1300
  deferAssistantOutput: false,
1184
1301
  logger: rlog,
1185
1302
  };
@@ -1187,7 +1304,8 @@ export class AgentLoop {
1187
1304
  HOOKS.PRE_MODEL_CALL,
1188
1305
  preModelCtx,
1189
1306
  );
1190
- providerOptions.systemPrompt = finalPreModelCtx.systemPrompt;
1307
+ providerOptions.systemPrompt =
1308
+ finalPreModelCtx.systemPrompt ?? undefined;
1191
1309
  // The hook owns the policy (it sees `callSite`/conversation and
1192
1310
  // self-gates); the loop honors whatever it decides.
1193
1311
  deferAssistantOutput = finalPreModelCtx.deferAssistantOutput;
@@ -1256,6 +1374,7 @@ export class AgentLoop {
1256
1374
  : this.provider.name,
1257
1375
  });
1258
1376
  }
1377
+ providerCallError = llmCallError;
1259
1378
  throw llmCallError;
1260
1379
  }
1261
1380
 
@@ -1279,49 +1398,72 @@ export class AgentLoop {
1279
1398
  if (streamingPending.length > 0) {
1280
1399
  const flushed = applySubstitutions(streamingPending, substitutionMap);
1281
1400
  if (flushed.length > 0) {
1401
+ streamedVisibleText = true;
1282
1402
  onEvent({ type: "text_delta", text: flushed });
1283
1403
  }
1284
1404
  streamingPending = "";
1285
1405
  }
1286
1406
 
1287
- // Run the `post-model-call` hook on a finalized message and, when
1288
- // output was deferred, emit the finalized text once (with sensitive-output
1289
- // substitution applied, matching the live stream). Fail-open: the hook
1290
- // receives a clone, so a throw — even mid in-place mutation — leaves the
1291
- // original message intact.
1407
+ // Run the `post-model-call` hook on a finalized message. Fail-open: the
1408
+ // hook receives a clone, so a throw — even mid in-place mutation —
1409
+ // leaves the original message intact and the outcome resolves to
1410
+ // `"stop"`.
1411
+ //
1412
+ // Returns the finalized message alongside the chain's retry `decision`
1413
+ // and resulting `messages`. The caller honors `decision` only at an
1414
+ // actionable outcome (a no-tool stop boundary); other call sites take
1415
+ // the finalized message and ignore the decision. Final output is not
1416
+ // emitted here — the caller emits it via `emitFinalAssistantText`
1417
+ // once it has decided to keep the turn, so a re-queried reply isn't
1418
+ // streamed-then-discarded.
1292
1419
  const finalizeAssistantMessage = async (
1293
1420
  message: Message,
1294
- ): Promise<Message> => {
1295
- let finalized = message;
1421
+ ): Promise<{
1422
+ finalized: Message;
1423
+ decision: PostModelCallDecision;
1424
+ messages: Message[];
1425
+ }> => {
1296
1426
  try {
1297
1427
  const ctx: PostModelCallContext = {
1298
1428
  conversationId: this.conversationId,
1299
- callSite,
1429
+ callSite: callSite ?? null,
1300
1430
  content: structuredClone(message.content),
1431
+ messages: [...history],
1301
1432
  stopReason: response.stopReason,
1433
+ decision: "stop",
1302
1434
  logger: rlog,
1303
1435
  };
1304
1436
  const result = await runHook(HOOKS.POST_MODEL_CALL, ctx);
1305
- finalized = { role: "assistant", content: result.content };
1437
+ return {
1438
+ finalized: { role: "assistant", content: result.content },
1439
+ decision: result.decision,
1440
+ messages: result.messages,
1441
+ };
1306
1442
  } catch (assistantMessageError) {
1307
1443
  rlog.error(
1308
1444
  { err: assistantMessageError },
1309
1445
  "post-model-call hook failed — keeping the original content",
1310
1446
  );
1311
- finalized = message;
1447
+ return { finalized: message, decision: "stop", messages: history };
1312
1448
  }
1313
- if (deferAssistantOutput) {
1314
- // The persisted message keeps sensitive-output placeholders; the
1315
- // stream shows real values — substitute before emitting.
1316
- const finalText = applySubstitutions(
1317
- assistantTextOf(finalized.content),
1318
- substitutionMap,
1319
- );
1320
- if (finalText.length > 0) {
1321
- onEvent({ type: "text_delta", text: finalText });
1322
- }
1449
+ };
1450
+
1451
+ // Surface the finalized assistant text when the client saw nothing live
1452
+ // this turn: a deferred turn held its live stream, and a turn the model
1453
+ // left visibly empty that a `post-model-call` hook rewrote into visible
1454
+ // text (e.g. a refusal turned into an apology) streamed nothing either.
1455
+ // Sensitive-output substitution is applied to match what the live stream
1456
+ // would have shown. A no-op when text already streamed live — that
1457
+ // stream stands. Call only for a turn being kept.
1458
+ const emitFinalAssistantText = (content: ContentBlock[]): void => {
1459
+ if (streamedVisibleText) return;
1460
+ const finalText = applySubstitutions(
1461
+ assistantTextOf(content),
1462
+ substitutionMap,
1463
+ );
1464
+ if (finalText.length > 0) {
1465
+ onEvent({ type: "text_delta", text: finalText });
1323
1466
  }
1324
- return finalized;
1325
1467
  };
1326
1468
 
1327
1469
  // Build the assistant message with placeholder-only text.
@@ -1369,14 +1511,16 @@ export class AgentLoop {
1369
1511
  "LLM response reached output token limit",
1370
1512
  );
1371
1513
  // Run the hook on the truncated reply so output-filter plugins still
1372
- // see it, and so a deferred turn gets its synthetic final emit (the
1373
- // live stream was suppressed; without this the client would see nothing).
1374
- const safeAssistantMessage = await finalizeAssistantMessage({
1375
- role: "assistant",
1376
- content: safeContent,
1377
- });
1514
+ // see it, and so a turn that streamed nothing live gets its final
1515
+ // emit (without this the client would see nothing). The retry decision
1516
+ // is ignored here: a max-tokens stop is terminal.
1517
+ const { finalized: safeAssistantMessage } =
1518
+ await finalizeAssistantMessage({
1519
+ role: "assistant",
1520
+ content: safeContent,
1521
+ });
1522
+ emitFinalAssistantText(safeAssistantMessage.content);
1378
1523
  history.push(safeAssistantMessage);
1379
- appendedNewMessages = true;
1380
1524
  await onEvent({
1381
1525
  type: "max_tokens_reached",
1382
1526
  stopReason: response.stopReason,
@@ -1385,80 +1529,76 @@ export class AgentLoop {
1385
1529
  type: "message_complete",
1386
1530
  message: safeAssistantMessage,
1387
1531
  });
1388
- await emitExit("max_tokens_reached");
1532
+ await stopTurn("max_tokens_reached");
1389
1533
  break;
1390
1534
  }
1391
1535
 
1392
- // The model's "stop" moment: a response with no tool calls is about to
1393
- // yield to the user. The `stop` hook (below) decides whether to accept
1394
- // the turn or re-query with a follow-up; `priorAssistantHadVisibleText`
1395
- // gates the ops log for the post-tool empty case.
1396
- const hasVisibleText = response.content.some(
1397
- (block) => block.type === "text" && block.text.trim().length > 0,
1398
- );
1399
- const priorAssistantHadVisibleText = producedVisibleTextThisRun;
1400
- if (hasVisibleText) {
1401
- producedVisibleTextThisRun = true;
1402
- }
1403
-
1404
- if (toolUseBlocks.length === 0) {
1405
- // The model stopped requesting tools — the run's stop boundary. The
1406
- // `stop` hook decides whether to let the turn end or re-query with a
1407
- // follow-up turn. It receives the full history and, when it asks to
1408
- // continue, appends the follow-up turn itself.
1409
- const stopCtx: StopContext = {
1410
- conversationId: this.conversationId,
1411
- messages: [...history],
1412
- responseContent: response.content,
1413
- stopReason: response.stopReason,
1414
- decision: "stop",
1415
- logger: rlog,
1416
- };
1417
- const finalStopCtx = await runHook(HOOKS.STOP, stopCtx);
1418
-
1419
- if (finalStopCtx.decision === "continue") {
1420
- // The loop owns the retry budget: a hook always asks to continue
1421
- // when a nudge is warranted, and the loop stops anyway once the
1422
- // budget is spent. This bounds the hook-driven re-query loop.
1423
- if (stopContinueRetries < MAX_STOP_CONTINUE_RETRIES) {
1424
- stopContinueRetries++;
1425
- rlog.warn(
1426
- { turn: toolUseTurns, retry: stopContinueRetries },
1427
- "Model returned empty response after tool results — retrying",
1428
- );
1429
- history = finalStopCtx.messages;
1430
- continue;
1431
- }
1432
-
1433
- // Budget spent — accept the empty turn. Emit a dedicated log line
1434
- // for the post-tool empty case so ops dashboards that grep on it
1435
- // keep working.
1436
- if (
1437
- !hasVisibleText &&
1438
- toolUseTurns > 0 &&
1439
- !priorAssistantHadVisibleText
1440
- ) {
1441
- rlog.error(
1442
- { turn: toolUseTurns, retries: stopContinueRetries },
1443
- "Model returned empty response after tool results — retries exhausted",
1444
- );
1445
- }
1536
+ // A response with no tool calls is the run's stop boundary. The
1537
+ // `post-model-call` hook (below) sees the finalized reply and decides
1538
+ // whether to accept the turn or re-query with a follow-up.
1539
+ const responseHasVisibleText = hasVisibleText(response.content);
1540
+
1541
+ // Run the `post-model-call` hook: transform the finalized reply and
1542
+ // surface its retry decision.
1543
+ const {
1544
+ finalized: finalizedAssistantMessage,
1545
+ decision: postModelCallDecision,
1546
+ messages: postModelCallMessages,
1547
+ } = await finalizeAssistantMessage(assistantMessage);
1548
+ assistantMessage = finalizedAssistantMessage;
1549
+
1550
+ // At the no-tool stop boundary the retry decision is actionable: a
1551
+ // recovery hook may repair history and ask to re-query (a tool-bearing
1552
+ // turn already continues, so its decision is ignored). A re-query
1553
+ // adopts the hook's `messages` and discards this turn rather than
1554
+ // persisting it; the per-run backstop keeps a misbehaving hook from
1555
+ // spinning forever.
1556
+ if (
1557
+ toolUseBlocks.length === 0 &&
1558
+ postModelCallDecision === "continue"
1559
+ ) {
1560
+ // A retry discards this reply and re-queries. That is only safe when
1561
+ // the reply was not already streamed to the client live: a deferred
1562
+ // turn suppressed its live stream, and a reply with no visible text
1563
+ // streamed nothing. Honoring a retry on an already-streamed visible
1564
+ // reply would leave the user looking at an answer the transcript
1565
+ // then silently replaces, with no retraction — so accept the turn
1566
+ // instead of discarding visible output.
1567
+ const replyWasStreamedLive =
1568
+ responseHasVisibleText && !deferAssistantOutput;
1569
+ if (replyWasStreamedLive) {
1570
+ rlog.warn(
1571
+ { turn: toolUseTurns },
1572
+ "post-model-call requested a retry on an already-streamed reply — keeping the turn to avoid discarding visible output",
1573
+ );
1574
+ } else if (postModelCallContinues < MAX_POST_MODEL_CALL_CONTINUES) {
1575
+ postModelCallContinues++;
1576
+ rlog.warn(
1577
+ { turn: toolUseTurns, retry: postModelCallContinues },
1578
+ "post-model-call requested a retry — re-querying the model",
1579
+ );
1580
+ history = postModelCallMessages;
1581
+ continue;
1582
+ } else {
1583
+ rlog.warn(
1584
+ { turn: toolUseTurns, retries: postModelCallContinues },
1585
+ "post-model-call retry backstop reached — accepting the turn",
1586
+ );
1446
1587
  }
1447
1588
  }
1448
1589
 
1449
- // Run the `post-model-call` hook + emit any deferred final text.
1450
- // On a no-tool turn this point is reached only after the `stop` hook
1451
- // resolves to "stop" (a `continue` already re-queried above), so a
1452
- // re-queried reply is never transformed-then-discarded.
1453
- assistantMessage = await finalizeAssistantMessage(assistantMessage);
1590
+ // The turn is being kept: surface the finalized text if the client saw
1591
+ // nothing live (a deferred stream, or a hook-rewritten empty turn).
1592
+ emitFinalAssistantText(assistantMessage.content);
1454
1593
 
1455
1594
  history.push(assistantMessage);
1456
- appendedNewMessages = true;
1457
1595
 
1458
1596
  await onEvent({ type: "message_complete", message: assistantMessage });
1459
1597
 
1460
1598
  if (toolUseBlocks.length === 0 || !this.toolExecutor) {
1461
- await emitExit("no_tool_calls");
1599
+ // The model stopped requesting tools and `post-model-call` settled on
1600
+ // ending the turn: the terminal `stop` chain fires via `stopTurn`.
1601
+ await stopTurn("no_tool_calls");
1462
1602
  break;
1463
1603
  }
1464
1604
 
@@ -1492,7 +1632,7 @@ export class AgentLoop {
1492
1632
  cancelled: true,
1493
1633
  });
1494
1634
  }
1495
- await emitExit("aborted_post_response");
1635
+ await stopTurn("aborted_post_response");
1496
1636
  break;
1497
1637
  }
1498
1638
 
@@ -1605,12 +1745,13 @@ export class AgentLoop {
1605
1745
  conversationId: this.conversationId,
1606
1746
  toolResponse: block as ToolResultContent,
1607
1747
  messages: history,
1748
+ additionalContext: null,
1608
1749
  maxInputTokens: contextWindowTokens,
1609
1750
  logger: rlog,
1610
1751
  };
1611
1752
  const finalCtx = await runHook(HOOKS.POST_TOOL_USE, postToolUseCtx);
1612
1753
  resultBlocks.push(finalCtx.toolResponse);
1613
- if (finalCtx.additionalContext !== undefined) {
1754
+ if (finalCtx.additionalContext !== null) {
1614
1755
  additionalContextBlocks.push({
1615
1756
  type: "text",
1616
1757
  text: finalCtx.additionalContext,
@@ -1654,7 +1795,7 @@ export class AgentLoop {
1654
1795
  // If cancelled during execution, push completed results and stop
1655
1796
  if (signal?.aborted) {
1656
1797
  history.push({ role: "user", content: resultBlocks });
1657
- await emitExit("aborted_during_tools");
1798
+ await stopTurn("aborted_during_tools");
1658
1799
  break;
1659
1800
  }
1660
1801
 
@@ -1662,7 +1803,7 @@ export class AgentLoop {
1662
1803
  // surface awaiting a button click), push results and stop the loop.
1663
1804
  if (toolResults.some(({ result }) => result.yieldToUser)) {
1664
1805
  history.push({ role: "user", content: resultBlocks });
1665
- await emitExit("yield_to_user");
1806
+ await stopTurn("yield_to_user");
1666
1807
  break;
1667
1808
  }
1668
1809
 
@@ -1691,6 +1832,13 @@ export class AgentLoop {
1691
1832
  history,
1692
1833
  });
1693
1834
  if (decision !== "continue") {
1835
+ // A handoff pauses this run so the orchestrator can drain a queued
1836
+ // message, then re-enters with a fresh run. It still ends *this*
1837
+ // turn, so fire the terminal `stop` chain to run teardown (clearing
1838
+ // per-turn state such as recovery bounds before the queued message
1839
+ // is processed) — but without emitting `agent_loop_exit`, since the
1840
+ // orchestrator owns the handoff signal and the conversation resumes.
1841
+ await runTerminalStop("checkpoint_handoff", { emitExit: false });
1694
1842
  exitReason = decision;
1695
1843
  break;
1696
1844
  }
@@ -1726,10 +1874,113 @@ export class AgentLoop {
1726
1874
  });
1727
1875
  }
1728
1876
  }
1729
- await emitExit("aborted_via_error");
1877
+ await stopTurn("aborted_via_error");
1730
1878
  break;
1731
1879
  }
1880
+
1732
1881
  const err = error instanceof Error ? error : new Error(String(error));
1882
+
1883
+ // Reactive context-overflow recovery. The provider rejected the call
1884
+ // because the prompt exceeded its window. Fold the provider's actual
1885
+ // token count into the per-provider calibration (ground truth the
1886
+ // estimator under-counted) and stash the overflow signal so the next
1887
+ // iteration's budget gate forwards it into the compaction plugin's
1888
+ // reduction ladder, which advances one rung before re-issuing the call.
1889
+ // When the ladder is already spent and the provider still rejects, end
1890
+ // the turn with the terminal reason the final rung implies instead of
1891
+ // looping forever. Recovery requires the budget gate to be active; when
1892
+ // it is disabled (e.g. agent wakes) there is no ladder to drive, so the
1893
+ // overflow falls through to the generic error path below.
1894
+ if (
1895
+ isContextOverflowError(error) &&
1896
+ (resolveContextWindow?.().overflowRecovery.enabled ?? false)
1897
+ ) {
1898
+ if (overflowLadderExhausted) {
1899
+ await stopTurn(
1900
+ overflowAutoCompressApplied
1901
+ ? "budget_yield_unrecovered"
1902
+ : "context_too_large",
1903
+ err,
1904
+ );
1905
+ break;
1906
+ }
1907
+ const actualTokens = parseActualTokensFromError(error);
1908
+ if (actualTokens !== null) {
1909
+ recordEstimate(
1910
+ getCalibrationProviderKey(this.provider),
1911
+ "",
1912
+ lastPreSendEstimatedTokens,
1913
+ actualTokens,
1914
+ );
1915
+ }
1916
+ pendingOverflowSignal = {
1917
+ actualTokens,
1918
+ isInteractive: !isNonInteractive,
1919
+ };
1920
+ budgetGateArmed = true;
1921
+ rlog.warn(
1922
+ {
1923
+ turn: toolUseTurns,
1924
+ estimated: lastPreSendEstimatedTokens,
1925
+ actualTokens,
1926
+ },
1927
+ "Context too large — recovering via the compaction reduction ladder",
1928
+ );
1929
+ continue;
1930
+ }
1931
+
1932
+ // A provider rejection is a model-call outcome: the loop has nothing
1933
+ // more to produce this turn unless a recovery hook repairs the history
1934
+ // and asks to retry. Run the `post-model-call` hook with the rejection
1935
+ // attached — a recovery hook (e.g. history-repair on an ordering
1936
+ // violation) can re-normalize the history and set `decision` to
1937
+ // `"continue"` to re-issue the call; hooks that only act on a real
1938
+ // reply ignore the rejection. The same per-run backstop bounds these
1939
+ // error-driven retries as the success-path ones. The chain is run
1940
+ // fail-open: a hook throw surfaces the original rejection.
1941
+ //
1942
+ // Confined to genuine provider rejections: a throw from elsewhere in
1943
+ // the turn body (tool execution, the success-path stop/post-model-call
1944
+ // hooks) is not a provider stop, so it falls straight through to the
1945
+ // error path below.
1946
+ if (error === providerCallError) {
1947
+ const errorOutcomeCtx: PostModelCallContext = {
1948
+ conversationId: this.conversationId,
1949
+ callSite: callSite ?? null,
1950
+ content: [],
1951
+ messages: [...history],
1952
+ stopReason: null,
1953
+ error: err,
1954
+ decision: "stop",
1955
+ logger: rlog,
1956
+ };
1957
+ let errorOutcome: PostModelCallContext = errorOutcomeCtx;
1958
+ try {
1959
+ errorOutcome = await runHook(
1960
+ HOOKS.POST_MODEL_CALL,
1961
+ errorOutcomeCtx,
1962
+ );
1963
+ } catch (postModelCallError) {
1964
+ rlog.error(
1965
+ { err: postModelCallError },
1966
+ "post-model-call hook failed on a provider rejection — surfacing the original error",
1967
+ );
1968
+ }
1969
+ if (
1970
+ errorOutcome.decision === "continue" &&
1971
+ postModelCallContinues < MAX_POST_MODEL_CALL_CONTINUES
1972
+ ) {
1973
+ postModelCallContinues++;
1974
+ history = errorOutcome.messages;
1975
+ // A recovery hook rewrites the history anywhere (deep repair merges
1976
+ // and drops messages), so the prior input boundary no longer maps
1977
+ // onto the new array; the repaired history is the base the retry's
1978
+ // output appends after.
1979
+ newMessagesStart = history.length;
1980
+ continue;
1981
+ }
1982
+ }
1983
+
1733
1984
  rlog.error(
1734
1985
  { err, turn: toolUseTurns, messageCount: history.length },
1735
1986
  "Agent loop error during turn processing",
@@ -1741,7 +1992,7 @@ export class AgentLoop {
1741
1992
  // Catch-block fallback. A break site that stamped a more specific
1742
1993
  // reason before unwinding here keeps it; the guard makes this a no-op.
1743
1994
  // Otherwise this is the genuine unhandled-error exit.
1744
- await emitExit("error");
1995
+ await stopTurn("error", err);
1745
1996
  break;
1746
1997
  }
1747
1998
  }
@@ -1758,7 +2009,6 @@ export class AgentLoop {
1758
2009
  return {
1759
2010
  history,
1760
2011
  exitReason,
1761
- appendedNewMessages,
1762
2012
  newMessages: history.slice(newMessagesStart),
1763
2013
  };
1764
2014
  }