@vellumai/assistant 0.8.10 → 0.8.11-staging.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (400) hide show
  1. package/bun.lock +62 -1
  2. package/docs/workspace-tools.md +196 -0
  3. package/examples/plugins/echo/README.md +3 -3
  4. package/knip.json +1 -0
  5. package/openapi.yaml +460 -128
  6. package/package.json +2 -1
  7. package/scripts/build-plugin-api.ts +299 -0
  8. package/src/__tests__/agent-loop-callsite-precedence.test.ts +7 -0
  9. package/src/__tests__/agent-loop-compaction-events.test.ts +197 -0
  10. package/src/__tests__/agent-loop-exit-reason.test.ts +93 -96
  11. package/src/__tests__/agent-loop-mutable-latest-user-message.test.ts +2 -0
  12. package/src/__tests__/agent-loop-output-hooks.test.ts +274 -1
  13. package/src/__tests__/agent-loop-override-profile.test.ts +3 -0
  14. package/src/__tests__/agent-loop-provider-error-recording.test.ts +4 -0
  15. package/src/__tests__/agent-loop-thinking.test.ts +4 -0
  16. package/src/__tests__/agent-loop.test.ts +578 -5
  17. package/src/__tests__/agent-wake-disk-pressure-callsite.test.ts +0 -1
  18. package/src/__tests__/approval-cascade.test.ts +1 -0
  19. package/src/__tests__/background-workers-disk-pressure.test.ts +0 -2
  20. package/src/__tests__/btw-routes.test.ts +0 -1
  21. package/src/__tests__/build-persisted-content.test.ts +75 -1
  22. package/src/__tests__/catalog-install-normalize.test.ts +141 -0
  23. package/src/__tests__/ces-startup-timeout.test.ts +60 -0
  24. package/src/__tests__/compaction-events.test.ts +1 -0
  25. package/src/__tests__/config-managed-gemini-defaults.test.ts +2 -46
  26. package/src/__tests__/context-overflow-reducer.test.ts +264 -124
  27. package/src/__tests__/context-window-manager-overflow-rung.test.ts +351 -0
  28. package/src/__tests__/conversation-abort-tool-results.test.ts +1 -1
  29. package/src/__tests__/conversation-agent-loop-inference-profile.test.ts +13 -5
  30. package/src/__tests__/conversation-agent-loop-overflow.test.ts +284 -455
  31. package/src/__tests__/conversation-agent-loop.test.ts +131 -551
  32. package/src/__tests__/conversation-app-control-instantiation.test.ts +13 -0
  33. package/src/__tests__/conversation-confirmation-signals.test.ts +1 -0
  34. package/src/__tests__/conversation-fork-crud.test.ts +259 -0
  35. package/src/__tests__/conversation-history-web-search.test.ts +1 -1
  36. package/src/__tests__/conversation-lifecycle.test.ts +257 -1
  37. package/src/__tests__/conversation-process-callsite.test.ts +1 -0
  38. package/src/__tests__/conversation-provider-retry-repair.test.ts +38 -377
  39. package/src/__tests__/conversation-queue.test.ts +1 -39
  40. package/src/__tests__/conversation-runtime-assembly.test.ts +119 -8
  41. package/src/__tests__/conversation-skill-tools.test.ts +491 -5
  42. package/src/__tests__/conversation-slash-queue.test.ts +1 -1
  43. package/src/__tests__/conversation-slash-unknown.test.ts +1 -0
  44. package/src/__tests__/conversation-speed-override.test.ts +1 -0
  45. package/src/__tests__/conversation-store.test.ts +74 -0
  46. package/src/__tests__/conversation-surfaces-app-control.test.ts +4 -1
  47. package/src/__tests__/conversation-tool-setup-attribution.test.ts +323 -0
  48. package/src/__tests__/conversation-tool-setup-tools-disabled.test.ts +34 -0
  49. package/src/__tests__/conversation-workspace-cache-state.test.ts +1 -0
  50. package/src/__tests__/conversation-workspace-injection.test.ts +1 -1
  51. package/src/__tests__/conversation-workspace-tool-tracking.test.ts +1 -0
  52. package/src/__tests__/corrected-target.test.ts +93 -0
  53. package/src/__tests__/credential-execution-feature-gates.test.ts +3 -5
  54. package/src/__tests__/credential-execution-tools.test.ts +23 -11
  55. package/src/__tests__/credential-security-invariants.test.ts +6 -1
  56. package/src/__tests__/db-schedule-syntax-migration.test.ts +80 -0
  57. package/src/__tests__/device-id.test.ts +70 -1
  58. package/src/__tests__/embedding-managed-proxy-selection.test.ts +6 -40
  59. package/src/__tests__/empty-response-hook.test.ts +242 -66
  60. package/src/__tests__/external-plugin-loader.test.ts +0 -31
  61. package/src/__tests__/get-skill-detail-audit.test.ts +43 -1
  62. package/src/__tests__/guardian-routing-invariants.test.ts +91 -0
  63. package/src/__tests__/history-repair-hook.test.ts +228 -3
  64. package/src/__tests__/host-app-control-proxy.test.ts +45 -0
  65. package/src/__tests__/host-browser-proxy.test.ts +254 -9
  66. package/src/__tests__/identity-routes.test.ts +1 -0
  67. package/src/__tests__/image-recovery-hook.test.ts +387 -0
  68. package/src/__tests__/injector-chain.test.ts +5 -4
  69. package/src/__tests__/injector-v3-suppression.test.ts +373 -47
  70. package/src/__tests__/intent-routing.test.ts +7 -0
  71. package/src/__tests__/memory-retrieval-hook.test.ts +117 -15
  72. package/src/__tests__/notification-decision-strategy.test.ts +3 -3
  73. package/src/__tests__/oauth-store.test.ts +0 -85
  74. package/src/__tests__/{context-overflow-policy.test.ts → overflow-policy.test.ts} +1 -1
  75. package/src/__tests__/parallel-tool.benchmark.test.ts +4 -0
  76. package/src/__tests__/persist-unsendable-image-downscale.test.ts +29 -9
  77. package/src/__tests__/persist-unsendable-image.test.ts +4 -4
  78. package/src/__tests__/persistence-secret-redaction.test.ts +78 -0
  79. package/src/__tests__/plugin-bootstrap.test.ts +82 -73
  80. package/src/__tests__/plugin-tool-contribution.test.ts +7 -4
  81. package/src/__tests__/plugin-types.test.ts +0 -8
  82. package/src/__tests__/provider-catalog-visibility.test.ts +1 -9
  83. package/src/__tests__/prune-old-conversations-job.test.ts +99 -0
  84. package/src/__tests__/registry.test.ts +240 -1
  85. package/src/__tests__/require-fresh-approval.test.ts +3 -0
  86. package/src/__tests__/schedule-routes.test.ts +116 -1
  87. package/src/__tests__/schedule-store.test.ts +28 -0
  88. package/src/__tests__/schedule-tools.test.ts +94 -1
  89. package/src/__tests__/server-history-render.test.ts +39 -0
  90. package/src/__tests__/skill-projection-feature-flag.test.ts +13 -0
  91. package/src/__tests__/skill-projection.benchmark.test.ts +25 -7
  92. package/src/__tests__/skills.test.ts +202 -0
  93. package/src/__tests__/slim-skill-category.test.ts +195 -0
  94. package/src/__tests__/strip-memory-injections.test.ts +33 -39
  95. package/src/__tests__/test-support/tool-invocation-seed.ts +79 -0
  96. package/src/__tests__/title-generate-hook.test.ts +9 -7
  97. package/src/__tests__/tool-audit-listener.test.ts +264 -1
  98. package/src/__tests__/tool-error-hook.test.ts +4 -3
  99. package/src/__tests__/tool-execution-pipeline.benchmark.test.ts +1 -0
  100. package/src/__tests__/tool-executor-lifecycle-events.test.ts +273 -0
  101. package/src/__tests__/tool-result-truncate-hook.test.ts +1 -0
  102. package/src/__tests__/tool-start-timestamp.test.ts +218 -0
  103. package/src/__tests__/tools-get-route.test.ts +202 -0
  104. package/src/__tests__/workspace-tool-loader.test.ts +319 -0
  105. package/src/__tests__/workspace-tools-watcher-flag.test.ts +70 -0
  106. package/src/agent/loop.ts +569 -319
  107. package/src/api/events/tool-result.ts +9 -0
  108. package/src/api/events/tool-use-start.ts +7 -0
  109. package/src/api/index.ts +10 -0
  110. package/src/api/responses/conversation-message.ts +135 -27
  111. package/src/api/responses/memory-v3-selection-log.ts +4 -4
  112. package/src/approvals/guardian-request-resolvers.ts +26 -0
  113. package/src/browser-session/backends/host-bridge.ts +29 -0
  114. package/src/browser-session/index.ts +1 -0
  115. package/src/browser-session/types.ts +5 -1
  116. package/src/cli/commands/__tests__/schedules.test.ts +62 -4
  117. package/src/cli/commands/__tests__/skills.test.ts +53 -0
  118. package/src/cli/commands/channel-verification-sessions.ts +6 -6
  119. package/src/cli/commands/inference-providers.ts +0 -8
  120. package/src/cli/commands/plugins.ts +2 -2
  121. package/src/cli/commands/schedules.ts +27 -4
  122. package/src/cli/commands/skills.ts +187 -146
  123. package/src/cli/commands/tools.ts +106 -0
  124. package/src/cli/lib/__tests__/install-from-github.test.ts +256 -328
  125. package/src/cli/lib/__tests__/plugin-catalog-cache.test.ts +6 -2
  126. package/src/cli/lib/__tests__/plugin-details.test.ts +10 -16
  127. package/src/cli/lib/__tests__/plugin-marketplace.test.ts +2 -2
  128. package/src/cli/lib/__tests__/search-plugins.test.ts +145 -240
  129. package/src/cli/lib/install-from-github.ts +187 -117
  130. package/src/cli/lib/plugin-catalog-cache.ts +9 -9
  131. package/src/cli/lib/plugin-details.ts +38 -68
  132. package/src/cli/lib/plugin-marketplace.ts +42 -14
  133. package/src/cli/lib/search-plugins.ts +29 -129
  134. package/src/cli/program.ts +2 -0
  135. package/src/config/bundled-skills/acp/SKILL.md +1 -0
  136. package/src/config/bundled-skills/app-builder/SKILL.md +1 -0
  137. package/src/config/bundled-skills/app-control/SKILL.md +1 -0
  138. package/src/config/bundled-skills/computer-use/SKILL.md +1 -0
  139. package/src/config/bundled-skills/contacts/SKILL.md +1 -0
  140. package/src/config/bundled-skills/document-editor/SKILL.md +1 -0
  141. package/src/config/bundled-skills/followups/SKILL.md +1 -0
  142. package/src/config/bundled-skills/image-studio/SKILL.md +1 -0
  143. package/src/config/bundled-skills/media-processing/SKILL.md +1 -0
  144. package/src/config/bundled-skills/messaging/SKILL.md +1 -0
  145. package/src/config/bundled-skills/phone-calls/SKILL.md +1 -0
  146. package/src/config/bundled-skills/playbooks/SKILL.md +1 -0
  147. package/src/config/bundled-skills/schedule/SKILL.md +1 -0
  148. package/src/config/bundled-skills/schedule/TOOLS.json +11 -3
  149. package/src/config/bundled-skills/sequences/SKILL.md +1 -0
  150. package/src/config/bundled-skills/settings/SKILL.md +1 -0
  151. package/src/config/bundled-skills/skill-management/SKILL.md +101 -1
  152. package/src/config/bundled-skills/subagent/SKILL.md +1 -0
  153. package/src/config/bundled-skills/transcribe/SKILL.md +1 -0
  154. package/src/config/env-registry.ts +23 -0
  155. package/src/config/feature-flag-registry.json +17 -81
  156. package/src/config/loader.ts +5 -22
  157. package/src/config/schema.ts +2 -0
  158. package/src/config/schemas/__tests__/compaction-logs.test.ts +56 -0
  159. package/src/config/schemas/__tests__/memory-v2.test.ts +0 -1
  160. package/src/config/schemas/__tests__/memory-v3.test.ts +61 -1
  161. package/src/config/schemas/compaction-logs.ts +79 -0
  162. package/src/config/schemas/memory-v2.ts +0 -8
  163. package/src/config/schemas/memory-v3.ts +104 -33
  164. package/src/config/seed-inference-profiles.ts +1 -1
  165. package/src/config/skills.ts +117 -47
  166. package/src/context/compactor.ts +11 -0
  167. package/src/context/strip-injections.ts +38 -4
  168. package/src/credential-execution/feature-gates.ts +0 -21
  169. package/src/credential-execution/startup-timeout.ts +32 -4
  170. package/src/daemon/__tests__/conversation-tool-setup-exclude.test.ts +18 -0
  171. package/src/daemon/conversation-agent-loop-handlers.ts +140 -91
  172. package/src/daemon/conversation-agent-loop.ts +123 -658
  173. package/src/daemon/conversation-error.ts +6 -33
  174. package/src/daemon/conversation-lifecycle.ts +1 -1
  175. package/src/daemon/conversation-runtime-assembly.ts +183 -22
  176. package/src/daemon/conversation-skill-tools.ts +137 -8
  177. package/src/daemon/conversation-slash.ts +0 -14
  178. package/src/daemon/conversation-store.ts +2 -19
  179. package/src/daemon/conversation-tool-setup.ts +87 -1
  180. package/src/daemon/conversation.ts +119 -50
  181. package/src/daemon/external-plugins-bootstrap.ts +36 -95
  182. package/src/daemon/handlers/config-channels.ts +11 -2
  183. package/src/daemon/handlers/shared.ts +9 -1
  184. package/src/daemon/handlers/skills.ts +10 -4
  185. package/src/daemon/host-app-control-proxy.ts +72 -57
  186. package/src/daemon/host-browser-proxy.ts +117 -22
  187. package/src/daemon/lifecycle.ts +34 -5
  188. package/src/daemon/message-protocol.ts +0 -7
  189. package/src/daemon/message-types/schedules.ts +1 -0
  190. package/src/daemon/message-types/skills.ts +17 -0
  191. package/src/daemon/providers-setup.ts +3 -0
  192. package/src/daemon/server.ts +3 -3
  193. package/src/daemon/tool-setup-types.ts +9 -3
  194. package/src/daemon/trust-context.ts +23 -0
  195. package/src/daemon/workspace-tools-watcher.ts +324 -0
  196. package/src/events/tool-audit-listener.ts +78 -16
  197. package/src/events/tool-metrics-listener.ts +2 -5
  198. package/src/memory/__tests__/compaction-log-writer-clickhouse.test.ts +227 -0
  199. package/src/memory/__tests__/conversation-queries.test.ts +176 -0
  200. package/src/memory/__tests__/jobs-worker-v2-schedule.test.ts +20 -32
  201. package/src/memory/compaction-log-writer-clickhouse.ts +418 -0
  202. package/src/memory/conversation-crud.ts +134 -3
  203. package/src/memory/conversation-queries.ts +64 -7
  204. package/src/memory/db-init.ts +14 -0
  205. package/src/memory/embedding-backend.test.ts +130 -1
  206. package/src/memory/embedding-backend.ts +79 -106
  207. package/src/memory/embedding-gemini.ts +5 -0
  208. package/src/memory/graph/__tests__/conversation-graph-memory-v2-routing.test.ts +12 -0
  209. package/src/memory/graph/__tests__/handle-remember-v2.test.ts +19 -0
  210. package/src/memory/graph/conversation-graph-memory.ts +36 -25
  211. package/src/memory/graph/tool-handlers.ts +3 -0
  212. package/src/memory/job-handlers/cleanup.ts +3 -1
  213. package/src/memory/jobs-store.ts +0 -28
  214. package/src/memory/jobs-worker.ts +10 -22
  215. package/src/memory/memory-marker.ts +29 -0
  216. package/src/memory/memory-retrospective-startup-cleanup.ts +1 -1
  217. package/src/memory/migrations/268-add-memory-v3-selections.ts +6 -0
  218. package/src/memory/migrations/270-schedule-description.ts +36 -0
  219. package/src/memory/migrations/275-tool-invocations-add-skill-id.test.ts +81 -0
  220. package/src/memory/migrations/275-tool-invocations-add-skill-id.ts +20 -0
  221. package/src/memory/migrations/276-tool-invocations-created-at-id-index.test.ts +68 -0
  222. package/src/memory/migrations/276-tool-invocations-created-at-id-index.ts +20 -0
  223. package/src/memory/migrations/277-add-memory-v3-ever-injected.ts +29 -0
  224. package/src/memory/migrations/278-tool-invocations-telemetry-columns.test.ts +96 -0
  225. package/src/memory/migrations/278-tool-invocations-telemetry-columns.ts +39 -0
  226. package/src/memory/migrations/279-create-skill-loaded-events.test.ts +84 -0
  227. package/src/memory/migrations/279-create-skill-loaded-events.ts +26 -0
  228. package/src/memory/migrations/280-conversations-surfaced-at.test.ts +88 -0
  229. package/src/memory/migrations/280-conversations-surfaced-at.ts +24 -0
  230. package/src/memory/migrations/index.ts +10 -0
  231. package/src/memory/migrations/registry.ts +8 -0
  232. package/src/memory/schema/conversations.ts +16 -0
  233. package/src/memory/schema/infrastructure.ts +26 -0
  234. package/src/memory/skill-loaded-events-store.test.ts +160 -0
  235. package/src/memory/skill-loaded-events-store.ts +95 -0
  236. package/src/memory/tool-executed-events-store.test.ts +219 -0
  237. package/src/memory/tool-executed-events-store.ts +102 -0
  238. package/src/memory/tool-usage-store.ts +15 -3
  239. package/src/memory/v2/__tests__/consolidation-job.test.ts +117 -12
  240. package/src/memory/v2/__tests__/consolidation-prompt-flag-gating-guard.test.ts +189 -0
  241. package/src/memory/v2/__tests__/injected-block-slugs.test.ts +90 -0
  242. package/src/memory/v2/__tests__/page-store.test.ts +33 -0
  243. package/src/memory/v2/__tests__/prompts-consolidation.test.ts +88 -15
  244. package/src/memory/v2/activation-store.ts +50 -1
  245. package/src/memory/v2/consolidation-job.ts +113 -29
  246. package/src/memory/v2/injected-block-slugs.ts +79 -0
  247. package/src/memory/v2/injection.ts +6 -1
  248. package/src/memory/v2/prompts/consolidation.ts +414 -13
  249. package/src/memory/v2/router.ts +2 -28
  250. package/src/memory/v2/static-context.ts +1 -1
  251. package/src/memory/v2/types.ts +16 -0
  252. package/src/notifications/__tests__/copy-composer.test.ts +244 -0
  253. package/src/notifications/access-request-copy.ts +298 -0
  254. package/src/notifications/adapters/slack.ts +3 -3
  255. package/src/notifications/adapters/telegram.ts +2 -1
  256. package/src/notifications/copy-composer.ts +49 -267
  257. package/src/notifications/decision-engine.ts +16 -35
  258. package/src/notifications/home-feed-side-effect.ts +1 -6
  259. package/src/oauth/oauth-store.ts +0 -9
  260. package/src/permissions/checker.test.ts +83 -1
  261. package/src/permissions/checker.ts +25 -2
  262. package/src/permissions/gateway-threshold-reader.test.ts +182 -0
  263. package/src/permissions/gateway-threshold-reader.ts +80 -0
  264. package/src/platform/client.ts +1 -3
  265. package/src/platform/feature-gate.ts +3 -12
  266. package/src/plugin-api/constants.ts +4 -2
  267. package/src/plugin-api/index.ts +63 -11
  268. package/src/plugin-api/types.ts +236 -71
  269. package/src/plugins/defaults/compaction/compact.ts +66 -2
  270. package/src/plugins/defaults/compaction/context-overflow-reducer.ts +240 -32
  271. package/src/plugins/defaults/compaction/corrected-target.ts +53 -0
  272. package/src/{daemon/context-overflow-policy.ts → plugins/defaults/compaction/overflow-policy.ts} +1 -1
  273. package/src/plugins/defaults/compaction/window-manager.ts +303 -1
  274. package/src/plugins/defaults/empty-response/hooks/post-model-call.ts +173 -0
  275. package/src/plugins/defaults/empty-response/hooks/stop.ts +11 -115
  276. package/src/plugins/defaults/empty-response/nudge-state-store.ts +46 -0
  277. package/src/plugins/defaults/history-repair/hooks/post-model-call.ts +50 -0
  278. package/src/plugins/defaults/history-repair/hooks/stop.ts +22 -0
  279. package/src/plugins/defaults/history-repair/repair-state-store.ts +51 -0
  280. package/src/plugins/defaults/history-repair/terminal.ts +39 -2
  281. package/src/plugins/defaults/image-recovery/detect.ts +25 -0
  282. package/src/plugins/defaults/image-recovery/hooks/post-model-call.ts +73 -0
  283. package/src/plugins/defaults/image-recovery/hooks/stop.ts +22 -0
  284. package/src/plugins/defaults/image-recovery/image-recovery-state-store.ts +48 -0
  285. package/src/plugins/defaults/image-recovery/package.json +14 -0
  286. package/src/{daemon/persist-unsendable-image.ts → plugins/defaults/image-recovery/recover.ts} +67 -14
  287. package/src/plugins/defaults/index.ts +71 -5
  288. package/src/plugins/defaults/memory-retrieval/hooks/post-compact.ts +76 -112
  289. package/src/plugins/defaults/memory-retrieval/hooks/{user-prompt-submit-temp.ts → user-prompt-submit.ts} +83 -74
  290. package/src/plugins/defaults/memory-retrieval/injector-chain.ts +14 -8
  291. package/src/plugins/defaults/memory-retrieval/injectors.ts +2 -18
  292. package/src/plugins/defaults/memory-retrieval/package.json +14 -0
  293. package/src/plugins/defaults/memory-v3-shadow/__tests__/carry-integration.test.ts +1157 -0
  294. package/src/plugins/defaults/memory-v3-shadow/__tests__/injection.test.ts +683 -0
  295. package/src/plugins/defaults/memory-v3-shadow/__tests__/live-integration.test.ts +161 -140
  296. package/src/plugins/defaults/memory-v3-shadow/__tests__/maintain-job.test.ts +160 -0
  297. package/src/plugins/defaults/memory-v3-shadow/__tests__/orchestrate.test.ts +335 -316
  298. package/src/plugins/defaults/memory-v3-shadow/__tests__/pool-select.test.ts +145 -53
  299. package/src/plugins/defaults/memory-v3-shadow/__tests__/render-injection.test.ts +38 -1
  300. package/src/plugins/defaults/memory-v3-shadow/__tests__/selection-log-store.test.ts +18 -8
  301. package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-integration.test.ts +112 -71
  302. package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-plugin.test.ts +192 -51
  303. package/src/plugins/defaults/memory-v3-shadow/__tests__/types.test.ts +4 -16
  304. package/src/plugins/defaults/memory-v3-shadow/card.test.ts +173 -0
  305. package/src/plugins/defaults/memory-v3-shadow/card.ts +116 -0
  306. package/src/plugins/defaults/memory-v3-shadow/core-set.test.ts +104 -0
  307. package/src/plugins/defaults/memory-v3-shadow/core-set.ts +59 -0
  308. package/src/plugins/defaults/memory-v3-shadow/ever-injected-store.test.ts +305 -0
  309. package/src/plugins/defaults/memory-v3-shadow/ever-injected-store.ts +278 -0
  310. package/src/plugins/defaults/memory-v3-shadow/hot-set.test.ts +138 -0
  311. package/src/plugins/defaults/memory-v3-shadow/hot-set.ts +85 -0
  312. package/src/plugins/defaults/memory-v3-shadow/injector.ts +331 -24
  313. package/src/plugins/defaults/memory-v3-shadow/maintain-job.ts +119 -13
  314. package/src/plugins/defaults/memory-v3-shadow/orchestrate.ts +169 -114
  315. package/src/plugins/defaults/memory-v3-shadow/page-content.ts +47 -16
  316. package/src/plugins/defaults/memory-v3-shadow/pool-select.ts +144 -66
  317. package/src/plugins/defaults/memory-v3-shadow/prune.test.ts +758 -0
  318. package/src/plugins/defaults/memory-v3-shadow/prune.ts +471 -0
  319. package/src/plugins/defaults/memory-v3-shadow/render-injection.ts +68 -16
  320. package/src/plugins/defaults/memory-v3-shadow/selection-log-store.ts +19 -12
  321. package/src/plugins/defaults/memory-v3-shadow/shadow-plugin.ts +96 -43
  322. package/src/plugins/defaults/memory-v3-shadow/types.ts +34 -17
  323. package/src/plugins/defaults/title-generate/hooks/stop.ts +9 -11
  324. package/src/plugins/pipeline.ts +8 -5
  325. package/src/plugins/types.ts +5 -51
  326. package/src/providers/cache-control.ts +26 -0
  327. package/src/providers/inference/__tests__/base-url-route-validation.test.ts +1 -2
  328. package/src/providers/model-catalog.ts +13 -1
  329. package/src/providers/openai/__tests__/tool-choice-mapping.test.ts +147 -0
  330. package/src/providers/openai/chat-completions-provider.ts +46 -0
  331. package/src/providers/openai/responses-provider.ts +45 -0
  332. package/src/providers/registry.ts +0 -8
  333. package/src/runtime/__tests__/agent-wake.test.ts +0 -1
  334. package/src/runtime/agent-wake.ts +13 -0
  335. package/src/runtime/routes/__tests__/acp-routes.test.ts +151 -0
  336. package/src/runtime/routes/__tests__/consolidation-routes.test.ts +12 -50
  337. package/src/runtime/routes/__tests__/conversation-surface-routes.test.ts +322 -0
  338. package/src/runtime/routes/__tests__/inference-provider-connection-routes.test.ts +0 -62
  339. package/src/runtime/routes/__tests__/plugins-routes.test.ts +31 -11
  340. package/src/runtime/routes/acp-routes.test.ts +106 -0
  341. package/src/runtime/routes/acp-routes.ts +248 -2
  342. package/src/runtime/routes/browser-tabs-routes.ts +1 -1
  343. package/src/runtime/routes/channel-verification-routes.ts +14 -5
  344. package/src/runtime/routes/chatgpt-subscription-auth-routes.ts +0 -14
  345. package/src/runtime/routes/consolidation-routes.ts +6 -82
  346. package/src/runtime/routes/conversation-list-routes.ts +6 -0
  347. package/src/runtime/routes/conversation-management-routes.ts +71 -0
  348. package/src/runtime/routes/identity-routes.ts +8 -0
  349. package/src/runtime/routes/inbound-stages/acl-enforcement.ts +33 -34
  350. package/src/runtime/routes/inference-provider-connection-routes.ts +0 -45
  351. package/src/runtime/routes/plugins-routes.ts +34 -45
  352. package/src/runtime/routes/schedule-routes.ts +43 -5
  353. package/src/runtime/routes/settings-routes.ts +140 -15
  354. package/src/runtime/routes/skills-routes.ts +18 -6
  355. package/src/runtime/services/__tests__/conversation-serializer.test.ts +140 -0
  356. package/src/runtime/services/conversation-serializer.ts +38 -1
  357. package/src/runtime/verification-outbound-actions.ts +147 -2
  358. package/src/runtime/verification-templates.ts +29 -3
  359. package/src/schedule/schedule-store.ts +19 -0
  360. package/src/skills/catalog-install.ts +77 -13
  361. package/src/tasks/task-scheduler.ts +1 -0
  362. package/src/telemetry/types.ts +66 -1
  363. package/src/telemetry/usage-telemetry-reporter.test.ts +542 -13
  364. package/src/telemetry/usage-telemetry-reporter.ts +213 -20
  365. package/src/tools/browser/__tests__/browser-execution-acquire.test.ts +49 -2
  366. package/src/tools/browser/__tests__/browser-status.test.ts +29 -5
  367. package/src/tools/browser/browser-execution.ts +27 -9
  368. package/src/tools/browser/cdp-client/__tests__/factory.test.ts +380 -4
  369. package/src/tools/browser/cdp-client/__tests__/host-bridge-cdp-client.test.ts +107 -0
  370. package/src/tools/browser/cdp-client/__tests__/types.test.ts +6 -1
  371. package/src/tools/browser/cdp-client/factory.ts +217 -17
  372. package/src/tools/browser/cdp-client/host-bridge-cdp-client.ts +67 -0
  373. package/src/tools/browser/cdp-client/types.ts +22 -2
  374. package/src/tools/credential-execution/make-authenticated-request.ts +2 -1
  375. package/src/tools/credential-execution/manage-secure-command-tool.ts +169 -164
  376. package/src/tools/credential-execution/run-authenticated-command.ts +2 -1
  377. package/src/tools/executor.ts +39 -7
  378. package/src/tools/registry.ts +387 -5
  379. package/src/tools/schedule/create.ts +16 -0
  380. package/src/tools/schedule/list.ts +12 -4
  381. package/src/tools/schedule/update.ts +12 -0
  382. package/src/tools/skills/load.ts +11 -6
  383. package/src/tools/terminal/safe-env.ts +2 -0
  384. package/src/tools/types.ts +65 -9
  385. package/src/tools/workspace-tools/loader.ts +673 -0
  386. package/src/usage/attribution.ts +28 -0
  387. package/src/util/device-id.ts +17 -3
  388. package/src/util/platform.ts +16 -0
  389. package/tsconfig.plugin-api.json +13 -0
  390. package/src/__tests__/plugin-external-api.test.ts +0 -68
  391. package/src/__tests__/plugin-skill-contribution.test.ts +0 -355
  392. package/src/daemon/message-types/browser.ts +0 -10
  393. package/src/notifications/__tests__/emit-signal-home-feed.test.ts +0 -187
  394. package/src/plugins/defaults/memory-v3-shadow/__tests__/fixtures/eval-turns.json +0 -36
  395. package/src/plugins/defaults/memory-v3-shadow/__tests__/fixtures/live-turns.json +0 -37
  396. package/src/plugins/defaults/memory-v3-shadow/__tests__/working-set-eviction.test.ts +0 -106
  397. package/src/plugins/defaults/memory-v3-shadow/__tests__/working-set-skeleton.test.ts +0 -44
  398. package/src/plugins/defaults/memory-v3-shadow/working-set.ts +0 -91
  399. package/src/plugins/external-api.ts +0 -114
  400. package/src/plugins/plugin-skill-contributions.ts +0 -292
@@ -6,7 +6,10 @@ import type {
6
6
  CheckpointInfo,
7
7
  } from "../agent/loop.js";
8
8
  import { AgentLoop } from "../agent/loop.js";
9
+ import type { StopContext } from "../plugin-api/types.js";
10
+ import { REFUSAL_FALLBACK_TEXT } from "../plugins/defaults/empty-response/hooks/post-model-call.js";
9
11
  import { resetPluginRegistryAndRegisterDefaults } from "../plugins/defaults/index.js";
12
+ import { registerPlugin } from "../plugins/registry.js";
10
13
  import type {
11
14
  ContentBlock,
12
15
  Message,
@@ -14,6 +17,7 @@ import type {
14
17
  ProviderResponse,
15
18
  ToolDefinition,
16
19
  } from "../providers/types.js";
20
+ import { ContextOverflowError } from "../providers/types.js";
17
21
  import {
18
22
  createMockProvider,
19
23
  textResponse,
@@ -64,6 +68,7 @@ describe("AgentLoop", () => {
64
68
 
65
69
  const events: AgentEvent[] = [];
66
70
  const { history } = await loop.run({
71
+ requestId: "test-request",
67
72
  messages: [userMessage],
68
73
  onEvent: collectEvents(events),
69
74
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -102,6 +107,7 @@ describe("AgentLoop", () => {
102
107
  });
103
108
  const events: AgentEvent[] = [];
104
109
  const { history } = await loop.run({
110
+ requestId: "test-request",
105
111
  messages: [userMessage],
106
112
  onEvent: collectEvents(events),
107
113
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -239,6 +245,7 @@ describe("AgentLoop", () => {
239
245
  toolExecutor: toolExecutor,
240
246
  });
241
247
  const { history } = await loop.run({
248
+ requestId: "test-request",
242
249
  messages: [userMessage],
243
250
  onEvent: () => {},
244
251
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -268,6 +275,7 @@ describe("AgentLoop", () => {
268
275
  tools: dummyTools,
269
276
  });
270
277
  const { history } = await loop.run({
278
+ requestId: "test-request",
271
279
  messages: [userMessage],
272
280
  onEvent: () => {},
273
281
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -295,6 +303,7 @@ describe("AgentLoop", () => {
295
303
  });
296
304
  const events: AgentEvent[] = [];
297
305
  const { history } = await loop.run({
306
+ requestId: "test-request",
298
307
  messages: [userMessage],
299
308
  onEvent: collectEvents(events),
300
309
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -311,6 +320,333 @@ describe("AgentLoop", () => {
311
320
  ).toBe("API rate limit exceeded");
312
321
  });
313
322
 
323
+ // 5b. Reactive context-overflow recovery
324
+ //
325
+ // When recovery is enabled the loop calibrates the estimator from the
326
+ // rejection, loops back, and the budget gate forwards the overflow signal
327
+ // into the compaction plugin's reduction ladder. That manager-owned path —
328
+ // single-rung recovery, multi-rung escalation, and the terminal
329
+ // `context_too_large` / `budget_yield_unrecovered` exit on exhaustion — is
330
+ // exercised end-to-end against a manager harness in
331
+ // `conversation-agent-loop-overflow.test.ts`. Loop-level coverage against a
332
+ // manager-store stub lands with the suite migration.
333
+ test.todo(
334
+ "drives the reduction ladder on a context-overflow rejection",
335
+ () => {},
336
+ );
337
+
338
+ test("surfaces a context-overflow error when recovery is unavailable", async () => {
339
+ /**
340
+ * Overflow recovery is gated on the budget gate being active. When it is
341
+ * disabled (e.g. agent wakes pass `overflowRecovery.enabled = false`, or no
342
+ * context window resolves) there is no ladder to drive, so a context
343
+ * overflow surfaces as an error on the first call rather than looping.
344
+ */
345
+
346
+ // GIVEN a provider that rejects every call as too large and a loop with no
347
+ // overflow-recovery budget gate
348
+ const { provider, calls } = createMockProvider([
349
+ new ContextOverflowError("prompt is too long", "mock", {
350
+ actualTokens: 250_000,
351
+ }),
352
+ ]);
353
+ const loop = new AgentLoop({
354
+ provider,
355
+ systemPrompt: "system",
356
+ conversationId: "test-conversation",
357
+ });
358
+ const events: AgentEvent[] = [];
359
+
360
+ // WHEN the loop runs
361
+ await loop.run({
362
+ requestId: "test-request",
363
+ messages: [userMessage],
364
+ onEvent: collectEvents(events),
365
+ trust: { sourceChannel: "vellum", trustClass: "unknown" },
366
+ });
367
+
368
+ // THEN it surfaced the error on the first call without retrying
369
+ expect(calls).toHaveLength(1);
370
+ expect(events.filter((e) => e.type === "error")).toHaveLength(1);
371
+ });
372
+
373
+ // 5c. Reactive ordering-error recovery
374
+ //
375
+ // When the provider rejects a call because the history violates
376
+ // tool-use/tool-result pairing or role-alternation rules, the loop runs
377
+ // `deepRepairHistory` directly and re-issues the call. It is bounded to a
378
+ // single repair pass per provider call, so a second consecutive ordering
379
+ // rejection falls through to the error path instead of looping forever.
380
+ test("repairs the history and re-issues on a provider ordering rejection", async () => {
381
+ // GIVEN a provider that rejects the first call with an ordering error and
382
+ // succeeds on the retry
383
+ const { provider, calls } = createMockProvider([
384
+ new Error(
385
+ "tool_result blocks that are not immediately after a tool_use block",
386
+ ),
387
+ textResponse("recovered"),
388
+ ]);
389
+ const loop = new AgentLoop({
390
+ provider,
391
+ systemPrompt: "system",
392
+ conversationId: "test-conversation",
393
+ });
394
+ const events: AgentEvent[] = [];
395
+
396
+ // WHEN the loop runs
397
+ const { history } = await loop.run({
398
+ messages: [userMessage],
399
+ onEvent: collectEvents(events),
400
+ trust: { sourceChannel: "vellum", trustClass: "unknown" },
401
+ requestId: "test-request",
402
+ });
403
+
404
+ // THEN it re-issued the call after repairing, surfaced no error, and
405
+ // appended the recovered assistant response
406
+ expect(calls).toHaveLength(2);
407
+ expect(events.filter((e) => e.type === "error")).toHaveLength(0);
408
+ expect(history[history.length - 1]).toEqual({
409
+ role: "assistant",
410
+ content: [{ type: "text", text: "recovered" }],
411
+ });
412
+ });
413
+
414
+ test("surfaces an ordering error after a single repair fails to recover", async () => {
415
+ // GIVEN a provider that rejects every call with an ordering error
416
+ const { provider, calls } = createMockProvider([
417
+ new Error("tool_use_id provided without a matching tool_result"),
418
+ ]);
419
+ const loop = new AgentLoop({
420
+ provider,
421
+ systemPrompt: "system",
422
+ conversationId: "test-conversation",
423
+ });
424
+ const events: AgentEvent[] = [];
425
+
426
+ // WHEN the loop runs
427
+ await loop.run({
428
+ messages: [userMessage],
429
+ onEvent: collectEvents(events),
430
+ trust: { sourceChannel: "vellum", trustClass: "unknown" },
431
+ requestId: "test-request",
432
+ });
433
+
434
+ // THEN it repaired once, re-issued once, then surfaced the error rather
435
+ // than retrying again
436
+ expect(calls).toHaveLength(2);
437
+ expect(events.filter((e) => e.type === "error")).toHaveLength(1);
438
+ });
439
+
440
+ test("reports only the retry's output as new messages when deep-repair shrinks the base", async () => {
441
+ // The wrapper reconstructs the persisted transcript from the loop's
442
+ // `newMessages` tail, so after `deepRepairHistory` removes a base message
443
+ // the new-message boundary must track the re-normalized history — otherwise
444
+ // the recovered assistant response is sliced off and lost.
445
+
446
+ // GIVEN an input whose leading assistant message deep-repair strips, a
447
+ // provider that rejects the first call on ordering grounds, then succeeds
448
+ const leadingAssistantMessage: Message = {
449
+ role: "assistant",
450
+ content: [{ type: "text", text: "stale leading turn" }],
451
+ };
452
+ const { provider } = createMockProvider([
453
+ new Error(
454
+ "tool_result blocks that are not immediately after a tool_use block",
455
+ ),
456
+ textResponse("recovered"),
457
+ ]);
458
+ const loop = new AgentLoop({
459
+ provider,
460
+ systemPrompt: "system",
461
+ conversationId: "test-conversation",
462
+ });
463
+
464
+ // WHEN the loop runs
465
+ const { history, newMessages } = await loop.run({
466
+ messages: [leadingAssistantMessage, userMessage],
467
+ onEvent: () => {},
468
+ trust: { sourceChannel: "vellum", trustClass: "unknown" },
469
+ requestId: "test-request",
470
+ });
471
+
472
+ // THEN deep-repair dropped the leading assistant from the base, and
473
+ // `newMessages` is exactly the retry's appended response
474
+ expect(history).toEqual([
475
+ userMessage,
476
+ { role: "assistant", content: [{ type: "text", text: "recovered" }] },
477
+ ]);
478
+ expect(newMessages).toEqual([
479
+ { role: "assistant", content: [{ type: "text", text: "recovered" }] },
480
+ ]);
481
+ });
482
+
483
+ test("repairs a fresh ordering rejection on a later run after a prior run left the bound spent", async () => {
484
+ // The ordering-repair bound is owned by the history-repair plugin's
485
+ // per-conversation state. A run scopes that bound, so the loop clears it on
486
+ // entry — otherwise a direct `run` caller that bypasses the daemon
487
+ // orchestrator (e.g. an agent wake) inherits a spent bound from a prior run
488
+ // on the same conversation and surfaces a repairable rejection instead of
489
+ // repairing it.
490
+
491
+ // GIVEN a first run that rejects on ordering grounds, repairs once, then
492
+ // rejects again — ending with the bound spent and the error surfaced
493
+ const { provider: firstProvider, calls: firstCalls } = createMockProvider([
494
+ new Error("tool_use_id provided without a matching tool_result"),
495
+ new Error("tool_use_id provided without a matching tool_result"),
496
+ ]);
497
+ const conversationId = "ordering-bound-across-runs";
498
+ const firstLoop = new AgentLoop({
499
+ provider: firstProvider,
500
+ systemPrompt: "system",
501
+ conversationId,
502
+ });
503
+ const firstEvents: AgentEvent[] = [];
504
+ await firstLoop.run({
505
+ messages: [userMessage],
506
+ onEvent: collectEvents(firstEvents),
507
+ trust: { sourceChannel: "vellum", trustClass: "unknown" },
508
+ requestId: "test-request",
509
+ });
510
+ expect(firstCalls).toHaveLength(2);
511
+ expect(firstEvents.filter((e) => e.type === "error")).toHaveLength(1);
512
+
513
+ // AND a second run on the same conversation whose first call rejects on
514
+ // ordering grounds, then succeeds on the retry
515
+ const { provider: secondProvider, calls: secondCalls } = createMockProvider(
516
+ [
517
+ new Error(
518
+ "tool_result blocks that are not immediately after a tool_use block",
519
+ ),
520
+ textResponse("recovered"),
521
+ ],
522
+ );
523
+ const secondLoop = new AgentLoop({
524
+ provider: secondProvider,
525
+ systemPrompt: "system",
526
+ conversationId,
527
+ });
528
+ const secondEvents: AgentEvent[] = [];
529
+
530
+ // WHEN the second run executes
531
+ const { history } = await secondLoop.run({
532
+ messages: [userMessage],
533
+ onEvent: collectEvents(secondEvents),
534
+ trust: { sourceChannel: "vellum", trustClass: "unknown" },
535
+ requestId: "test-request",
536
+ });
537
+
538
+ // THEN it repaired and re-issued (the bound did not leak across runs),
539
+ // surfaced no error, and appended the recovered response
540
+ expect(secondCalls).toHaveLength(2);
541
+ expect(secondEvents.filter((e) => e.type === "error")).toHaveLength(0);
542
+ expect(history[history.length - 1]).toEqual({
543
+ role: "assistant",
544
+ content: [{ type: "text", text: "recovered" }],
545
+ });
546
+ });
547
+
548
+ test("isolates a throwing stop hook on a successful stop and still emits the terminal exit", async () => {
549
+ // The `stop` chain is the loop's terminal teardown: a throwing teardown
550
+ // hook (e.g. a third-party plugin) is logged and contained, never escalated
551
+ // into a turn error. The turn's real outcome stands — a successful no-tool
552
+ // stop — and the terminal `agent_loop_exit` still fires so the run stays
553
+ // observable.
554
+
555
+ // GIVEN a registered stop hook that always throws, and a provider that
556
+ // returns a successful no-tool response (a successful stop)
557
+ registerPlugin({
558
+ manifest: { name: "throwing-stop", version: "0.0.1" },
559
+ hooks: {
560
+ stop: async () => {
561
+ throw new Error("stop hook boom");
562
+ },
563
+ },
564
+ });
565
+ const { provider, calls } = createMockProvider([textResponse("done")]);
566
+ const loop = new AgentLoop({
567
+ provider,
568
+ systemPrompt: "system",
569
+ conversationId: "test-conversation",
570
+ });
571
+ const events: AgentEvent[] = [];
572
+
573
+ // WHEN the loop runs
574
+ let rejected = false;
575
+ await loop
576
+ .run({
577
+ messages: [userMessage],
578
+ onEvent: collectEvents(events),
579
+ trust: { sourceChannel: "vellum", trustClass: "unknown" },
580
+ requestId: "test-request",
581
+ })
582
+ .catch(() => {
583
+ rejected = true;
584
+ });
585
+
586
+ // THEN the loop did not reject, did not re-issue the call, surfaced no
587
+ // error event, and still emitted the terminal exit with the real reason
588
+ expect(rejected).toBe(false);
589
+ expect(calls).toHaveLength(1);
590
+ expect(events.filter((e) => e.type === "error")).toHaveLength(0);
591
+ const exitEvents = events.filter((e) => e.type === "agent_loop_exit");
592
+ expect(exitEvents).toHaveLength(1);
593
+ expect(exitEvents[0]).toMatchObject({ reason: "no_tool_calls" });
594
+ });
595
+
596
+ test("fires the stop teardown chain on a checkpoint handoff without emitting agent_loop_exit", async () => {
597
+ // A handoff pauses this run so the orchestrator can drain a queued message,
598
+ // then re-enters with a fresh run. It still ends *this* turn, so the
599
+ // terminal `stop` chain must fire — that is what runs per-turn teardown
600
+ // (e.g. clearing the recovery bounds the post-model-call hooks set) before
601
+ // the queued message is processed. But `agent_loop_exit` must NOT be
602
+ // emitted, because the handoff is a control transfer, not a terminal exit.
603
+
604
+ // GIVEN a stop hook recording every exit reason it observes, and a provider
605
+ // whose first reply requests a tool so the loop reaches a checkpoint
606
+ const stopReasons: string[] = [];
607
+ registerPlugin({
608
+ manifest: { name: "recording-stop", version: "0.0.1" },
609
+ hooks: {
610
+ stop: async (ctx: StopContext) => {
611
+ stopReasons.push(ctx.exitReason);
612
+ },
613
+ },
614
+ });
615
+ const { provider } = createMockProvider([
616
+ toolUseResponse("t1", "read_file", { path: "/a.txt" }),
617
+ textResponse("never reached"),
618
+ ]);
619
+ const toolExecutor = async () => ({ content: "ok", isError: false });
620
+ const loop = new AgentLoop({
621
+ provider,
622
+ systemPrompt: "system",
623
+ conversationId: "test-conversation",
624
+ tools: dummyTools,
625
+ toolExecutor,
626
+ });
627
+ const events: AgentEvent[] = [];
628
+
629
+ // AND a checkpoint callback that hands off at the first opportunity
630
+ const onCheckpoint = (_info: CheckpointInfo): CheckpointDecision =>
631
+ "handoff";
632
+
633
+ // WHEN the loop runs and yields control at the checkpoint
634
+ const { exitReason } = await loop.run({
635
+ requestId: "test-request",
636
+ messages: [userMessage],
637
+ onEvent: collectEvents(events),
638
+ trust: { sourceChannel: "vellum", trustClass: "unknown" },
639
+ onCheckpoint,
640
+ });
641
+
642
+ // THEN the terminal stop chain fired exactly once with the handoff reason
643
+ // (teardown ran), the run reported the handoff, and no agent_loop_exit was
644
+ // emitted
645
+ expect(stopReasons).toEqual(["checkpoint_handoff"]);
646
+ expect(exitReason).toBe("handoff");
647
+ expect(events.filter((e) => e.type === "agent_loop_exit")).toHaveLength(0);
648
+ });
649
+
314
650
  // 6. Abort signal — verify the loop respects AbortSignal
315
651
  test("stops when abort signal is triggered before provider call", async () => {
316
652
  const controller = new AbortController();
@@ -323,6 +659,7 @@ describe("AgentLoop", () => {
323
659
  conversationId: "test-conversation",
324
660
  });
325
661
  const { history } = await loop.run({
662
+ requestId: "test-request",
326
663
  messages: [userMessage],
327
664
  onEvent: () => {},
328
665
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -360,6 +697,7 @@ describe("AgentLoop", () => {
360
697
  toolExecutor: toolExecutor,
361
698
  });
362
699
  const { history } = await loop.run({
700
+ requestId: "test-request",
363
701
  messages: [userMessage],
364
702
  onEvent: () => {},
365
703
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -413,6 +751,7 @@ describe("AgentLoop", () => {
413
751
  });
414
752
  const start = Date.now();
415
753
  const { history } = await loop.run({
754
+ requestId: "test-request",
416
755
  messages: [userMessage],
417
756
  onEvent: () => {},
418
757
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -462,6 +801,7 @@ describe("AgentLoop", () => {
462
801
 
463
802
  const events: AgentEvent[] = [];
464
803
  await loop.run({
804
+ requestId: "test-request",
465
805
  messages: [userMessage],
466
806
  onEvent: collectEvents(events),
467
807
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -484,6 +824,7 @@ describe("AgentLoop", () => {
484
824
 
485
825
  const events: AgentEvent[] = [];
486
826
  await loop.run({
827
+ requestId: "test-request",
487
828
  messages: [userMessage],
488
829
  onEvent: collectEvents(events),
489
830
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -510,6 +851,7 @@ describe("AgentLoop", () => {
510
851
 
511
852
  const events: AgentEvent[] = [];
512
853
  await loop.run({
854
+ requestId: "test-request",
513
855
  messages: [userMessage],
514
856
  onEvent: collectEvents(events),
515
857
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -540,6 +882,7 @@ describe("AgentLoop", () => {
540
882
 
541
883
  const events: AgentEvent[] = [];
542
884
  await loop.run({
885
+ requestId: "test-request",
543
886
  messages: [userMessage],
544
887
  onEvent: collectEvents(events),
545
888
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -590,6 +933,7 @@ describe("AgentLoop", () => {
590
933
  });
591
934
 
592
935
  await loop.run({
936
+ requestId: "test-request",
593
937
  messages: [userMessage],
594
938
  onEvent: () => {},
595
939
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -632,6 +976,7 @@ describe("AgentLoop", () => {
632
976
  });
633
977
  const events: AgentEvent[] = [];
634
978
  await loop.run({
979
+ requestId: "test-request",
635
980
  messages: [userMessage],
636
981
  onEvent: collectEvents(events),
637
982
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -660,6 +1005,7 @@ describe("AgentLoop", () => {
660
1005
  });
661
1006
 
662
1007
  await loop.run({
1008
+ requestId: "test-request",
663
1009
  messages: [userMessage],
664
1010
  onEvent: () => {},
665
1011
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -679,6 +1025,7 @@ describe("AgentLoop", () => {
679
1025
  });
680
1026
 
681
1027
  await loop.run({
1028
+ requestId: "test-request",
682
1029
  messages: [userMessage],
683
1030
  onEvent: () => {},
684
1031
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -744,6 +1091,7 @@ describe("AgentLoop", () => {
744
1091
  });
745
1092
  const events: AgentEvent[] = [];
746
1093
  const { history } = await loop.run({
1094
+ requestId: "test-request",
747
1095
  messages: [userMessage],
748
1096
  onEvent: collectEvents(events),
749
1097
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -845,6 +1193,7 @@ describe("AgentLoop", () => {
845
1193
  });
846
1194
  const events: AgentEvent[] = [];
847
1195
  const { history } = await loop.run({
1196
+ requestId: "test-request",
848
1197
  messages: [userMessage],
849
1198
  onEvent: collectEvents(events),
850
1199
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -925,6 +1274,7 @@ describe("AgentLoop", () => {
925
1274
  });
926
1275
  const events: AgentEvent[] = [];
927
1276
  await loop.run({
1277
+ requestId: "test-request",
928
1278
  messages: [userMessage],
929
1279
  onEvent: collectEvents(events),
930
1280
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -970,6 +1320,7 @@ describe("AgentLoop", () => {
970
1320
  };
971
1321
 
972
1322
  await loop.run({
1323
+ requestId: "test-request",
973
1324
  messages: [userMessage],
974
1325
  onEvent: () => {},
975
1326
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1006,6 +1357,7 @@ describe("AgentLoop", () => {
1006
1357
  const onCheckpoint = (): CheckpointDecision => "continue";
1007
1358
 
1008
1359
  const { history } = await loop.run({
1360
+ requestId: "test-request",
1009
1361
  messages: [userMessage],
1010
1362
  onEvent: () => {},
1011
1363
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1036,9 +1388,10 @@ describe("AgentLoop", () => {
1036
1388
  toolExecutor: toolExecutor,
1037
1389
  });
1038
1390
 
1039
- const onCheckpoint = (): CheckpointDecision => "budget";
1391
+ const onCheckpoint = (): CheckpointDecision => "handoff";
1040
1392
 
1041
1393
  const { history } = await loop.run({
1394
+ requestId: "test-request",
1042
1395
  messages: [userMessage],
1043
1396
  onEvent: () => {},
1044
1397
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1070,6 +1423,7 @@ describe("AgentLoop", () => {
1070
1423
  });
1071
1424
 
1072
1425
  const { history } = await loop.run({
1426
+ requestId: "test-request",
1073
1427
  messages: [userMessage],
1074
1428
  onEvent: () => {},
1075
1429
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1106,6 +1460,7 @@ describe("AgentLoop", () => {
1106
1460
  };
1107
1461
 
1108
1462
  await loop.run({
1463
+ requestId: "test-request",
1109
1464
  messages: [userMessage],
1110
1465
  onEvent: () => {},
1111
1466
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1137,6 +1492,7 @@ describe("AgentLoop", () => {
1137
1492
  };
1138
1493
 
1139
1494
  const { history } = await loop.run({
1495
+ requestId: "test-request",
1140
1496
  messages: [userMessage],
1141
1497
  onEvent: () => {},
1142
1498
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1199,6 +1555,7 @@ describe("AgentLoop", () => {
1199
1555
  };
1200
1556
 
1201
1557
  await loop.run({
1558
+ requestId: "test-request",
1202
1559
  messages: [userMessage],
1203
1560
  onEvent: () => {},
1204
1561
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1235,11 +1592,12 @@ describe("AgentLoop", () => {
1235
1592
  const onCheckpoint = (checkpoint: CheckpointInfo): CheckpointDecision => {
1236
1593
  checkpoints.push(checkpoint);
1237
1594
  // Yield on turn 3 (0-indexed)
1238
- return checkpoint.turnIndex === 3 ? "budget" : "continue";
1595
+ return checkpoint.turnIndex === 3 ? "handoff" : "continue";
1239
1596
  };
1240
1597
 
1241
1598
  const events: AgentEvent[] = [];
1242
1599
  const { history } = await loop.run({
1600
+ requestId: "test-request",
1243
1601
  messages: [userMessage],
1244
1602
  onEvent: collectEvents(events),
1245
1603
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1308,10 +1666,11 @@ describe("AgentLoop", () => {
1308
1666
 
1309
1667
  const onCheckpoint = (checkpoint: CheckpointInfo): CheckpointDecision => {
1310
1668
  // Yield on the second turn (turnIndex 1)
1311
- return checkpoint.turnIndex === 1 ? "budget" : "continue";
1669
+ return checkpoint.turnIndex === 1 ? "handoff" : "continue";
1312
1670
  };
1313
1671
 
1314
1672
  const { history } = await loop.run({
1673
+ requestId: "test-request",
1315
1674
  messages: [userMessage],
1316
1675
  onEvent: () => {},
1317
1676
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1339,6 +1698,7 @@ describe("AgentLoop", () => {
1339
1698
  });
1340
1699
 
1341
1700
  await loop.run({
1701
+ requestId: "test-request",
1342
1702
  messages: [userMessage],
1343
1703
  onEvent: () => {},
1344
1704
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1381,6 +1741,7 @@ describe("AgentLoop", () => {
1381
1741
  resolveTools: resolveTools,
1382
1742
  });
1383
1743
  await loop.run({
1744
+ requestId: "test-request",
1384
1745
  messages: [userMessage],
1385
1746
  onEvent: () => {},
1386
1747
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1420,6 +1781,7 @@ describe("AgentLoop", () => {
1420
1781
  resolveTools: resolveTools,
1421
1782
  });
1422
1783
  await loop.run({
1784
+ requestId: "test-request",
1423
1785
  messages: [userMessage],
1424
1786
  onEvent: () => {},
1425
1787
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1484,6 +1846,7 @@ describe("AgentLoop", () => {
1484
1846
  resolveTools: resolveTools,
1485
1847
  });
1486
1848
  await loop.run({
1849
+ requestId: "test-request",
1487
1850
  messages: [userMessage],
1488
1851
  onEvent: () => {},
1489
1852
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1514,6 +1877,7 @@ describe("AgentLoop", () => {
1514
1877
  resolveTools: resolveTools,
1515
1878
  });
1516
1879
  await loop.run({
1880
+ requestId: "test-request",
1517
1881
  messages: [userMessage],
1518
1882
  onEvent: () => {},
1519
1883
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1551,6 +1915,7 @@ describe("AgentLoop", () => {
1551
1915
  });
1552
1916
  const events: AgentEvent[] = [];
1553
1917
  const { history } = await loop.run({
1918
+ requestId: "test-request",
1554
1919
  messages: [userMessage],
1555
1920
  onEvent: collectEvents(events),
1556
1921
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1607,6 +1972,7 @@ describe("AgentLoop", () => {
1607
1972
  });
1608
1973
  const events: AgentEvent[] = [];
1609
1974
  const { history } = await loop.run({
1975
+ requestId: "test-request",
1610
1976
  messages: [userMessage],
1611
1977
  onEvent: collectEvents(events),
1612
1978
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1671,6 +2037,7 @@ describe("AgentLoop", () => {
1671
2037
  });
1672
2038
  const events: AgentEvent[] = [];
1673
2039
  const { history } = await loop.run({
2040
+ requestId: "test-request",
1674
2041
  messages: [userMessage],
1675
2042
  onEvent: collectEvents(events),
1676
2043
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1744,6 +2111,7 @@ describe("AgentLoop", () => {
1744
2111
  });
1745
2112
  const events: AgentEvent[] = [];
1746
2113
  await loop.run({
2114
+ requestId: "test-request",
1747
2115
  messages: [userMessage],
1748
2116
  onEvent: collectEvents(events),
1749
2117
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1783,6 +2151,7 @@ describe("AgentLoop", () => {
1783
2151
  });
1784
2152
  const events: AgentEvent[] = [];
1785
2153
  const { history } = await loop.run({
2154
+ requestId: "test-request",
1786
2155
  messages: [userMessage],
1787
2156
  onEvent: collectEvents(events),
1788
2157
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1833,6 +2202,7 @@ describe("AgentLoop", () => {
1833
2202
  toolExecutor: toolExecutor,
1834
2203
  });
1835
2204
  await loop.run({
2205
+ requestId: "test-request",
1836
2206
  messages: [userMessage],
1837
2207
  onEvent: () => {},
1838
2208
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1898,6 +2268,7 @@ describe("AgentLoop", () => {
1898
2268
  toolExecutor: toolExecutor,
1899
2269
  });
1900
2270
  await loop.run({
2271
+ requestId: "test-request",
1901
2272
  messages: [userMessage],
1902
2273
  onEvent: () => {},
1903
2274
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -1952,6 +2323,7 @@ describe("AgentLoop", () => {
1952
2323
  });
1953
2324
  const events: AgentEvent[] = [];
1954
2325
  const { history } = await loop.run({
2326
+ requestId: "test-request",
1955
2327
  messages: [userMessage],
1956
2328
  onEvent: collectEvents(events),
1957
2329
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -2034,6 +2406,7 @@ describe("AgentLoop", () => {
2034
2406
  });
2035
2407
  const events: AgentEvent[] = [];
2036
2408
  const { history } = await loop.run({
2409
+ requestId: "test-request",
2037
2410
  messages: [userMessage],
2038
2411
  onEvent: collectEvents(events),
2039
2412
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -2100,6 +2473,7 @@ describe("AgentLoop", () => {
2100
2473
  });
2101
2474
  const events: AgentEvent[] = [];
2102
2475
  const { history } = await loop.run({
2476
+ requestId: "test-request",
2103
2477
  messages: [userMessage],
2104
2478
  onEvent: collectEvents(events),
2105
2479
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -2114,12 +2488,23 @@ describe("AgentLoop", () => {
2114
2488
  );
2115
2489
  expect(messageCompletes).toHaveLength(2);
2116
2490
 
2117
- // The last assistant message in history is the empty one
2491
+ // An organic empty `end_turn` (not a refusal) is left as-is — the fallback
2492
+ // is refusal-specific, so the exhausted empty turn stays empty.
2118
2493
  const lastAssistant = [...history]
2119
2494
  .reverse()
2120
2495
  .find((m) => m.role === "assistant");
2121
2496
  expect(lastAssistant).toBeDefined();
2122
2497
  expect(lastAssistant!.content).toEqual([]);
2498
+
2499
+ // AND no fallback text is streamed for a non-refusal empty turn.
2500
+ const textDeltas = events.filter((e) => e.type === "text_delta");
2501
+ expect(
2502
+ textDeltas.some(
2503
+ (e) =>
2504
+ (e as { type: "text_delta"; text: string }).text ===
2505
+ REFUSAL_FALLBACK_TEXT,
2506
+ ),
2507
+ ).toBe(false);
2123
2508
  });
2124
2509
 
2125
2510
  test("does not retry empty response on first turn (no prior tool use)", async () => {
@@ -2139,6 +2524,7 @@ describe("AgentLoop", () => {
2139
2524
  });
2140
2525
  const events: AgentEvent[] = [];
2141
2526
  const { history } = await loop.run({
2527
+ requestId: "test-request",
2142
2528
  messages: [userMessage],
2143
2529
  onEvent: collectEvents(events),
2144
2530
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -2146,7 +2532,192 @@ describe("AgentLoop", () => {
2146
2532
 
2147
2533
  // Should NOT retry — this is the first turn with no tool use history
2148
2534
  expect(calls).toHaveLength(1);
2149
- expect(history).toHaveLength(2); // user + empty assistant
2535
+ // user + assistant; an organic empty `end_turn` is left as-is (the fallback
2536
+ // is refusal-specific), so the empty turn stays empty.
2537
+ expect(history).toHaveLength(2);
2538
+ expect(history[1].content).toEqual([]);
2539
+ });
2540
+
2541
+ // Refusal stop boundary — the provider zeroed the response with
2542
+ // `stopReason: "refusal"`. The default `stop` hook rewrites the empty turn
2543
+ // into a user-facing fallback (no retry — a safety-classifier refusal
2544
+ // re-fires on a re-query), and the loop persists + streams it rather than the
2545
+ // empty assistant bubble the user would otherwise see.
2546
+ test("rewrites a refusal into a user-facing fallback without retrying", async () => {
2547
+ // GIVEN a provider that refuses (empty content, stopReason "refusal").
2548
+ const refusalResponse: ProviderResponse = {
2549
+ content: [],
2550
+ model: "mock-model",
2551
+ usage: { inputTokens: 10, outputTokens: 0 },
2552
+ stopReason: "refusal",
2553
+ };
2554
+ const { provider, calls } = createMockProvider([
2555
+ refusalResponse,
2556
+ refusalResponse,
2557
+ ]);
2558
+ const loop = new AgentLoop({
2559
+ provider: provider,
2560
+ systemPrompt: "system",
2561
+ conversationId: "test-conversation",
2562
+ });
2563
+
2564
+ // WHEN the loop runs.
2565
+ const events: AgentEvent[] = [];
2566
+ const { history } = await loop.run({
2567
+ requestId: "test-request",
2568
+ messages: [userMessage],
2569
+ onEvent: collectEvents(events),
2570
+ trust: { sourceChannel: "vellum", trustClass: "unknown" },
2571
+ });
2572
+
2573
+ // THEN the loop calls the provider once: the refusal is rewritten in place
2574
+ // with no nudge-driven retry.
2575
+ expect(calls).toHaveLength(1);
2576
+
2577
+ // AND the user-visible assistant turn is the fallback, not an empty bubble.
2578
+ const lastAssistant = [...history]
2579
+ .reverse()
2580
+ .find((m) => m.role === "assistant");
2581
+ expect(lastAssistant!.content).toEqual([
2582
+ { type: "text", text: REFUSAL_FALLBACK_TEXT },
2583
+ ]);
2584
+
2585
+ // AND the fallback is streamed so the client renders it.
2586
+ const textDeltas = events.filter((e) => e.type === "text_delta");
2587
+ expect(
2588
+ textDeltas.some(
2589
+ (e) =>
2590
+ (e as { type: "text_delta"; text: string }).text ===
2591
+ REFUSAL_FALLBACK_TEXT,
2592
+ ),
2593
+ ).toBe(true);
2594
+ });
2595
+
2596
+ // The fallback must not pile a spurious apology beneath a real answer: when
2597
+ // an earlier turn this run already produced visible text (text alongside a
2598
+ // tool call), a refusal on the trailing turn after the tool result must not
2599
+ // rewrite the turn even though a refusal would normally trigger it.
2600
+ test("does not substitute a fallback when the run already produced visible text", async () => {
2601
+ // GIVEN a first turn with text + a tool call, then a refusal after the
2602
+ // tool result.
2603
+ const textPlusToolUse: ProviderResponse = {
2604
+ content: [
2605
+ { type: "text", text: "Here is your answer." },
2606
+ {
2607
+ type: "tool_use",
2608
+ id: "t1",
2609
+ name: "read_file",
2610
+ input: { path: "/a" },
2611
+ },
2612
+ ],
2613
+ model: "mock-model",
2614
+ usage: { inputTokens: 10, outputTokens: 5 },
2615
+ stopReason: "tool_use",
2616
+ };
2617
+ const refusalResponse: ProviderResponse = {
2618
+ content: [],
2619
+ model: "mock-model",
2620
+ usage: { inputTokens: 10, outputTokens: 0 },
2621
+ stopReason: "refusal",
2622
+ };
2623
+ const { provider } = createMockProvider([textPlusToolUse, refusalResponse]);
2624
+ const loop = new AgentLoop({
2625
+ provider: provider,
2626
+ systemPrompt: "system",
2627
+ conversationId: "test-conversation",
2628
+ tools: dummyTools,
2629
+ toolExecutor: async () => ({ content: "data", isError: false }),
2630
+ });
2631
+
2632
+ // WHEN the loop runs.
2633
+ const events: AgentEvent[] = [];
2634
+ const { history } = await loop.run({
2635
+ requestId: "test-request",
2636
+ messages: [userMessage],
2637
+ onEvent: collectEvents(events),
2638
+ trust: { sourceChannel: "vellum", trustClass: "unknown" },
2639
+ });
2640
+
2641
+ // THEN no fallback text is emitted or persisted — the real answer stands
2642
+ // alone and the trailing empty turn stays empty.
2643
+ const fallbackEmitted = events.some(
2644
+ (e) =>
2645
+ e.type === "text_delta" &&
2646
+ (e as { type: "text_delta"; text: string }).text ===
2647
+ REFUSAL_FALLBACK_TEXT,
2648
+ );
2649
+ expect(fallbackEmitted).toBe(false);
2650
+ const fallbackInHistory = history.some(
2651
+ (m) =>
2652
+ m.role === "assistant" &&
2653
+ m.content.some(
2654
+ (b) =>
2655
+ b.type === "text" &&
2656
+ "text" in b &&
2657
+ (b as { text: string }).text === REFUSAL_FALLBACK_TEXT,
2658
+ ),
2659
+ );
2660
+ expect(fallbackInHistory).toBe(false);
2661
+ });
2662
+
2663
+ // A native web-search turn lands at the stop boundary with no `tool_use` and
2664
+ // no visible text, but with `stopReason: "end_turn"` (not a refusal) and
2665
+ // `server_tool_use`/`web_search_tool_result` blocks that render the search
2666
+ // card. Because the fallback is refusal-specific, it must not fire here and
2667
+ // the server-tool blocks must persist untouched.
2668
+ test("does not substitute a fallback for a server-tool turn with no text", async () => {
2669
+ // GIVEN a turn carrying only native web-search blocks (no text, no
2670
+ // client-side tool_use).
2671
+ const serverToolBlocks: ContentBlock[] = [
2672
+ {
2673
+ type: "server_tool_use",
2674
+ id: "srv1",
2675
+ name: "web_search",
2676
+ input: { query: "weather" },
2677
+ },
2678
+ {
2679
+ type: "web_search_tool_result",
2680
+ tool_use_id: "srv1",
2681
+ content: [{ title: "result" }],
2682
+ },
2683
+ ];
2684
+ const webSearchResponse: ProviderResponse = {
2685
+ content: serverToolBlocks,
2686
+ model: "mock-model",
2687
+ usage: { inputTokens: 10, outputTokens: 5 },
2688
+ stopReason: "end_turn",
2689
+ };
2690
+ const { provider, calls } = createMockProvider([webSearchResponse]);
2691
+ const loop = new AgentLoop({
2692
+ provider: provider,
2693
+ systemPrompt: "system",
2694
+ conversationId: "test-conversation",
2695
+ });
2696
+
2697
+ // WHEN the loop runs.
2698
+ const events: AgentEvent[] = [];
2699
+ const { history } = await loop.run({
2700
+ requestId: "test-request",
2701
+ messages: [userMessage],
2702
+ onEvent: collectEvents(events),
2703
+ trust: { sourceChannel: "vellum", trustClass: "unknown" },
2704
+ });
2705
+
2706
+ // THEN the provider is called once (no nudge for a non-refusal turn) and
2707
+ // the server-tool blocks are persisted untouched, with no fallback added.
2708
+ expect(calls).toHaveLength(1);
2709
+ const lastAssistant = [...history]
2710
+ .reverse()
2711
+ .find((m) => m.role === "assistant");
2712
+ expect(lastAssistant!.content).toEqual(serverToolBlocks);
2713
+
2714
+ const fallbackEmitted = events.some(
2715
+ (e) =>
2716
+ e.type === "text_delta" &&
2717
+ (e as { type: "text_delta"; text: string }).text ===
2718
+ REFUSAL_FALLBACK_TEXT,
2719
+ );
2720
+ expect(fallbackEmitted).toBe(false);
2150
2721
  });
2151
2722
 
2152
2723
  // PR 6: callSite threading from AgentLoop.run() into provider config.
@@ -2161,6 +2732,7 @@ describe("AgentLoop", () => {
2161
2732
  conversationId: "test-conversation",
2162
2733
  });
2163
2734
  await loop.run({
2735
+ requestId: "test-request",
2164
2736
  messages: [userMessage],
2165
2737
  onEvent: () => {},
2166
2738
  trust: { sourceChannel: "vellum", trustClass: "unknown" },
@@ -2180,6 +2752,7 @@ describe("AgentLoop", () => {
2180
2752
  conversationId: "test-conversation",
2181
2753
  });
2182
2754
  await loop.run({
2755
+ requestId: "test-request",
2183
2756
  messages: [userMessage],
2184
2757
  onEvent: () => {},
2185
2758
  trust: { sourceChannel: "vellum", trustClass: "unknown" },