@vellumai/assistant 0.11.4 → 0.11.5-staging.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (367) hide show
  1. package/AGENTS.md +2 -2
  2. package/ARCHITECTURE.md +29 -4
  3. package/README.md +1 -1
  4. package/docs/architecture/memory.md +17 -5
  5. package/docs/architecture/security.md +96 -152
  6. package/docs/guardian-request-flow.md +13 -2
  7. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/package.json +1 -0
  8. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/__tests__/url-normalization.test.ts +135 -0
  9. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/index.ts +1 -0
  10. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/url-normalization.ts +107 -0
  11. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/package.json +1 -0
  12. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/__tests__/url-normalization.test.ts +135 -0
  13. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/index.ts +1 -0
  14. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/url-normalization.ts +107 -0
  15. package/node_modules/@vellumai/gateway-client/src/gateway-ipc-contracts.ts +166 -1
  16. package/node_modules/@vellumai/gateway-client/src/http-delivery.ts +7 -14
  17. package/node_modules/@vellumai/gateway-client/src/index.ts +2 -0
  18. package/node_modules/@vellumai/gateway-client/src/outbound-contract.ts +29 -31
  19. package/node_modules/@vellumai/service-contracts/package.json +1 -0
  20. package/node_modules/@vellumai/service-contracts/src/__tests__/url-normalization.test.ts +135 -0
  21. package/node_modules/@vellumai/service-contracts/src/index.ts +1 -0
  22. package/node_modules/@vellumai/service-contracts/src/url-normalization.ts +107 -0
  23. package/openapi.yaml +193 -3
  24. package/package.json +1 -1
  25. package/src/__tests__/anthropic-provider.test.ts +10 -3
  26. package/src/__tests__/approval-interception-trust-gates.test.ts +40 -0
  27. package/src/__tests__/background-workers-disk-pressure.test.ts +1 -1
  28. package/src/__tests__/channel-approval-routes.test.ts +4 -0
  29. package/src/__tests__/channel-inbound-disk-pressure.test.ts +5 -6
  30. package/src/__tests__/channel-reply-delivery.test.ts +26 -13
  31. package/src/__tests__/checker.test.ts +20 -314
  32. package/src/__tests__/cli-memory-v2-reembed-skills.test.ts +6 -2
  33. package/src/__tests__/client-os-metadata-persistence.test.ts +16 -0
  34. package/src/__tests__/compactor-retained-image-validation.test.ts +142 -0
  35. package/src/__tests__/config-loader-backfill.test.ts +19 -10
  36. package/src/__tests__/container-cpu-sampler.test.ts +153 -0
  37. package/src/__tests__/context-search-memory-v2-source.test.ts +0 -1
  38. package/src/__tests__/conversation-attachments.test.ts +0 -1
  39. package/src/__tests__/conversation-error.test.ts +17 -0
  40. package/src/__tests__/conversation-lifecycle.test.ts +20 -26
  41. package/src/__tests__/conversation-pairing.test.ts +186 -0
  42. package/src/__tests__/conversation-runtime-assembly.test.ts +21 -2
  43. package/src/__tests__/credential-broker-server-use.test.ts +24 -5
  44. package/src/__tests__/credential-routes.test.ts +22 -3
  45. package/src/__tests__/credential-security-invariants.test.ts +2 -1
  46. package/src/__tests__/daemon-credential-client.test.ts +88 -0
  47. package/src/__tests__/default-plugin-names-guard.test.ts +12 -1
  48. package/src/__tests__/dm-backfill.test.ts +63 -0
  49. package/src/__tests__/dynamic-skill-background-guard.test.ts +0 -2
  50. package/src/__tests__/edit-propagation.test.ts +107 -4
  51. package/src/__tests__/events-client-registration.test.ts +28 -0
  52. package/src/__tests__/file-write-tool.test.ts +4 -2
  53. package/src/__tests__/guardian-prompt-notice-privacy.test.ts +133 -0
  54. package/src/__tests__/guardian-verify-setup-skill-regression.test.ts +143 -97
  55. package/src/__tests__/helpers/gateway-classify-mock.ts +27 -6
  56. package/src/__tests__/host-proxy-interface.test.ts +11 -1
  57. package/src/__tests__/host-shell-tool.test.ts +23 -4
  58. package/src/__tests__/image-conversion.test.ts +143 -1
  59. package/src/__tests__/inline-skill-load-permissions.test.ts +47 -36
  60. package/src/__tests__/input-repairs.test.ts +207 -0
  61. package/src/__tests__/live-workspace-guard.test.ts +75 -0
  62. package/src/__tests__/managed-profile-guard.test.ts +4 -2
  63. package/src/__tests__/mcp-abort-signal.test.ts +1 -1
  64. package/src/__tests__/mcp-client-auth.test.ts +1 -1
  65. package/src/__tests__/mcp-tool-annotations-risk.test.ts +1 -1
  66. package/src/__tests__/media-resolve-image-validation.test.ts +310 -0
  67. package/src/__tests__/mtime-cache.test.ts +2 -0
  68. package/src/__tests__/notification-vellum-adapter.test.ts +124 -1
  69. package/src/__tests__/openai-provider.test.ts +22 -0
  70. package/src/__tests__/personal-memory-auth-bypass.test.ts +22 -6
  71. package/src/__tests__/platform-bash-auto-approve.test.ts +0 -4
  72. package/src/__tests__/platform.test.ts +16 -1
  73. package/src/__tests__/plugin-api-resolve-credential.test.ts +22 -13
  74. package/src/__tests__/plugin-api-shim.test.ts +5 -0
  75. package/src/__tests__/plugin-api-store-credential.test.ts +268 -0
  76. package/src/__tests__/plugin-config-data-migration.test.ts +8 -1
  77. package/src/__tests__/plugin-disabled-state.test.ts +2 -0
  78. package/src/__tests__/plugin-effective-enabled-set.test.ts +55 -3
  79. package/src/__tests__/plugin-execution-context.test.ts +73 -0
  80. package/src/__tests__/plugin-import-boundary-guard.test.ts +0 -1
  81. package/src/__tests__/plugin-import-boundary-reverse-guard.test.ts +0 -1
  82. package/src/__tests__/reaction-persistence.test.ts +200 -7
  83. package/src/__tests__/require-fresh-approval.test.ts +0 -4
  84. package/src/__tests__/resource-pressure-guard.test.ts +428 -0
  85. package/src/__tests__/resource-pressure-routes.test.ts +113 -0
  86. package/src/__tests__/risk-classification-boundary-guard.test.ts +115 -0
  87. package/src/__tests__/run-conversation-turn-persistence.test.ts +27 -1
  88. package/src/__tests__/secret-routes-platform-proxy.test.ts +7 -0
  89. package/src/__tests__/skill-tool-factory.test.ts +161 -0
  90. package/src/__tests__/tool-execution-pipeline.benchmark.test.ts +1 -1
  91. package/src/__tests__/tool-executor-lifecycle-events.test.ts +132 -5
  92. package/src/__tests__/tool-executor.test.ts +63 -38
  93. package/src/__tests__/tool-policy.test.ts +97 -1
  94. package/src/__tests__/trusted-contact-inline-approval-integration.test.ts +7 -4
  95. package/src/__tests__/user-plugin-loader.test.ts +2 -0
  96. package/src/__tests__/validate-input.test.ts +243 -7
  97. package/src/__tests__/verification-control-plane-policy.test.ts +0 -2
  98. package/src/__tests__/voice-scoped-grant-consumer.test.ts +2 -0
  99. package/src/__tests__/voice-session-bridge.test.ts +42 -0
  100. package/src/acp/__tests__/acp-claude-oauth.test.ts +97 -29
  101. package/src/acp/__tests__/prepare-agent-env.test.ts +376 -153
  102. package/src/acp/acp-claude-oauth.ts +38 -21
  103. package/src/acp/prepare-agent-env.ts +146 -55
  104. package/src/agent/loop-tool-dedup.test.ts +161 -0
  105. package/src/agent/loop.ts +68 -22
  106. package/src/api/constants/app-tools.ts +34 -0
  107. package/src/api/events/resource-pressure-status-changed.ts +53 -0
  108. package/src/api/index.ts +19 -0
  109. package/src/api/responses/resource-pressure-status.ts +25 -0
  110. package/src/approvals/guardian-channel-delivery.ts +70 -6
  111. package/src/approvals/guardian-request-resolvers.ts +32 -66
  112. package/src/calls/__tests__/voice-session-bridge.test.ts +114 -0
  113. package/src/calls/voice-session-bridge.ts +60 -6
  114. package/src/channels/__tests__/message-audience.test.ts +49 -0
  115. package/src/channels/__tests__/types.test.ts +17 -3
  116. package/src/channels/message-audience.ts +40 -0
  117. package/src/channels/types.ts +16 -12
  118. package/src/cli/__tests__/catalog-search-help.test.ts +50 -0
  119. package/src/cli/commands/__tests__/channel-verification-sessions.test.ts +31 -0
  120. package/src/cli/commands/__tests__/conversations-search.test.ts +289 -0
  121. package/src/cli/commands/__tests__/conversations-slack.test.ts +1 -0
  122. package/src/cli/commands/__tests__/inference-providers.test.ts +68 -0
  123. package/src/cli/commands/__tests__/keys.test.ts +89 -8
  124. package/src/cli/commands/__tests__/plugins.test.ts +57 -1
  125. package/src/cli/commands/channel-verification-sessions.help.ts +11 -9
  126. package/src/cli/commands/channel-verification-sessions.ts +11 -21
  127. package/src/cli/commands/conversations.help.ts +34 -0
  128. package/src/cli/commands/conversations.ts +125 -0
  129. package/src/cli/commands/inference-providers.ts +9 -4
  130. package/src/cli/commands/inference.help.ts +9 -4
  131. package/src/cli/commands/keys.help.ts +13 -1
  132. package/src/cli/commands/keys.ts +26 -15
  133. package/src/cli/commands/memory/__tests__/memory-v2.test.ts +57 -7
  134. package/src/cli/commands/memory/index.help.ts +41 -27
  135. package/src/cli/commands/memory/index.ts +2 -0
  136. package/src/cli/commands/memory/memory-v2.ts +58 -54
  137. package/src/cli/commands/memory/memory-validate.ts +18 -0
  138. package/src/cli/commands/monitoring.ts +1 -0
  139. package/src/cli/commands/plugins.help.ts +28 -31
  140. package/src/cli/commands/plugins.ts +10 -0
  141. package/src/cli/commands/skills.help.ts +13 -16
  142. package/src/cli/commands/trust.ts +3 -14
  143. package/src/cli/lib/__tests__/install-from-github.test.ts +38 -0
  144. package/src/cli/lib/__tests__/merge-plugin-tree.test.ts +34 -0
  145. package/src/cli/lib/__tests__/plugin-surfaces.test.ts +32 -1
  146. package/src/cli/lib/__tests__/toggle-plugin.test.ts +2 -1
  147. package/src/cli/lib/__tests__/upgrade-plugin.test.ts +64 -0
  148. package/src/cli/lib/bundled-marketplace.json +13 -0
  149. package/src/cli/lib/daemon-credential-client.ts +25 -3
  150. package/src/cli/lib/install-from-github.ts +35 -24
  151. package/src/cli/lib/merge-plugin-tree.ts +22 -4
  152. package/src/cli/lib/plugin-surfaces.ts +27 -0
  153. package/src/config/__tests__/balanced-model-experiment.test.ts +7 -7
  154. package/src/config/__tests__/default-profile-catalog.test.ts +3 -3
  155. package/src/config/default-profile-catalog.ts +5 -6
  156. package/src/config/feature-flag-registry.json +32 -32
  157. package/src/config/webhook-routing.ts +8 -0
  158. package/src/context/compactor.ts +31 -2
  159. package/src/daemon/__tests__/provider-rejection-log-fields.test.ts +272 -0
  160. package/src/daemon/conversation-agent-loop-handlers.ts +4 -1
  161. package/src/daemon/conversation-error.ts +13 -0
  162. package/src/daemon/conversation-messaging.ts +22 -2
  163. package/src/daemon/conversation-process.ts +11 -0
  164. package/src/daemon/conversation-store.ts +10 -0
  165. package/src/daemon/conversation-surfaces.ts +21 -8
  166. package/src/daemon/conversation.ts +12 -10
  167. package/src/daemon/lifecycle.ts +6 -2
  168. package/src/daemon/message-types/conversations.ts +5 -7
  169. package/src/daemon/provider-rejection-log-fields.ts +124 -0
  170. package/src/daemon/resource-pressure-guard-lifecycle.ts +66 -0
  171. package/src/daemon/resource-pressure-guard.ts +387 -0
  172. package/src/daemon/shutdown-handlers.ts +2 -0
  173. package/src/daemon/startup-error.ts +16 -0
  174. package/src/daemon/trust-context.ts +14 -10
  175. package/src/daemon/unsendable-image-notice.ts +76 -0
  176. package/src/hooks/registry.ts +65 -51
  177. package/src/ipc/__tests__/socket-path.test.ts +15 -7
  178. package/src/ipc/gateway-client.test.ts +1 -1
  179. package/src/ipc/gateway-client.ts +14 -21
  180. package/src/ipc/socket-cleanup.ts +2 -11
  181. package/src/mcp/__tests__/mcp-auth-orchestrator.test.ts +1 -1
  182. package/src/messaging/provider-message-metadata.ts +128 -0
  183. package/src/messaging/providers/__tests__/transport-dispatch.test.ts +116 -68
  184. package/src/messaging/providers/channel-transport.ts +90 -13
  185. package/src/messaging/providers/discord/send.ts +16 -0
  186. package/src/messaging/providers/discord/transport.ts +11 -1
  187. package/src/messaging/providers/index.ts +109 -16
  188. package/src/messaging/providers/slack/message-metadata.ts +45 -0
  189. package/src/messaging/providers/slack/render-transcript.ts +1 -4
  190. package/src/messaging/providers/slack/send.test.ts +6 -9
  191. package/src/messaging/providers/slack/send.ts +40 -28
  192. package/src/messaging/providers/slack/transport.ts +22 -26
  193. package/src/messaging/providers/telegram-bot/send.test.ts +2 -8
  194. package/src/messaging/providers/telegram-bot/transport.ts +3 -6
  195. package/src/messaging/read-provider-metadata.test.ts +157 -0
  196. package/src/messaging/read-provider-metadata.ts +51 -0
  197. package/src/notifications/__tests__/assistant-reply-producer.test.ts +94 -0
  198. package/src/notifications/__tests__/edit-notification.test.ts +174 -2
  199. package/src/notifications/__tests__/home-feed-side-effect.test.ts +536 -4
  200. package/src/notifications/adapters/macos.ts +55 -0
  201. package/src/notifications/adapters/slack.ts +10 -5
  202. package/src/notifications/assistant-reply-producer.ts +37 -4
  203. package/src/notifications/conversation-pairing.ts +111 -21
  204. package/src/notifications/edit-notification.ts +38 -8
  205. package/src/notifications/emit-signal.ts +12 -14
  206. package/src/notifications/home-feed-side-effect.ts +218 -15
  207. package/src/notifications/types.ts +5 -0
  208. package/src/permissions/AGENTS.md +16 -0
  209. package/src/permissions/checker.test.ts +75 -122
  210. package/src/permissions/checker.ts +44 -560
  211. package/src/permissions/confirmation-guardian-request.test.ts +28 -3
  212. package/src/permissions/confirmation-guardian-request.ts +5 -1
  213. package/src/persistence/conversation-crud.ts +23 -20
  214. package/src/persistence/conversation-queries.ts +14 -1
  215. package/src/persistence/conversation-types.ts +22 -0
  216. package/src/persistence/delivery-crud.ts +122 -15
  217. package/src/plugin-api/__tests__/import-graph-partial-mock.test.ts +67 -0
  218. package/src/plugin-api/conversation-turn.ts +27 -0
  219. package/src/plugin-api/credential-scope.test.ts +15 -0
  220. package/src/plugin-api/credential-scope.ts +14 -0
  221. package/src/plugin-api/index.ts +19 -1
  222. package/src/plugin-api/resolve-credential.ts +15 -10
  223. package/src/plugin-api/store-credential.ts +148 -0
  224. package/src/plugin-api/system-card.ts +36 -0
  225. package/src/plugin-api/vision-support.test.ts +82 -0
  226. package/src/plugin-api/vision-support.ts +14 -2
  227. package/src/plugins/defaults/image-fallback/__tests__/image-fallback.test.ts +361 -110
  228. package/src/plugins/defaults/image-fallback/__tests__/vision-recovery.test.ts +4 -0
  229. package/src/plugins/defaults/image-fallback/hooks/user-prompt-submit.ts +58 -0
  230. package/src/plugins/defaults/image-fallback/src/caption-blocks.ts +64 -23
  231. package/src/plugins/defaults/image-recovery/detect.ts +24 -1
  232. package/src/plugins/defaults/main.ts +15 -0
  233. package/src/plugins/defaults/memory/AGENTS.md +11 -3
  234. package/src/plugins/defaults/memory/__tests__/memory-tier-boundary-guard.test.ts +0 -1
  235. package/src/plugins/defaults/memory/graph-topology/pending-buffer.ts +9 -12
  236. package/src/plugins/defaults/memory/memory-retrospective-prompt.ts +1 -1
  237. package/src/plugins/defaults/memory/src/memory-v2-routes.ts +12 -10
  238. package/src/plugins/defaults/memory/substrate/__tests__/consolidation-job.test.ts +114 -0
  239. package/src/plugins/defaults/memory/substrate/__tests__/consolidation-prompt-flag-gating-guard.test.ts +8 -0
  240. package/src/plugins/defaults/memory/substrate/__tests__/edge-index.test.ts +0 -30
  241. package/src/plugins/defaults/memory/substrate/__tests__/ingest.test.ts +73 -1
  242. package/src/plugins/defaults/memory/substrate/__tests__/page-index.test.ts +37 -0
  243. package/src/plugins/defaults/memory/substrate/__tests__/page-links.test.ts +208 -0
  244. package/src/plugins/defaults/memory/substrate/__tests__/prompts-consolidation.test.ts +158 -1
  245. package/src/plugins/defaults/memory/substrate/__tests__/static-context.test.ts +1 -61
  246. package/src/plugins/defaults/memory/substrate/consolidation-job.ts +68 -6
  247. package/src/plugins/defaults/memory/substrate/edge-index.ts +0 -31
  248. package/src/plugins/defaults/memory/substrate/ingest.ts +59 -0
  249. package/src/plugins/defaults/memory/substrate/page-index.ts +35 -8
  250. package/src/plugins/defaults/memory/substrate/page-links.ts +133 -0
  251. package/src/plugins/defaults/memory/substrate/page-store.ts +10 -0
  252. package/src/plugins/defaults/memory/substrate/prompts/consolidation.ts +124 -34
  253. package/src/plugins/defaults/memory/substrate/static-context.ts +0 -29
  254. package/src/plugins/defaults/memory/v3/__tests__/shadow-plugin.test.ts +3 -1
  255. package/src/plugins/defaults/memory/v3/card.ts +9 -9
  256. package/src/plugins/defaults/memory/v3/core-set.test.ts +7 -0
  257. package/src/plugins/defaults/memory/v3/core-set.ts +5 -1
  258. package/src/plugins/defaults/memory/v3/edge.ts +13 -57
  259. package/src/plugins/mtime-cache.ts +1 -1
  260. package/src/plugins/pipeline.ts +14 -7
  261. package/src/plugins/plugin-execution-context.ts +35 -11
  262. package/src/plugins/plugin-tree-walk.ts +63 -4
  263. package/src/providers/anthropic/__tests__/pause-turn-continuation.test.ts +170 -0
  264. package/src/providers/anthropic/client.ts +332 -241
  265. package/src/providers/connection-resolution.ts +2 -1
  266. package/src/providers/content-blocks.ts +22 -0
  267. package/src/providers/gemini/client.ts +5 -2
  268. package/src/providers/inference/__tests__/adapter-factory-litellm.test.ts +47 -1
  269. package/src/providers/inference/__tests__/adapter-factory-ollama.test.ts +83 -0
  270. package/src/providers/inference/__tests__/adapter-factory-openai-compatible.test.ts +98 -1
  271. package/src/providers/inference/__tests__/base-url-route-validation.test.ts +38 -0
  272. package/src/providers/inference/__tests__/base-url-security.test.ts +10 -0
  273. package/src/providers/inference/__tests__/connections-ollama.test.ts +90 -0
  274. package/src/providers/inference/__tests__/missing-credential-guard.test.ts +117 -0
  275. package/src/providers/inference/adapter-factory.ts +33 -2
  276. package/src/providers/inference/auth.ts +13 -2
  277. package/src/providers/inference/credential-usage.ts +37 -0
  278. package/src/providers/inference/missing-credential-guard.ts +110 -0
  279. package/src/providers/inference/resolve-auth.ts +7 -7
  280. package/src/providers/media-resolve.ts +176 -15
  281. package/src/providers/openai/__tests__/chat-completions-provider-reasoning.test.ts +576 -46
  282. package/src/providers/openai/__tests__/tool-choice-mapping.test.ts +125 -2
  283. package/src/providers/openai/chat-completions-provider.ts +322 -42
  284. package/src/routes/route-host-protocol.ts +7 -0
  285. package/src/routes/worker.ts +31 -10
  286. package/src/runtime/AGENTS.md +35 -0
  287. package/src/runtime/__tests__/runtime-http-port-collision.test.ts +131 -0
  288. package/src/runtime/__tests__/web-presence.test.ts +234 -0
  289. package/src/runtime/assistant-event-hub.ts +38 -0
  290. package/src/runtime/channel-reply-delivery.ts +54 -31
  291. package/src/runtime/channel-retry-sweep.ts +6 -4
  292. package/src/runtime/effective-capabilities.test.ts +19 -0
  293. package/src/runtime/effective-capabilities.ts +25 -0
  294. package/src/runtime/guardian-reply-router.ts +6 -2
  295. package/src/runtime/http-errors.ts +1 -0
  296. package/src/runtime/http-server.ts +20 -5
  297. package/src/runtime/routes/__tests__/client-routes.test.ts +150 -0
  298. package/src/runtime/routes/__tests__/conversation-list-routes.test.ts +133 -0
  299. package/src/runtime/routes/__tests__/credential-delete-in-use.test.ts +155 -0
  300. package/src/runtime/routes/__tests__/inference-provider-connection-routes.test.ts +52 -1
  301. package/src/runtime/routes/__tests__/monitoring-routes.test.ts +1 -0
  302. package/src/runtime/routes/__tests__/pending-interactions-route.test.ts +175 -0
  303. package/src/runtime/routes/__tests__/plugins-routes.test.ts +88 -35
  304. package/src/runtime/routes/__tests__/surface-action-routes.test.ts +55 -6
  305. package/src/runtime/routes/__tests__/system-card-turn-termination.test.ts +107 -0
  306. package/src/runtime/routes/__tests__/user-route-dispatcher-host.test.ts +52 -1
  307. package/src/runtime/routes/__tests__/user-route-dispatcher.test.ts +75 -4
  308. package/src/runtime/routes/approval-routes.ts +37 -4
  309. package/src/runtime/routes/approval-strategies/guardian-text-engine-strategy.ts +16 -22
  310. package/src/runtime/routes/canned-message-complete.ts +68 -29
  311. package/src/runtime/routes/client-routes.ts +113 -0
  312. package/src/runtime/routes/conversation-list-routes.ts +22 -7
  313. package/src/runtime/routes/conversation-management-routes.ts +15 -9
  314. package/src/runtime/routes/conversation-routes.ts +36 -22
  315. package/src/runtime/routes/credential-in-use.ts +85 -0
  316. package/src/runtime/routes/credential-routes.ts +55 -85
  317. package/src/runtime/routes/guardian-approval-interception.ts +22 -22
  318. package/src/runtime/routes/guardian-approval-reply-helpers.ts +14 -12
  319. package/src/runtime/routes/identity-routes.ts +5 -121
  320. package/src/runtime/routes/inbound-message-handler.ts +71 -27
  321. package/src/runtime/routes/inbound-stages/acl-enforcement.ts +16 -17
  322. package/src/runtime/routes/inbound-stages/background-dispatch.test.ts +77 -50
  323. package/src/runtime/routes/inbound-stages/background-dispatch.ts +61 -97
  324. package/src/runtime/routes/inbound-stages/edit-intercept.ts +101 -77
  325. package/src/runtime/routes/inbound-stages/guardian-reply-intercept.ts +17 -8
  326. package/src/runtime/routes/inbound-stages/reaction-intercept.test.ts +91 -7
  327. package/src/runtime/routes/inbound-stages/reaction-intercept.ts +72 -70
  328. package/src/runtime/routes/index.ts +2 -0
  329. package/src/runtime/routes/inference-provider-connection-routes.ts +9 -7
  330. package/src/runtime/routes/monitoring-routes.ts +9 -1
  331. package/src/runtime/routes/plugins-routes.ts +17 -5
  332. package/src/runtime/routes/resource-pressure-routes.ts +22 -0
  333. package/src/runtime/routes/secret-routes.ts +39 -8
  334. package/src/runtime/routes/settings-routes.ts +6 -6
  335. package/src/runtime/routes/surface-action-routes.ts +20 -19
  336. package/src/runtime/routes/user-route-dispatcher.ts +31 -2
  337. package/src/runtime/routes/user-route-resolution.ts +18 -1
  338. package/src/runtime/slack-reply-session.test.ts +23 -13
  339. package/src/runtime/slack-reply-session.ts +38 -55
  340. package/src/runtime/web-presence.ts +90 -0
  341. package/src/skills/validate-input.ts +177 -40
  342. package/src/subagent/__tests__/consult-context-gating.test.ts +6 -10
  343. package/src/subagent/consult-context.ts +7 -23
  344. package/src/tools/credentials/broker.ts +24 -79
  345. package/src/tools/credentials/ref-parse.ts +35 -0
  346. package/src/tools/credentials/resolve.ts +4 -10
  347. package/src/tools/credentials/store.ts +168 -0
  348. package/src/tools/credentials/tool-policy.ts +67 -0
  349. package/src/tools/executor.ts +53 -15
  350. package/src/tools/network/__tests__/web-search.test.ts +41 -1
  351. package/src/tools/network/url-safety.ts +9 -16
  352. package/src/tools/network/web-search.ts +21 -3
  353. package/src/tools/permission-checker.ts +43 -52
  354. package/src/tools/schema-transforms.ts +40 -5
  355. package/src/tools/shared/input-repairs.ts +160 -0
  356. package/src/tools/skills/skill-tool-factory.ts +18 -4
  357. package/src/tools/subagent/spawn.ts +0 -1
  358. package/src/tools/tool-approval-handler.ts +16 -10
  359. package/src/tools/tool-types.ts +26 -13
  360. package/src/tools/types.ts +6 -6
  361. package/src/util/__tests__/cgroup-memory.test.ts +3 -0
  362. package/src/util/cgroup-memory.ts +3 -0
  363. package/src/util/container-cpu-sampler.ts +250 -0
  364. package/src/util/image-conversion.ts +178 -14
  365. package/src/util/platform.ts +93 -10
  366. package/src/permissions/ipc-risk-types.ts +0 -143
  367. package/src/permissions/risk-types.ts +0 -76
@@ -17,16 +17,20 @@ import {
17
17
  getSecureKeyAsync,
18
18
  setSecureKeyAsync,
19
19
  } from "../security/secure-keys.js";
20
+ import { getLogger } from "../util/logger.js";
20
21
  import {
21
22
  ACP_OAUTH_TOKEN_FIELD,
22
23
  ACP_SERVICE,
23
24
  classifyAnthropicToken,
24
25
  } from "./acp-credentials.js";
25
26
  import {
26
- acpSpawnCanReadCredential,
27
- grantAcpSpawnPolicy,
27
+ ACP_CLAUDE_OAUTH_USAGE_DESCRIPTION,
28
+ acpSpawnCredentialDenialReason,
29
+ repairAcpSpawnPolicy,
28
30
  } from "./prepare-agent-env.js";
29
31
 
32
+ const log = getLogger("acp:claude-oauth");
33
+
30
34
  /**
31
35
  * Verified Claude Code public OAuth client. PKCE-only (no client secret);
32
36
  * the single `user:inference` scope is what the ACP adapter's
@@ -109,8 +113,9 @@ export function parseManualClaudeCode(input: string): {
109
113
 
110
114
  /**
111
115
  * Store a captured Claude OAuth token in the `acp/claude_oauth_token` vault
112
- * field and provision the `acp_spawn` read policy so the broker can inject it
113
- * at spawn time. Throws when the backing store rejects the write.
116
+ * field and provision the policy the broker applies at spawn time: grant the
117
+ * `acp_spawn` read and lift any domain restriction. Throws when the backing
118
+ * store rejects the write.
114
119
  */
115
120
  export async function storeAcpClaudeToken(token: string): Promise<void> {
116
121
  const stored = await setSecureKeyAsync(
@@ -120,13 +125,13 @@ export async function storeAcpClaudeToken(token: string): Promise<void> {
120
125
  if (!stored) {
121
126
  throw new Error("Failed to store Claude OAuth token in secure storage.");
122
127
  }
123
- // Force-grant acp_spawn (union) rather than merely ensure it: an explicit
124
- // Connect is a deliberate opt-in to ACP, so this repairs a credential whose
125
- // explicit allowedTools omitted acp_spawn — otherwise the broker keeps denying
126
- // the spawn read and the Connect card dead-loops on every auto-continue.
127
- grantAcpSpawnPolicy(
128
+ // Repair rather than merely ensure the policy: an explicit Connect is a
129
+ // deliberate opt-in to ACP, so this widens a credential the broker would
130
+ // otherwise keep denying the spawn read on, which would dead-loop the Connect
131
+ // card on every auto-continue.
132
+ repairAcpSpawnPolicy(
128
133
  ACP_OAUTH_TOKEN_FIELD,
129
- "Claude OAuth token for ACP agent authentication",
134
+ ACP_CLAUDE_OAUTH_USAGE_DESCRIPTION,
130
135
  );
131
136
  }
132
137
 
@@ -142,20 +147,32 @@ export async function storeAcpClaudeToken(token: string): Promise<void> {
142
147
  * spawn, so keeping Connect offered (rather than self-dismissing) lets the user
143
148
  * repair the bad entry by connecting a real OAuth token.
144
149
  *
145
- * Likewise, a token the `acp_spawn` policy can't read (an explicit
146
- * `allowedTools` that omits `acp_spawn`) is NOT connected: the vault holds a
147
- * value but the spawn's broker read is denied, so self-dismissing the card would
148
- * hide the only repair CTA while every spawn keeps failing. Keep the card up in
149
- * that denied-policy case too.
150
+ * Likewise, a token the spawn's broker read would be denied (an explicit
151
+ * `allowedTools` that omits `acp_spawn`, or a domain-restricted policy) is NOT
152
+ * connected: the vault holds a value but every spawn fails, so self-dismissing
153
+ * the card would hide the only repair CTA. That half of the answer is delegated
154
+ * to `acpSpawnCredentialDenialReason`, which evaluates the exact policy the
155
+ * spawn-time broker read applies, so "connected" means precisely "the spawn
156
+ * would get this token". The token-shape guard stays here instead: the broker
157
+ * knows nothing about Anthropic token formats.
150
158
  */
151
159
  export async function hasAcpClaudeToken(): Promise<boolean> {
152
160
  const token = await getSecureKeyAsync(
153
161
  credentialKey(ACP_SERVICE, ACP_OAUTH_TOKEN_FIELD),
154
162
  );
155
- return (
156
- token != null &&
157
- token.length > 0 &&
158
- classifyAnthropicToken(token) !== "api_key" &&
159
- acpSpawnCanReadCredential(ACP_OAUTH_TOKEN_FIELD)
160
- );
163
+ if (token == null || token.length === 0) {
164
+ return false;
165
+ }
166
+ if (classifyAnthropicToken(token) === "api_key") {
167
+ return false;
168
+ }
169
+ const denialReason = acpSpawnCredentialDenialReason(ACP_OAUTH_TOKEN_FIELD);
170
+ if (denialReason !== undefined) {
171
+ log.debug(
172
+ { field: ACP_OAUTH_TOKEN_FIELD, reason: denialReason },
173
+ "Connect Claude status: token present but spawn read would be denied",
174
+ );
175
+ return false;
176
+ }
177
+ return true;
161
178
  }
@@ -25,9 +25,11 @@ import { basename } from "node:path";
25
25
  import { FailedDependencyError } from "../runtime/routes/errors.js";
26
26
  import { credentialBroker } from "../tools/credentials/broker.js";
27
27
  import {
28
+ type CredentialMetadata,
28
29
  getCredentialMetadata,
29
30
  upsertCredentialMetadata,
30
31
  } from "../tools/credentials/metadata-store.js";
32
+ import { serverUseDenialReason } from "../tools/credentials/tool-policy.js";
31
33
  import { getLogger } from "../util/logger.js";
32
34
  import {
33
35
  ACP_OAUTH_TOKEN_FIELD,
@@ -40,91 +42,148 @@ const log = getLogger("acp:prepare-agent-env");
40
42
 
41
43
  const ACP_SPAWN_TOOL = "acp_spawn";
42
44
 
45
+ /**
46
+ * `usageDescription` recorded on `acp/claude_oauth_token` when a record is
47
+ * created, shared by the spawn-time ensure and the Connect repair so the two
48
+ * paths can't describe the same credential differently.
49
+ */
50
+ export const ACP_CLAUDE_OAUTH_USAGE_DESCRIPTION =
51
+ "Claude OAuth token for ACP agent authentication";
52
+
43
53
  /**
44
54
  * Stable, machine-readable marker carried on the `FailedDependencyError.details`
45
55
  * when a `claude-agent-acp` spawn is missing `CLAUDE_CODE_OAUTH_TOKEN`. Threaded
46
56
  * through the tool result / error payload as a structured field so clients can
47
57
  * offer the inline "Connect Claude Code" flow instead of re-parsing the human
48
58
  * message string. Kept in lockstep with the web literal in
49
- * `clients/web/src/domains/chat/transcript/acp-connect-affordance.tsx`.
59
+ * `clients/web/src/domains/chat/utils/acp-connect.ts`.
50
60
  */
51
61
  export const ACP_CLAUDE_OAUTH_MISSING_CODE = "acp_claude_oauth_missing";
52
62
 
53
63
  /**
54
- * Ensure an `acp/<field>` credential has metadata that allows the
55
- * `acp_spawn` tool to read it, but only for legacy/unmanaged cases:
64
+ * The metadata an `acp/<field>` credential has once the `acp_spawn` read policy
65
+ * is ensured, given what is stored now. This is the single definition of that
66
+ * decision: {@link ensureAcpCredentialPolicy} persists the result and
67
+ * {@link acpSpawnCredentialDenialReason} evaluates it without writing.
68
+ *
69
+ * The policy is only repaired for legacy/unmanaged cases:
70
+ *
71
+ * - No metadata at all: a record with `allowedTools: ["acp_spawn"]` and the
72
+ * caller's `usageDescription`.
73
+ * - Metadata with an empty `allowedTools`: default provisioning path (user ran
74
+ * `credentials set` without `--allowed-tools`), so `acp_spawn` is added.
75
+ * - Metadata with a non-empty `allowedTools`: explicit policy set by the
76
+ * user/admin, returned as the very same object so callers can tell by
77
+ * identity that there is nothing to persist. It stands even when `acp_spawn`
78
+ * is absent; the broker denies the read and the caller decides whether that's
79
+ * fatal.
56
80
  *
57
- * - No metadata at all: create with `allowedTools: ["acp_spawn"]`.
58
- * - Metadata exists with an empty `allowedTools`: default provisioning
59
- * path (user ran `credentials set` without `--allowed-tools`), add it.
60
- * - Metadata exists with a non-empty `allowedTools`: explicit policy set
61
- * by the user/admin. Respect it even if `acp_spawn` is absent; the
62
- * broker will deny the read and the caller decides whether that's fatal.
81
+ * Everything else on an existing record, `allowedDomains` included, is
82
+ * preserved.
63
83
  */
64
- export function ensureAcpCredentialPolicy(
84
+ function projectEnsuredAcpPolicy(
85
+ meta: CredentialMetadata | undefined,
65
86
  field: string,
66
- usageDescription: string,
67
- ): void {
68
- const meta = getCredentialMetadata(ACP_SERVICE, field);
87
+ usageDescription?: string,
88
+ ): CredentialMetadata {
69
89
  if (!meta) {
70
- upsertCredentialMetadata(ACP_SERVICE, field, {
90
+ return {
91
+ credentialId: "",
92
+ service: ACP_SERVICE,
93
+ field,
71
94
  allowedTools: [ACP_SPAWN_TOOL],
95
+ allowedDomains: [],
72
96
  usageDescription,
73
- });
74
- return;
97
+ createdAt: 0,
98
+ updatedAt: 0,
99
+ };
75
100
  }
76
- const tools = meta.allowedTools ?? [];
77
- if (tools.length === 0) {
78
- upsertCredentialMetadata(ACP_SERVICE, field, {
79
- allowedTools: [ACP_SPAWN_TOOL],
80
- });
101
+ if ((meta.allowedTools ?? []).length === 0) {
102
+ return { ...meta, allowedTools: [ACP_SPAWN_TOOL] };
81
103
  }
104
+ return meta;
82
105
  }
83
106
 
84
107
  /**
85
- * Force-grant the `acp_spawn` read policy on `acp/<field>`, unioning it into any
86
- * existing `allowedTools`. Unlike {@link ensureAcpCredentialPolicy} (which
87
- * PRESERVES an explicit non-empty policy so a passive spawn can't silently widen
88
- * it), this is for the EXPLICIT Connect flow: a user connecting Claude is a
89
- * deliberate opt-in to `acp_spawn`, so granting it makes the CTA actually repair
90
- * a policy-denied credential instead of dead-looping the missing-token card.
108
+ * Bring the stored metadata for `acp/<field>` up to the policy
109
+ * {@link projectEnsuredAcpPolicy} describes, writing only the fields that
110
+ * projection decides and only when it differs from what is stored.
91
111
  */
92
- export function grantAcpSpawnPolicy(
112
+ export function ensureAcpCredentialPolicy(
93
113
  field: string,
94
114
  usageDescription: string,
95
115
  ): void {
96
116
  const meta = getCredentialMetadata(ACP_SERVICE, field);
97
- if (!meta) {
98
- upsertCredentialMetadata(ACP_SERVICE, field, {
99
- allowedTools: [ACP_SPAWN_TOOL],
100
- usageDescription,
101
- });
117
+ const ensured = projectEnsuredAcpPolicy(meta, field, usageDescription);
118
+ if (ensured === meta) {
102
119
  return;
103
120
  }
104
- const tools = meta.allowedTools ?? [];
105
- if (!tools.includes(ACP_SPAWN_TOOL)) {
106
- upsertCredentialMetadata(ACP_SERVICE, field, {
107
- allowedTools: [...tools, ACP_SPAWN_TOOL],
108
- });
109
- }
121
+ upsertCredentialMetadata(ACP_SERVICE, field, {
122
+ allowedTools: ensured.allowedTools,
123
+ usageDescription: ensured.usageDescription,
124
+ });
110
125
  }
111
126
 
112
127
  /**
113
- * Whether the `acp_spawn` broker read for `acp/<field>` would actually be
114
- * permitted, mirroring {@link ensureAcpCredentialPolicy}'s grant rules: a
115
- * missing or empty `allowedTools` is auto-granted `acp_spawn` at spawn time, so
116
- * it can read; a non-empty explicit policy is respected as-is, so it can read
117
- * only when it lists `acp_spawn`. Lets a connected-status check avoid reporting
118
- * "connected" for a token the spawn is policy-denied from reading (which would
119
- * otherwise hide the repair CTA and trap the user in a missing-token loop).
128
+ * Make `acp/<field>` readable by the spawn: union `acp_spawn` into any existing
129
+ * `allowedTools` and drop any domain restriction, in ONE write and only when the
130
+ * stored record fails either half. This is the whole repair the Connect flow
131
+ * performs, so a new dimension of {@link serverUseDenialReason} is repaired in
132
+ * exactly one place.
133
+ *
134
+ * Unlike {@link ensureAcpCredentialPolicy} (which PRESERVES an explicit non-empty
135
+ * policy so a passive spawn can't silently widen it), this is for the EXPLICIT
136
+ * Connect flow: a user connecting Claude is a deliberate opt-in to `acp_spawn`,
137
+ * so granting it makes the CTA actually repair a policy-denied credential instead
138
+ * of dead-looping the missing-token card. Domains are cleared under the same
139
+ * opt-in: this field is OAuth-only and server-use-only, and the broker refuses a
140
+ * domain-restricted credential server-side, so a lingering restriction would keep
141
+ * every spawn failing even after a successful connect.
120
142
  */
121
- export function acpSpawnCanReadCredential(field: string): boolean {
143
+ export function repairAcpSpawnPolicy(
144
+ field: string,
145
+ usageDescription: string,
146
+ ): void {
122
147
  const meta = getCredentialMetadata(ACP_SERVICE, field);
123
- if (!meta) {
124
- return true;
148
+ const tools = meta?.allowedTools ?? [];
149
+ const spawnAllowed = tools.includes(ACP_SPAWN_TOOL);
150
+ const domainUnrestricted = (meta?.allowedDomains ?? []).length === 0;
151
+ if (meta && spawnAllowed && domainUnrestricted) {
152
+ return;
125
153
  }
126
- const tools = meta.allowedTools ?? [];
127
- return tools.length === 0 || tools.includes(ACP_SPAWN_TOOL);
154
+ upsertCredentialMetadata(ACP_SERVICE, field, {
155
+ allowedTools: spawnAllowed ? tools : [...tools, ACP_SPAWN_TOOL],
156
+ allowedDomains: [],
157
+ // Only a fresh record takes the description; an existing one keeps its own.
158
+ ...(meta ? {} : { usageDescription }),
159
+ });
160
+ }
161
+
162
+ /**
163
+ * Why the `acp_spawn` broker read for `acp/<field>` would be denied, or
164
+ * `undefined` when it would be permitted. Lets a connected-status check avoid
165
+ * reporting "connected" for a token the spawn is policy-denied from reading
166
+ * (which would otherwise hide the repair CTA and trap the user in a
167
+ * missing-token loop).
168
+ *
169
+ * The verdict comes from `serverUseDenialReason`, the single policy source the
170
+ * broker itself consults, so the status check and the spawn read can never
171
+ * disagree. The stored metadata is first run through
172
+ * {@link projectEnsuredAcpPolicy}, the same repair the spawn persists, computed
173
+ * in memory and never written: this runs on a side-effect-free GET route, so it
174
+ * has to predict what the spawn's ensure-then-read sequence would do rather
175
+ * than perform it.
176
+ */
177
+ export function acpSpawnCredentialDenialReason(
178
+ field: string,
179
+ ): string | undefined {
180
+ const meta = getCredentialMetadata(ACP_SERVICE, field);
181
+ return serverUseDenialReason(
182
+ projectEnsuredAcpPolicy(meta, field),
183
+ ACP_SPAWN_TOOL,
184
+ ACP_SERVICE,
185
+ field,
186
+ );
128
187
  }
129
188
 
130
189
  /**
@@ -243,25 +302,57 @@ export async function prepareAgentEnv(
243
302
  };
244
303
 
245
304
  dropApiKeyOauthToken();
305
+ let missReason: string | undefined;
246
306
  if (!env.CLAUDE_CODE_OAUTH_TOKEN) {
247
- await injectCredential(
307
+ missReason = await injectCredential(
248
308
  env,
249
309
  ACP_OAUTH_TOKEN_FIELD,
250
310
  "CLAUDE_CODE_OAUTH_TOKEN",
251
- "Claude OAuth token for ACP agent authentication",
311
+ ACP_CLAUDE_OAUTH_USAGE_DESCRIPTION,
252
312
  );
253
313
  }
314
+ // Any api-key-shaped value still standing here came from the vault read:
315
+ // the config override was already dropped above, and the read only runs
316
+ // when the override left the var unset.
317
+ const storedValueIsApiKeyShaped =
318
+ env.CLAUDE_CODE_OAUTH_TOKEN !== undefined &&
319
+ classifyAnthropicToken(env.CLAUDE_CODE_OAUTH_TOKEN) === "api_key";
254
320
  dropApiKeyOauthToken();
255
321
  if (!env.CLAUDE_CODE_OAUTH_TOKEN) {
322
+ // The operator's record of WHY the spawn has no token. `missReason` is
323
+ // the broker's own reason string and the rest are policy verdicts, so no
324
+ // field can carry the credential value.
325
+ const policyDenialReason = acpSpawnCredentialDenialReason(
326
+ ACP_OAUTH_TOKEN_FIELD,
327
+ );
328
+ log.warn(
329
+ {
330
+ field: ACP_OAUTH_TOKEN_FIELD,
331
+ missReason,
332
+ policyBlocked: policyDenialReason !== undefined,
333
+ apiKeyShaped: storedValueIsApiKeyShaped,
334
+ },
335
+ "Claude OAuth token not injected for acp_spawn",
336
+ );
256
337
  // Carry the stable marker as structured `details` so the client renders
257
338
  // the inline "Connect Claude Code" card. The message itself is the tool
258
339
  // result the model reads at the failure moment, so it directs the model
259
340
  // AT that card and away from CLI/token-paste workarounds — otherwise the
260
341
  // model relays a `claude setup-token` / paste-a-token flow that the card
261
342
  // exists to replace. The CLI command stays only as a headless fallback.
343
+ // A policy-blocked read is a different repair story from an absent value,
344
+ // so the opening states which one happened. The guidance after it is
345
+ // shared: the Connect card fixes both.
346
+ const opening = policyDenialReason
347
+ ? "claude-agent-acp cannot read the Claude OAuth token: the credential " +
348
+ "policy on acp/claude_oauth_token blocks the acp_spawn read, so " +
349
+ "CLAUDE_CODE_OAUTH_TOKEN is not set for the spawn. Clicking Connect " +
350
+ "signs in again and repairs that policy. "
351
+ : "claude-agent-acp needs a Claude OAuth token (CLAUDE_CODE_OAUTH_TOKEN), " +
352
+ "which is not set. ";
262
353
  throw new FailedDependencyError(
263
- "claude-agent-acp needs a Claude OAuth token (CLAUDE_CODE_OAUTH_TOKEN), " +
264
- 'which is not set. The app shows the user an inline "Connect Claude ' +
354
+ opening +
355
+ 'The app shows the user an inline "Connect Claude ' +
265
356
  'Code" card. Reply with ONE short sentence: ask them to click Connect ' +
266
357
  "in that card to sign in, and tell them you'll continue automatically " +
267
358
  "once they're connected. Do NOT say where the card is — never say " +
@@ -0,0 +1,161 @@
1
+ /**
2
+ * Verifies the agent loop coalesces `tool_use` blocks by call id before it
3
+ * dispatches them: a provider that emits the same call twice under one id gets
4
+ * one execution, one `tool_use` block in history, and one correlated
5
+ * `tool_result`. Distinct ids for the same tool name still run independently.
6
+ * Drives the REAL loop, mocking only the provider boundary.
7
+ */
8
+ import { describe, expect, test } from "bun:test";
9
+
10
+ import { createMockProvider } from "../__tests__/helpers/mock-provider.js";
11
+ import type { ContentBlock, ProviderResponse } from "../providers/types.js";
12
+ import { AgentLoop } from "./loop.js";
13
+
14
+ const endTurn = (text: string): ProviderResponse => ({
15
+ content: [{ type: "text", text }],
16
+ model: "mock-model",
17
+ usage: { inputTokens: 1, outputTokens: 1 },
18
+ stopReason: "end_turn",
19
+ });
20
+
21
+ const toolUseTurn = (
22
+ blocks: Array<{ id: string; name: string }>,
23
+ ): ProviderResponse => ({
24
+ content: [
25
+ { type: "text", text: "working" },
26
+ ...blocks.map((b) => ({
27
+ type: "tool_use" as const,
28
+ id: b.id,
29
+ name: b.name,
30
+ input: {},
31
+ })),
32
+ ],
33
+ model: "mock-model",
34
+ usage: { inputTokens: 1, outputTokens: 1 },
35
+ stopReason: "tool_use",
36
+ });
37
+
38
+ function blocksOfType<T extends ContentBlock["type"]>(
39
+ history: Array<{ content: ContentBlock[] }>,
40
+ type: T,
41
+ ): Array<Extract<ContentBlock, { type: T }>> {
42
+ return history
43
+ .flatMap((m) => m.content)
44
+ .filter((b): b is Extract<ContentBlock, { type: T }> => b.type === type);
45
+ }
46
+
47
+ function buildLoop(
48
+ provider: ReturnType<typeof createMockProvider>["provider"],
49
+ conversationId: string,
50
+ executed: string[],
51
+ ) {
52
+ return new AgentLoop({
53
+ provider,
54
+ systemPrompt: "sys",
55
+ conversationId,
56
+ tools: [
57
+ { name: "read_file", description: "", input_schema: { type: "object" } },
58
+ ],
59
+ toolExecutor: async (name) => {
60
+ executed.push(name);
61
+ return { content: `ran ${name}`, isError: false };
62
+ },
63
+ });
64
+ }
65
+
66
+ const baseRun = {
67
+ requestId: "req-dedup",
68
+ onEvent: () => {},
69
+ callSite: "mainAgent" as const,
70
+ trust: { sourceChannel: "vellum" as const, trustClass: "unknown" as const },
71
+ messages: [
72
+ { role: "user" as const, content: [{ type: "text" as const, text: "go" }] },
73
+ ],
74
+ };
75
+
76
+ describe("AgentLoop: duplicate tool_use ids", () => {
77
+ /** A call the provider emits twice under one id runs a single time. */
78
+ test("executes a call id once when the provider emits it twice", async () => {
79
+ // GIVEN a provider turn carrying the same tool_use id twice
80
+ const { provider } = createMockProvider([
81
+ toolUseTurn([
82
+ { id: "call-dup", name: "read_file" },
83
+ { id: "call-dup", name: "read_file" },
84
+ ]),
85
+ endTurn("done"),
86
+ ]);
87
+
88
+ // WHEN the loop runs the turn
89
+ const executed: string[] = [];
90
+ const toolUseEventIds: string[] = [];
91
+ const { history } = await buildLoop(provider, "dedup-1", executed).run({
92
+ ...baseRun,
93
+ onEvent: (event) => {
94
+ if (event.type === "tool_use") {
95
+ toolUseEventIds.push(event.id);
96
+ }
97
+ },
98
+ });
99
+
100
+ // THEN the tool runs once and the client sees one tool_use
101
+ expect(executed).toEqual(["read_file"]);
102
+ expect(toolUseEventIds).toEqual(["call-dup"]);
103
+
104
+ // AND history stays well-formed: providers require one tool_result per
105
+ // tool_use id, so the coalesced copy must not survive into history.
106
+ const toolUses = blocksOfType(history, "tool_use");
107
+ expect(toolUses.map((b) => b.id)).toEqual(["call-dup"]);
108
+ const results = blocksOfType(history, "tool_result");
109
+ expect(results.map((b) => b.tool_use_id)).toEqual(["call-dup"]);
110
+ expect(results[0]!.content).toBe("ran read_file");
111
+ expect(results[0]!.is_error).toBe(false);
112
+ });
113
+
114
+ /** Two independent calls of one tool are not collapsed by name. */
115
+ test("runs repeat calls of one tool when their ids differ", async () => {
116
+ // GIVEN a provider turn calling one tool twice under distinct ids
117
+ const { provider } = createMockProvider([
118
+ toolUseTurn([
119
+ { id: "call-a", name: "read_file" },
120
+ { id: "call-b", name: "read_file" },
121
+ ]),
122
+ endTurn("done"),
123
+ ]);
124
+
125
+ // WHEN the loop runs the turn
126
+ const executed: string[] = [];
127
+ const { history } = await buildLoop(provider, "dedup-2", executed).run(
128
+ baseRun,
129
+ );
130
+
131
+ // THEN both calls execute and each gets its own result
132
+ expect(executed).toEqual(["read_file", "read_file"]);
133
+ expect(
134
+ blocksOfType(history, "tool_result").map((b) => b.tool_use_id),
135
+ ).toEqual(["call-a", "call-b"]);
136
+ });
137
+
138
+ /** An id-less call still executes, under a generated id. */
139
+ test("assigns a call id when the provider emits a tool_use without one", async () => {
140
+ // GIVEN a provider turn whose tool_use block carries no id
141
+ const { provider } = createMockProvider([
142
+ toolUseTurn([{ id: "", name: "read_file" }]),
143
+ endTurn("done"),
144
+ ]);
145
+
146
+ // WHEN the loop runs the turn
147
+ const executed: string[] = [];
148
+ const { history } = await buildLoop(provider, "dedup-3", executed).run(
149
+ baseRun,
150
+ );
151
+
152
+ // THEN the call executes under a generated id its result correlates to
153
+ expect(executed).toEqual(["read_file"]);
154
+ const toolUses = blocksOfType(history, "tool_use");
155
+ expect(toolUses).toHaveLength(1);
156
+ expect(toolUses[0]!.id.length).toBeGreaterThan(0);
157
+ expect(
158
+ blocksOfType(history, "tool_result").map((b) => b.tool_use_id),
159
+ ).toEqual([toolUses[0]!.id]);
160
+ });
161
+ });
package/src/agent/loop.ts CHANGED
@@ -31,6 +31,7 @@ import { defaultCompact } from "../plugins/defaults/compaction/compact.js";
31
31
  import type { ContextWindowResult } from "../plugins/defaults/compaction/window-manager.js";
32
32
  import { runHook } from "../plugins/pipeline.js";
33
33
  import type { CompactionCircuitEvent } from "../plugins/types.js";
34
+ import { hasVisibleText } from "../providers/content-blocks.js";
34
35
  import { isMaxTokensStopReason } from "../providers/stop-reasons.js";
35
36
  import { normalizeThinkingConfigForWire } from "../providers/thinking-config.js";
36
37
  import type {
@@ -528,13 +529,6 @@ function assistantTextOf(content: ReadonlyArray<ContentBlock>): string {
528
529
  return text;
529
530
  }
530
531
 
531
- /** Whether `content` carries at least one non-empty `text` block. */
532
- function hasVisibleText(content: ReadonlyArray<ContentBlock>): boolean {
533
- return content.some(
534
- (block) => block.type === "text" && block.text.trim().length > 0,
535
- );
536
- }
537
-
538
532
  type AgentLoopContextWindowResolver = () => {
539
533
  maxInputTokens: number;
540
534
  overflowRecovery: { enabled: boolean; safetyMarginRatio: number };
@@ -704,10 +698,56 @@ export type LoopToolExecutor = (
704
698
  errorCode?: string;
705
699
  }>;
706
700
 
701
+ type ToolUseBlock = Extract<ContentBlock, { type: "tool_use" }>;
702
+
703
+ interface NormalizedToolUse {
704
+ /** Assistant content with at most one `tool_use` block per call id. */
705
+ content: ContentBlock[];
706
+ /** The `tool_use` blocks in `content`, in order. */
707
+ toolUseBlocks: ToolUseBlock[];
708
+ /** Coalesced copies: a call id and name for each block dropped. */
709
+ duplicates: Array<{ id: string; name: string }>;
710
+ }
711
+
712
+ /**
713
+ * Resolves an assistant reply's `tool_use` blocks into an executable set keyed
714
+ * by call id: a block with no id gets one, and a repeat of an id already in the
715
+ * reply is dropped so the call runs once and its single `tool_result` correlates
716
+ * unambiguously. Providers occasionally emit the same call twice under one id,
717
+ * and both Anthropic and OpenAI require one `tool_result` per `tool_use` id, so
718
+ * the duplicate has no well-formed representation downstream.
719
+ */
720
+ function normalizeToolUseBlocks(
721
+ content: ReadonlyArray<ContentBlock>,
722
+ ): NormalizedToolUse {
723
+ const nextContent: ContentBlock[] = [];
724
+ const toolUseBlocks: ToolUseBlock[] = [];
725
+ const duplicates: Array<{ id: string; name: string }> = [];
726
+ const seenIds = new Set<string>();
727
+
728
+ for (const block of content) {
729
+ if (block.type !== "tool_use") {
730
+ nextContent.push(block);
731
+ continue;
732
+ }
733
+ if (seenIds.has(block.id)) {
734
+ duplicates.push({ id: block.id, name: block.name });
735
+ continue;
736
+ }
737
+ const normalized: ToolUseBlock =
738
+ block.id.length === 0 ? { ...block, id: crypto.randomUUID() } : block;
739
+ seenIds.add(normalized.id);
740
+ nextContent.push(normalized);
741
+ toolUseBlocks.push(normalized);
742
+ }
743
+
744
+ return { content: nextContent, toolUseBlocks, duplicates };
745
+ }
746
+
707
747
  /**
708
748
  * The benign result returned for a sibling tool call that was deferred because
709
749
  * an exclusive tool ran in the same turn. Phrased so the model treats it as a
710
- * "not run yet" signal — read the exclusive tool's output, then re-issue this
750
+ * "not run yet" signal: read the exclusive tool's output, then re-issue this
711
751
  * call if it is still the right next step.
712
752
  */
713
753
  function deferredForExclusiveMessage(exclusiveToolName: string): string {
@@ -1176,7 +1216,7 @@ export class AgentLoop {
1176
1216
  "Agent loop iteration start",
1177
1217
  );
1178
1218
 
1179
- let toolUseBlocks: Extract<ContentBlock, { type: "tool_use" }>[] = [];
1219
+ let toolUseBlocks: ToolUseBlock[] = [];
1180
1220
  // The provider rejection thrown by this iteration's call, if any. Set in
1181
1221
  // the inner provider catch and read by the outer catch to confine
1182
1222
  // error-stop recovery to genuine provider rejections — a throw from
@@ -1864,8 +1904,7 @@ export class AgentLoop {
1864
1904
  // the `post-model-call` hook below, which may add or drop tool calls;
1865
1905
  // this raw set drives only the completion log and the max-tokens branch.
1866
1906
  const modelToolUseBlocks = response.content.filter(
1867
- (block): block is Extract<ContentBlock, { type: "tool_use" }> =>
1868
- block.type === "tool_use",
1907
+ (block): block is ToolUseBlock => block.type === "tool_use",
1869
1908
  );
1870
1909
 
1871
1910
  rlog.info(
@@ -1979,19 +2018,26 @@ export class AgentLoop {
1979
2018
  // if the model had called it (the supported way for a plugin to surface
1980
2019
  // a card or take a follow-up action), or drop one the model emitted, so
1981
2020
  // the loop runs whatever the assistant message ends up carrying.
1982
- // Normalize ids so the executor and tool_result correlation stay 1:1 —
1983
- // a hook-added block may carry an empty or duplicate id.
1984
- toolUseBlocks = assistantMessage.content.filter(
1985
- (block): block is Extract<ContentBlock, { type: "tool_use" }> =>
1986
- block.type === "tool_use",
2021
+ // Normalizing ids keeps executor dispatch and tool_result correlation
2022
+ // 1:1 for the rest of the turn.
2023
+ const normalizedToolUse = normalizeToolUseBlocks(
2024
+ assistantMessage.content,
1987
2025
  );
1988
- const seenToolUseIds = new Set<string>();
1989
- for (const block of toolUseBlocks) {
1990
- if (block.id.length === 0 || seenToolUseIds.has(block.id)) {
1991
- block.id = crypto.randomUUID();
1992
- }
1993
- seenToolUseIds.add(block.id);
2026
+ for (const duplicate of normalizedToolUse.duplicates) {
2027
+ rlog.warn(
2028
+ {
2029
+ turn: toolUseTurns,
2030
+ duplicateId: duplicate.id,
2031
+ duplicateName: duplicate.name,
2032
+ },
2033
+ "Duplicate tool_use id in the assistant reply, coalescing into a single call",
2034
+ );
1994
2035
  }
2036
+ assistantMessage = {
2037
+ ...assistantMessage,
2038
+ content: normalizedToolUse.content,
2039
+ };
2040
+ toolUseBlocks = normalizedToolUse.toolUseBlocks;
1995
2041
 
1996
2042
  // At the no-tool stop boundary the retry decision is actionable: a
1997
2043
  // recovery hook may repair history and ask to re-query (a tool-bearing