@vellumai/assistant 0.8.10 → 0.8.11-staging.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (400) hide show
  1. package/bun.lock +62 -1
  2. package/docs/workspace-tools.md +196 -0
  3. package/examples/plugins/echo/README.md +3 -3
  4. package/knip.json +1 -0
  5. package/openapi.yaml +460 -128
  6. package/package.json +2 -1
  7. package/scripts/build-plugin-api.ts +299 -0
  8. package/src/__tests__/agent-loop-callsite-precedence.test.ts +7 -0
  9. package/src/__tests__/agent-loop-compaction-events.test.ts +197 -0
  10. package/src/__tests__/agent-loop-exit-reason.test.ts +93 -96
  11. package/src/__tests__/agent-loop-mutable-latest-user-message.test.ts +2 -0
  12. package/src/__tests__/agent-loop-output-hooks.test.ts +274 -1
  13. package/src/__tests__/agent-loop-override-profile.test.ts +3 -0
  14. package/src/__tests__/agent-loop-provider-error-recording.test.ts +4 -0
  15. package/src/__tests__/agent-loop-thinking.test.ts +4 -0
  16. package/src/__tests__/agent-loop.test.ts +578 -5
  17. package/src/__tests__/agent-wake-disk-pressure-callsite.test.ts +0 -1
  18. package/src/__tests__/approval-cascade.test.ts +1 -0
  19. package/src/__tests__/background-workers-disk-pressure.test.ts +0 -2
  20. package/src/__tests__/btw-routes.test.ts +0 -1
  21. package/src/__tests__/build-persisted-content.test.ts +75 -1
  22. package/src/__tests__/catalog-install-normalize.test.ts +141 -0
  23. package/src/__tests__/ces-startup-timeout.test.ts +60 -0
  24. package/src/__tests__/compaction-events.test.ts +1 -0
  25. package/src/__tests__/config-managed-gemini-defaults.test.ts +2 -46
  26. package/src/__tests__/context-overflow-reducer.test.ts +264 -124
  27. package/src/__tests__/context-window-manager-overflow-rung.test.ts +351 -0
  28. package/src/__tests__/conversation-abort-tool-results.test.ts +1 -1
  29. package/src/__tests__/conversation-agent-loop-inference-profile.test.ts +13 -5
  30. package/src/__tests__/conversation-agent-loop-overflow.test.ts +284 -455
  31. package/src/__tests__/conversation-agent-loop.test.ts +131 -551
  32. package/src/__tests__/conversation-app-control-instantiation.test.ts +13 -0
  33. package/src/__tests__/conversation-confirmation-signals.test.ts +1 -0
  34. package/src/__tests__/conversation-fork-crud.test.ts +259 -0
  35. package/src/__tests__/conversation-history-web-search.test.ts +1 -1
  36. package/src/__tests__/conversation-lifecycle.test.ts +257 -1
  37. package/src/__tests__/conversation-process-callsite.test.ts +1 -0
  38. package/src/__tests__/conversation-provider-retry-repair.test.ts +38 -377
  39. package/src/__tests__/conversation-queue.test.ts +1 -39
  40. package/src/__tests__/conversation-runtime-assembly.test.ts +119 -8
  41. package/src/__tests__/conversation-skill-tools.test.ts +491 -5
  42. package/src/__tests__/conversation-slash-queue.test.ts +1 -1
  43. package/src/__tests__/conversation-slash-unknown.test.ts +1 -0
  44. package/src/__tests__/conversation-speed-override.test.ts +1 -0
  45. package/src/__tests__/conversation-store.test.ts +74 -0
  46. package/src/__tests__/conversation-surfaces-app-control.test.ts +4 -1
  47. package/src/__tests__/conversation-tool-setup-attribution.test.ts +323 -0
  48. package/src/__tests__/conversation-tool-setup-tools-disabled.test.ts +34 -0
  49. package/src/__tests__/conversation-workspace-cache-state.test.ts +1 -0
  50. package/src/__tests__/conversation-workspace-injection.test.ts +1 -1
  51. package/src/__tests__/conversation-workspace-tool-tracking.test.ts +1 -0
  52. package/src/__tests__/corrected-target.test.ts +93 -0
  53. package/src/__tests__/credential-execution-feature-gates.test.ts +3 -5
  54. package/src/__tests__/credential-execution-tools.test.ts +23 -11
  55. package/src/__tests__/credential-security-invariants.test.ts +6 -1
  56. package/src/__tests__/db-schedule-syntax-migration.test.ts +80 -0
  57. package/src/__tests__/device-id.test.ts +70 -1
  58. package/src/__tests__/embedding-managed-proxy-selection.test.ts +6 -40
  59. package/src/__tests__/empty-response-hook.test.ts +242 -66
  60. package/src/__tests__/external-plugin-loader.test.ts +0 -31
  61. package/src/__tests__/get-skill-detail-audit.test.ts +43 -1
  62. package/src/__tests__/guardian-routing-invariants.test.ts +91 -0
  63. package/src/__tests__/history-repair-hook.test.ts +228 -3
  64. package/src/__tests__/host-app-control-proxy.test.ts +45 -0
  65. package/src/__tests__/host-browser-proxy.test.ts +254 -9
  66. package/src/__tests__/identity-routes.test.ts +1 -0
  67. package/src/__tests__/image-recovery-hook.test.ts +387 -0
  68. package/src/__tests__/injector-chain.test.ts +5 -4
  69. package/src/__tests__/injector-v3-suppression.test.ts +373 -47
  70. package/src/__tests__/intent-routing.test.ts +7 -0
  71. package/src/__tests__/memory-retrieval-hook.test.ts +117 -15
  72. package/src/__tests__/notification-decision-strategy.test.ts +3 -3
  73. package/src/__tests__/oauth-store.test.ts +0 -85
  74. package/src/__tests__/{context-overflow-policy.test.ts → overflow-policy.test.ts} +1 -1
  75. package/src/__tests__/parallel-tool.benchmark.test.ts +4 -0
  76. package/src/__tests__/persist-unsendable-image-downscale.test.ts +29 -9
  77. package/src/__tests__/persist-unsendable-image.test.ts +4 -4
  78. package/src/__tests__/persistence-secret-redaction.test.ts +78 -0
  79. package/src/__tests__/plugin-bootstrap.test.ts +82 -73
  80. package/src/__tests__/plugin-tool-contribution.test.ts +7 -4
  81. package/src/__tests__/plugin-types.test.ts +0 -8
  82. package/src/__tests__/provider-catalog-visibility.test.ts +1 -9
  83. package/src/__tests__/prune-old-conversations-job.test.ts +99 -0
  84. package/src/__tests__/registry.test.ts +240 -1
  85. package/src/__tests__/require-fresh-approval.test.ts +3 -0
  86. package/src/__tests__/schedule-routes.test.ts +116 -1
  87. package/src/__tests__/schedule-store.test.ts +28 -0
  88. package/src/__tests__/schedule-tools.test.ts +94 -1
  89. package/src/__tests__/server-history-render.test.ts +39 -0
  90. package/src/__tests__/skill-projection-feature-flag.test.ts +13 -0
  91. package/src/__tests__/skill-projection.benchmark.test.ts +25 -7
  92. package/src/__tests__/skills.test.ts +202 -0
  93. package/src/__tests__/slim-skill-category.test.ts +195 -0
  94. package/src/__tests__/strip-memory-injections.test.ts +33 -39
  95. package/src/__tests__/test-support/tool-invocation-seed.ts +79 -0
  96. package/src/__tests__/title-generate-hook.test.ts +9 -7
  97. package/src/__tests__/tool-audit-listener.test.ts +264 -1
  98. package/src/__tests__/tool-error-hook.test.ts +4 -3
  99. package/src/__tests__/tool-execution-pipeline.benchmark.test.ts +1 -0
  100. package/src/__tests__/tool-executor-lifecycle-events.test.ts +273 -0
  101. package/src/__tests__/tool-result-truncate-hook.test.ts +1 -0
  102. package/src/__tests__/tool-start-timestamp.test.ts +218 -0
  103. package/src/__tests__/tools-get-route.test.ts +202 -0
  104. package/src/__tests__/workspace-tool-loader.test.ts +319 -0
  105. package/src/__tests__/workspace-tools-watcher-flag.test.ts +70 -0
  106. package/src/agent/loop.ts +569 -319
  107. package/src/api/events/tool-result.ts +9 -0
  108. package/src/api/events/tool-use-start.ts +7 -0
  109. package/src/api/index.ts +10 -0
  110. package/src/api/responses/conversation-message.ts +135 -27
  111. package/src/api/responses/memory-v3-selection-log.ts +4 -4
  112. package/src/approvals/guardian-request-resolvers.ts +26 -0
  113. package/src/browser-session/backends/host-bridge.ts +29 -0
  114. package/src/browser-session/index.ts +1 -0
  115. package/src/browser-session/types.ts +5 -1
  116. package/src/cli/commands/__tests__/schedules.test.ts +62 -4
  117. package/src/cli/commands/__tests__/skills.test.ts +53 -0
  118. package/src/cli/commands/channel-verification-sessions.ts +6 -6
  119. package/src/cli/commands/inference-providers.ts +0 -8
  120. package/src/cli/commands/plugins.ts +2 -2
  121. package/src/cli/commands/schedules.ts +27 -4
  122. package/src/cli/commands/skills.ts +187 -146
  123. package/src/cli/commands/tools.ts +106 -0
  124. package/src/cli/lib/__tests__/install-from-github.test.ts +256 -328
  125. package/src/cli/lib/__tests__/plugin-catalog-cache.test.ts +6 -2
  126. package/src/cli/lib/__tests__/plugin-details.test.ts +10 -16
  127. package/src/cli/lib/__tests__/plugin-marketplace.test.ts +2 -2
  128. package/src/cli/lib/__tests__/search-plugins.test.ts +145 -240
  129. package/src/cli/lib/install-from-github.ts +187 -117
  130. package/src/cli/lib/plugin-catalog-cache.ts +9 -9
  131. package/src/cli/lib/plugin-details.ts +38 -68
  132. package/src/cli/lib/plugin-marketplace.ts +42 -14
  133. package/src/cli/lib/search-plugins.ts +29 -129
  134. package/src/cli/program.ts +2 -0
  135. package/src/config/bundled-skills/acp/SKILL.md +1 -0
  136. package/src/config/bundled-skills/app-builder/SKILL.md +1 -0
  137. package/src/config/bundled-skills/app-control/SKILL.md +1 -0
  138. package/src/config/bundled-skills/computer-use/SKILL.md +1 -0
  139. package/src/config/bundled-skills/contacts/SKILL.md +1 -0
  140. package/src/config/bundled-skills/document-editor/SKILL.md +1 -0
  141. package/src/config/bundled-skills/followups/SKILL.md +1 -0
  142. package/src/config/bundled-skills/image-studio/SKILL.md +1 -0
  143. package/src/config/bundled-skills/media-processing/SKILL.md +1 -0
  144. package/src/config/bundled-skills/messaging/SKILL.md +1 -0
  145. package/src/config/bundled-skills/phone-calls/SKILL.md +1 -0
  146. package/src/config/bundled-skills/playbooks/SKILL.md +1 -0
  147. package/src/config/bundled-skills/schedule/SKILL.md +1 -0
  148. package/src/config/bundled-skills/schedule/TOOLS.json +11 -3
  149. package/src/config/bundled-skills/sequences/SKILL.md +1 -0
  150. package/src/config/bundled-skills/settings/SKILL.md +1 -0
  151. package/src/config/bundled-skills/skill-management/SKILL.md +101 -1
  152. package/src/config/bundled-skills/subagent/SKILL.md +1 -0
  153. package/src/config/bundled-skills/transcribe/SKILL.md +1 -0
  154. package/src/config/env-registry.ts +23 -0
  155. package/src/config/feature-flag-registry.json +17 -81
  156. package/src/config/loader.ts +5 -22
  157. package/src/config/schema.ts +2 -0
  158. package/src/config/schemas/__tests__/compaction-logs.test.ts +56 -0
  159. package/src/config/schemas/__tests__/memory-v2.test.ts +0 -1
  160. package/src/config/schemas/__tests__/memory-v3.test.ts +61 -1
  161. package/src/config/schemas/compaction-logs.ts +79 -0
  162. package/src/config/schemas/memory-v2.ts +0 -8
  163. package/src/config/schemas/memory-v3.ts +104 -33
  164. package/src/config/seed-inference-profiles.ts +1 -1
  165. package/src/config/skills.ts +117 -47
  166. package/src/context/compactor.ts +11 -0
  167. package/src/context/strip-injections.ts +38 -4
  168. package/src/credential-execution/feature-gates.ts +0 -21
  169. package/src/credential-execution/startup-timeout.ts +32 -4
  170. package/src/daemon/__tests__/conversation-tool-setup-exclude.test.ts +18 -0
  171. package/src/daemon/conversation-agent-loop-handlers.ts +140 -91
  172. package/src/daemon/conversation-agent-loop.ts +123 -658
  173. package/src/daemon/conversation-error.ts +6 -33
  174. package/src/daemon/conversation-lifecycle.ts +1 -1
  175. package/src/daemon/conversation-runtime-assembly.ts +183 -22
  176. package/src/daemon/conversation-skill-tools.ts +137 -8
  177. package/src/daemon/conversation-slash.ts +0 -14
  178. package/src/daemon/conversation-store.ts +2 -19
  179. package/src/daemon/conversation-tool-setup.ts +87 -1
  180. package/src/daemon/conversation.ts +119 -50
  181. package/src/daemon/external-plugins-bootstrap.ts +36 -95
  182. package/src/daemon/handlers/config-channels.ts +11 -2
  183. package/src/daemon/handlers/shared.ts +9 -1
  184. package/src/daemon/handlers/skills.ts +10 -4
  185. package/src/daemon/host-app-control-proxy.ts +72 -57
  186. package/src/daemon/host-browser-proxy.ts +117 -22
  187. package/src/daemon/lifecycle.ts +34 -5
  188. package/src/daemon/message-protocol.ts +0 -7
  189. package/src/daemon/message-types/schedules.ts +1 -0
  190. package/src/daemon/message-types/skills.ts +17 -0
  191. package/src/daemon/providers-setup.ts +3 -0
  192. package/src/daemon/server.ts +3 -3
  193. package/src/daemon/tool-setup-types.ts +9 -3
  194. package/src/daemon/trust-context.ts +23 -0
  195. package/src/daemon/workspace-tools-watcher.ts +324 -0
  196. package/src/events/tool-audit-listener.ts +78 -16
  197. package/src/events/tool-metrics-listener.ts +2 -5
  198. package/src/memory/__tests__/compaction-log-writer-clickhouse.test.ts +227 -0
  199. package/src/memory/__tests__/conversation-queries.test.ts +176 -0
  200. package/src/memory/__tests__/jobs-worker-v2-schedule.test.ts +20 -32
  201. package/src/memory/compaction-log-writer-clickhouse.ts +418 -0
  202. package/src/memory/conversation-crud.ts +134 -3
  203. package/src/memory/conversation-queries.ts +64 -7
  204. package/src/memory/db-init.ts +14 -0
  205. package/src/memory/embedding-backend.test.ts +130 -1
  206. package/src/memory/embedding-backend.ts +79 -106
  207. package/src/memory/embedding-gemini.ts +5 -0
  208. package/src/memory/graph/__tests__/conversation-graph-memory-v2-routing.test.ts +12 -0
  209. package/src/memory/graph/__tests__/handle-remember-v2.test.ts +19 -0
  210. package/src/memory/graph/conversation-graph-memory.ts +36 -25
  211. package/src/memory/graph/tool-handlers.ts +3 -0
  212. package/src/memory/job-handlers/cleanup.ts +3 -1
  213. package/src/memory/jobs-store.ts +0 -28
  214. package/src/memory/jobs-worker.ts +10 -22
  215. package/src/memory/memory-marker.ts +29 -0
  216. package/src/memory/memory-retrospective-startup-cleanup.ts +1 -1
  217. package/src/memory/migrations/268-add-memory-v3-selections.ts +6 -0
  218. package/src/memory/migrations/270-schedule-description.ts +36 -0
  219. package/src/memory/migrations/275-tool-invocations-add-skill-id.test.ts +81 -0
  220. package/src/memory/migrations/275-tool-invocations-add-skill-id.ts +20 -0
  221. package/src/memory/migrations/276-tool-invocations-created-at-id-index.test.ts +68 -0
  222. package/src/memory/migrations/276-tool-invocations-created-at-id-index.ts +20 -0
  223. package/src/memory/migrations/277-add-memory-v3-ever-injected.ts +29 -0
  224. package/src/memory/migrations/278-tool-invocations-telemetry-columns.test.ts +96 -0
  225. package/src/memory/migrations/278-tool-invocations-telemetry-columns.ts +39 -0
  226. package/src/memory/migrations/279-create-skill-loaded-events.test.ts +84 -0
  227. package/src/memory/migrations/279-create-skill-loaded-events.ts +26 -0
  228. package/src/memory/migrations/280-conversations-surfaced-at.test.ts +88 -0
  229. package/src/memory/migrations/280-conversations-surfaced-at.ts +24 -0
  230. package/src/memory/migrations/index.ts +10 -0
  231. package/src/memory/migrations/registry.ts +8 -0
  232. package/src/memory/schema/conversations.ts +16 -0
  233. package/src/memory/schema/infrastructure.ts +26 -0
  234. package/src/memory/skill-loaded-events-store.test.ts +160 -0
  235. package/src/memory/skill-loaded-events-store.ts +95 -0
  236. package/src/memory/tool-executed-events-store.test.ts +219 -0
  237. package/src/memory/tool-executed-events-store.ts +102 -0
  238. package/src/memory/tool-usage-store.ts +15 -3
  239. package/src/memory/v2/__tests__/consolidation-job.test.ts +117 -12
  240. package/src/memory/v2/__tests__/consolidation-prompt-flag-gating-guard.test.ts +189 -0
  241. package/src/memory/v2/__tests__/injected-block-slugs.test.ts +90 -0
  242. package/src/memory/v2/__tests__/page-store.test.ts +33 -0
  243. package/src/memory/v2/__tests__/prompts-consolidation.test.ts +88 -15
  244. package/src/memory/v2/activation-store.ts +50 -1
  245. package/src/memory/v2/consolidation-job.ts +113 -29
  246. package/src/memory/v2/injected-block-slugs.ts +79 -0
  247. package/src/memory/v2/injection.ts +6 -1
  248. package/src/memory/v2/prompts/consolidation.ts +414 -13
  249. package/src/memory/v2/router.ts +2 -28
  250. package/src/memory/v2/static-context.ts +1 -1
  251. package/src/memory/v2/types.ts +16 -0
  252. package/src/notifications/__tests__/copy-composer.test.ts +244 -0
  253. package/src/notifications/access-request-copy.ts +298 -0
  254. package/src/notifications/adapters/slack.ts +3 -3
  255. package/src/notifications/adapters/telegram.ts +2 -1
  256. package/src/notifications/copy-composer.ts +49 -267
  257. package/src/notifications/decision-engine.ts +16 -35
  258. package/src/notifications/home-feed-side-effect.ts +1 -6
  259. package/src/oauth/oauth-store.ts +0 -9
  260. package/src/permissions/checker.test.ts +83 -1
  261. package/src/permissions/checker.ts +25 -2
  262. package/src/permissions/gateway-threshold-reader.test.ts +182 -0
  263. package/src/permissions/gateway-threshold-reader.ts +80 -0
  264. package/src/platform/client.ts +1 -3
  265. package/src/platform/feature-gate.ts +3 -12
  266. package/src/plugin-api/constants.ts +4 -2
  267. package/src/plugin-api/index.ts +63 -11
  268. package/src/plugin-api/types.ts +236 -71
  269. package/src/plugins/defaults/compaction/compact.ts +66 -2
  270. package/src/plugins/defaults/compaction/context-overflow-reducer.ts +240 -32
  271. package/src/plugins/defaults/compaction/corrected-target.ts +53 -0
  272. package/src/{daemon/context-overflow-policy.ts → plugins/defaults/compaction/overflow-policy.ts} +1 -1
  273. package/src/plugins/defaults/compaction/window-manager.ts +303 -1
  274. package/src/plugins/defaults/empty-response/hooks/post-model-call.ts +173 -0
  275. package/src/plugins/defaults/empty-response/hooks/stop.ts +11 -115
  276. package/src/plugins/defaults/empty-response/nudge-state-store.ts +46 -0
  277. package/src/plugins/defaults/history-repair/hooks/post-model-call.ts +50 -0
  278. package/src/plugins/defaults/history-repair/hooks/stop.ts +22 -0
  279. package/src/plugins/defaults/history-repair/repair-state-store.ts +51 -0
  280. package/src/plugins/defaults/history-repair/terminal.ts +39 -2
  281. package/src/plugins/defaults/image-recovery/detect.ts +25 -0
  282. package/src/plugins/defaults/image-recovery/hooks/post-model-call.ts +73 -0
  283. package/src/plugins/defaults/image-recovery/hooks/stop.ts +22 -0
  284. package/src/plugins/defaults/image-recovery/image-recovery-state-store.ts +48 -0
  285. package/src/plugins/defaults/image-recovery/package.json +14 -0
  286. package/src/{daemon/persist-unsendable-image.ts → plugins/defaults/image-recovery/recover.ts} +67 -14
  287. package/src/plugins/defaults/index.ts +71 -5
  288. package/src/plugins/defaults/memory-retrieval/hooks/post-compact.ts +76 -112
  289. package/src/plugins/defaults/memory-retrieval/hooks/{user-prompt-submit-temp.ts → user-prompt-submit.ts} +83 -74
  290. package/src/plugins/defaults/memory-retrieval/injector-chain.ts +14 -8
  291. package/src/plugins/defaults/memory-retrieval/injectors.ts +2 -18
  292. package/src/plugins/defaults/memory-retrieval/package.json +14 -0
  293. package/src/plugins/defaults/memory-v3-shadow/__tests__/carry-integration.test.ts +1157 -0
  294. package/src/plugins/defaults/memory-v3-shadow/__tests__/injection.test.ts +683 -0
  295. package/src/plugins/defaults/memory-v3-shadow/__tests__/live-integration.test.ts +161 -140
  296. package/src/plugins/defaults/memory-v3-shadow/__tests__/maintain-job.test.ts +160 -0
  297. package/src/plugins/defaults/memory-v3-shadow/__tests__/orchestrate.test.ts +335 -316
  298. package/src/plugins/defaults/memory-v3-shadow/__tests__/pool-select.test.ts +145 -53
  299. package/src/plugins/defaults/memory-v3-shadow/__tests__/render-injection.test.ts +38 -1
  300. package/src/plugins/defaults/memory-v3-shadow/__tests__/selection-log-store.test.ts +18 -8
  301. package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-integration.test.ts +112 -71
  302. package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-plugin.test.ts +192 -51
  303. package/src/plugins/defaults/memory-v3-shadow/__tests__/types.test.ts +4 -16
  304. package/src/plugins/defaults/memory-v3-shadow/card.test.ts +173 -0
  305. package/src/plugins/defaults/memory-v3-shadow/card.ts +116 -0
  306. package/src/plugins/defaults/memory-v3-shadow/core-set.test.ts +104 -0
  307. package/src/plugins/defaults/memory-v3-shadow/core-set.ts +59 -0
  308. package/src/plugins/defaults/memory-v3-shadow/ever-injected-store.test.ts +305 -0
  309. package/src/plugins/defaults/memory-v3-shadow/ever-injected-store.ts +278 -0
  310. package/src/plugins/defaults/memory-v3-shadow/hot-set.test.ts +138 -0
  311. package/src/plugins/defaults/memory-v3-shadow/hot-set.ts +85 -0
  312. package/src/plugins/defaults/memory-v3-shadow/injector.ts +331 -24
  313. package/src/plugins/defaults/memory-v3-shadow/maintain-job.ts +119 -13
  314. package/src/plugins/defaults/memory-v3-shadow/orchestrate.ts +169 -114
  315. package/src/plugins/defaults/memory-v3-shadow/page-content.ts +47 -16
  316. package/src/plugins/defaults/memory-v3-shadow/pool-select.ts +144 -66
  317. package/src/plugins/defaults/memory-v3-shadow/prune.test.ts +758 -0
  318. package/src/plugins/defaults/memory-v3-shadow/prune.ts +471 -0
  319. package/src/plugins/defaults/memory-v3-shadow/render-injection.ts +68 -16
  320. package/src/plugins/defaults/memory-v3-shadow/selection-log-store.ts +19 -12
  321. package/src/plugins/defaults/memory-v3-shadow/shadow-plugin.ts +96 -43
  322. package/src/plugins/defaults/memory-v3-shadow/types.ts +34 -17
  323. package/src/plugins/defaults/title-generate/hooks/stop.ts +9 -11
  324. package/src/plugins/pipeline.ts +8 -5
  325. package/src/plugins/types.ts +5 -51
  326. package/src/providers/cache-control.ts +26 -0
  327. package/src/providers/inference/__tests__/base-url-route-validation.test.ts +1 -2
  328. package/src/providers/model-catalog.ts +13 -1
  329. package/src/providers/openai/__tests__/tool-choice-mapping.test.ts +147 -0
  330. package/src/providers/openai/chat-completions-provider.ts +46 -0
  331. package/src/providers/openai/responses-provider.ts +45 -0
  332. package/src/providers/registry.ts +0 -8
  333. package/src/runtime/__tests__/agent-wake.test.ts +0 -1
  334. package/src/runtime/agent-wake.ts +13 -0
  335. package/src/runtime/routes/__tests__/acp-routes.test.ts +151 -0
  336. package/src/runtime/routes/__tests__/consolidation-routes.test.ts +12 -50
  337. package/src/runtime/routes/__tests__/conversation-surface-routes.test.ts +322 -0
  338. package/src/runtime/routes/__tests__/inference-provider-connection-routes.test.ts +0 -62
  339. package/src/runtime/routes/__tests__/plugins-routes.test.ts +31 -11
  340. package/src/runtime/routes/acp-routes.test.ts +106 -0
  341. package/src/runtime/routes/acp-routes.ts +248 -2
  342. package/src/runtime/routes/browser-tabs-routes.ts +1 -1
  343. package/src/runtime/routes/channel-verification-routes.ts +14 -5
  344. package/src/runtime/routes/chatgpt-subscription-auth-routes.ts +0 -14
  345. package/src/runtime/routes/consolidation-routes.ts +6 -82
  346. package/src/runtime/routes/conversation-list-routes.ts +6 -0
  347. package/src/runtime/routes/conversation-management-routes.ts +71 -0
  348. package/src/runtime/routes/identity-routes.ts +8 -0
  349. package/src/runtime/routes/inbound-stages/acl-enforcement.ts +33 -34
  350. package/src/runtime/routes/inference-provider-connection-routes.ts +0 -45
  351. package/src/runtime/routes/plugins-routes.ts +34 -45
  352. package/src/runtime/routes/schedule-routes.ts +43 -5
  353. package/src/runtime/routes/settings-routes.ts +140 -15
  354. package/src/runtime/routes/skills-routes.ts +18 -6
  355. package/src/runtime/services/__tests__/conversation-serializer.test.ts +140 -0
  356. package/src/runtime/services/conversation-serializer.ts +38 -1
  357. package/src/runtime/verification-outbound-actions.ts +147 -2
  358. package/src/runtime/verification-templates.ts +29 -3
  359. package/src/schedule/schedule-store.ts +19 -0
  360. package/src/skills/catalog-install.ts +77 -13
  361. package/src/tasks/task-scheduler.ts +1 -0
  362. package/src/telemetry/types.ts +66 -1
  363. package/src/telemetry/usage-telemetry-reporter.test.ts +542 -13
  364. package/src/telemetry/usage-telemetry-reporter.ts +213 -20
  365. package/src/tools/browser/__tests__/browser-execution-acquire.test.ts +49 -2
  366. package/src/tools/browser/__tests__/browser-status.test.ts +29 -5
  367. package/src/tools/browser/browser-execution.ts +27 -9
  368. package/src/tools/browser/cdp-client/__tests__/factory.test.ts +380 -4
  369. package/src/tools/browser/cdp-client/__tests__/host-bridge-cdp-client.test.ts +107 -0
  370. package/src/tools/browser/cdp-client/__tests__/types.test.ts +6 -1
  371. package/src/tools/browser/cdp-client/factory.ts +217 -17
  372. package/src/tools/browser/cdp-client/host-bridge-cdp-client.ts +67 -0
  373. package/src/tools/browser/cdp-client/types.ts +22 -2
  374. package/src/tools/credential-execution/make-authenticated-request.ts +2 -1
  375. package/src/tools/credential-execution/manage-secure-command-tool.ts +169 -164
  376. package/src/tools/credential-execution/run-authenticated-command.ts +2 -1
  377. package/src/tools/executor.ts +39 -7
  378. package/src/tools/registry.ts +387 -5
  379. package/src/tools/schedule/create.ts +16 -0
  380. package/src/tools/schedule/list.ts +12 -4
  381. package/src/tools/schedule/update.ts +12 -0
  382. package/src/tools/skills/load.ts +11 -6
  383. package/src/tools/terminal/safe-env.ts +2 -0
  384. package/src/tools/types.ts +65 -9
  385. package/src/tools/workspace-tools/loader.ts +673 -0
  386. package/src/usage/attribution.ts +28 -0
  387. package/src/util/device-id.ts +17 -3
  388. package/src/util/platform.ts +16 -0
  389. package/tsconfig.plugin-api.json +13 -0
  390. package/src/__tests__/plugin-external-api.test.ts +0 -68
  391. package/src/__tests__/plugin-skill-contribution.test.ts +0 -355
  392. package/src/daemon/message-types/browser.ts +0 -10
  393. package/src/notifications/__tests__/emit-signal-home-feed.test.ts +0 -187
  394. package/src/plugins/defaults/memory-v3-shadow/__tests__/fixtures/eval-turns.json +0 -36
  395. package/src/plugins/defaults/memory-v3-shadow/__tests__/fixtures/live-turns.json +0 -37
  396. package/src/plugins/defaults/memory-v3-shadow/__tests__/working-set-eviction.test.ts +0 -106
  397. package/src/plugins/defaults/memory-v3-shadow/__tests__/working-set-skeleton.test.ts +0 -44
  398. package/src/plugins/defaults/memory-v3-shadow/working-set.ts +0 -91
  399. package/src/plugins/external-api.ts +0 -114
  400. package/src/plugins/plugin-skill-contributions.ts +0 -292
@@ -19,6 +19,7 @@ import type { LLMConfig } from "../config/schemas/llm.js";
19
19
  import type { ServerMessage } from "../daemon/message-protocol.js";
20
20
  import { resetPluginRegistryAndRegisterDefaults } from "../plugins/defaults/index.js";
21
21
  import type { Message, Provider, ToolDefinition } from "../providers/types.js";
22
+ import { ContextOverflowError } from "../providers/types.js";
22
23
 
23
24
  const conversationCrudRealSnapshot = {
24
25
  ...(createRequire(import.meta.url)(
@@ -151,42 +152,106 @@ mock.module("../context/token-estimator.js", () => ({
151
152
  let mockReducerStepFn:
152
153
  | ((msgs: Message[], cfg: unknown, state: unknown) => unknown)
153
154
  | null = null;
155
+ const makeInitialReducerState = () => ({
156
+ appliedTiers: [] as string[],
157
+ injectionMode: "full" as const,
158
+ exhausted: false,
159
+ });
160
+ const runMockReducer = async (
161
+ msgs: Message[],
162
+ cfg: unknown,
163
+ state: unknown,
164
+ ) => {
165
+ if (mockReducerStepFn) return mockReducerStepFn(msgs, cfg, state);
166
+ return {
167
+ messages: msgs,
168
+ tier: "forced_compaction",
169
+ state: {
170
+ appliedTiers: [
171
+ "forced_compaction",
172
+ "tool_result_truncation",
173
+ "media_stubbing",
174
+ "injection_downgrade",
175
+ ],
176
+ injectionMode: "full",
177
+ exhausted: true,
178
+ },
179
+ estimatedTokens: 1000,
180
+ };
181
+ };
154
182
  mock.module(
155
183
  "../plugins/defaults/compaction/context-overflow-reducer.js",
156
184
  () => ({
157
- createInitialReducerState: () => ({
158
- appliedTiers: [],
159
- injectionMode: "full" as const,
160
- exhausted: false,
161
- }),
162
- reduceContextOverflow: async (
163
- msgs: Message[],
164
- cfg: unknown,
165
- state: unknown,
166
- ) => {
167
- if (mockReducerStepFn) return mockReducerStepFn(msgs, cfg, state);
168
- return {
169
- messages: msgs,
170
- tier: "forced_compaction",
185
+ createInitialReducerState: makeInitialReducerState,
186
+ reduceContextOverflow: runMockReducer,
187
+ }),
188
+ );
189
+
190
+ // Stand-in for `ContextWindowManager`'s turn-scoped overflow ladder. Threads
191
+ // reducer state across a turn's rungs and delegates each rung to the mocked
192
+ // reducer, mirroring `reduceOverflowOneRung` / `resetOverflowRecovery`.
193
+ // `recoverContextOverflow` adapts a rung into the `ContextWindowResult` the
194
+ // agent loop's compaction path consumes, mirroring the real manager's
195
+ // `overflowStepToResult` so the loop sees the rung's reduced history, injection
196
+ // mode, terminal auto-compress flag, and exhaustion.
197
+ function makeOverflowLadderStub(): {
198
+ resetOverflowRecovery: () => void;
199
+ reduceOverflowOneRung: (
200
+ msgs: Message[],
201
+ opts: unknown,
202
+ signal?: AbortSignal,
203
+ ) => Promise<unknown>;
204
+ recoverContextOverflow: (
205
+ msgs: Message[],
206
+ opts: unknown,
207
+ signal?: AbortSignal,
208
+ ) => Promise<unknown>;
209
+ } {
210
+ let state: unknown;
211
+ const reduceOverflowOneRung = async (msgs: Message[], opts: unknown) => {
212
+ if (!state) state = makeInitialReducerState();
213
+ const step = (await runMockReducer(msgs, opts, state)) as {
214
+ state: unknown;
215
+ };
216
+ state = step.state;
217
+ return step;
218
+ };
219
+ return {
220
+ resetOverflowRecovery: () => {
221
+ state = undefined;
222
+ },
223
+ reduceOverflowOneRung,
224
+ recoverContextOverflow: async (msgs: Message[], opts: unknown) => {
225
+ const step = (await reduceOverflowOneRung(msgs, opts)) as {
226
+ messages: Message[];
227
+ estimatedTokens?: number;
171
228
  state: {
172
- appliedTiers: [
173
- "forced_compaction",
174
- "tool_result_truncation",
175
- "media_stubbing",
176
- "injection_downgrade",
177
- ],
178
- injectionMode: "full",
179
- exhausted: true,
180
- },
181
- estimatedTokens: 1000,
229
+ appliedTiers: string[];
230
+ injectionMode: string;
231
+ exhausted: boolean;
232
+ };
233
+ compactionResult?: Record<string, unknown>;
234
+ };
235
+ const base = step.compactionResult ?? {
236
+ compacted: false,
237
+ messages: step.messages,
238
+ };
239
+ return {
240
+ ...base,
241
+ messages: step.messages,
242
+ injectionMode: step.state.injectionMode,
243
+ autoCompressApplied: step.state.appliedTiers.includes(
244
+ "auto_compress_latest_turn",
245
+ ),
246
+ exhausted: step.state.exhausted,
182
247
  };
183
248
  },
184
- }),
185
- );
249
+ };
250
+ }
186
251
 
187
252
  // Policy: default to fail_gracefully
188
253
  let mockOverflowAction: string = "fail_gracefully";
189
- mock.module("../daemon/context-overflow-policy.js", () => ({
254
+ mock.module("../plugins/defaults/compaction/overflow-policy.js", () => ({
190
255
  resolveOverflowAction: () => mockOverflowAction,
191
256
  }));
192
257
 
@@ -315,6 +380,7 @@ mock.module("../plugins/defaults/history-repair/terminal.js", () => ({
315
380
  },
316
381
  }),
317
382
  deepRepairHistory: (msgs: Message[]) => ({ messages: msgs, stats: {} }),
383
+ isRepairableOrderingError: () => false,
318
384
  }));
319
385
 
320
386
  const recordUsageMock = mock((..._args: unknown[]) => {});
@@ -405,11 +471,13 @@ mock.module("../daemon/conversation-error.js", () => ({
405
471
  /context.?length.?exceeded|prompt.?is.?too.?long|too many.*input.*tokens/i.test(
406
472
  msg,
407
473
  ),
408
- }));
409
-
410
- mock.module("../daemon/conversation-slash.js", () => ({
411
- isProviderOrderingError: (msg: string) =>
412
- /ordering|before.*after|messages.*order/i.test(msg),
474
+ budgetYieldUnrecoveredClassification: () => ({
475
+ code: "BUDGET_YIELD_UNRECOVERED",
476
+ userMessage:
477
+ "I tried to compact this conversation but couldn't fit the next step into the model's context window. Send another message to continue.",
478
+ retryable: true,
479
+ errorCategory: "budget_yield_unrecovered",
480
+ }),
413
481
  }));
414
482
 
415
483
  mock.module("../util/truncate.js", () => ({
@@ -427,6 +495,7 @@ mock.module("../memory/llm-request-log-store.js", () => ({
427
495
  recordRequestLog: () => {},
428
496
  backfillMessageIdOnLogs: () => {},
429
497
  setAgentLoopExitReasonOnLatestLog: setAgentLoopExitReasonOnLatestLogMock,
498
+ recordSyntheticAgentErrorMessageLog: () => {},
430
499
  }));
431
500
 
432
501
  mock.module("../memory/archive-store.js", () => ({
@@ -599,6 +668,16 @@ function makeCtx(
599
668
 
600
669
  ...ctxOverrides,
601
670
  } as unknown as Conversation;
671
+ // The convergence driver resolves the turn-scoped overflow ladder off the
672
+ // manager; give every fake manager the ladder methods unless a test supplied
673
+ // its own.
674
+ const manager = ctx.contextWindowManager as unknown as Record<
675
+ string,
676
+ unknown
677
+ >;
678
+ if (typeof manager.reduceOverflowOneRung !== "function") {
679
+ Object.assign(manager, makeOverflowLadderStub());
680
+ }
602
681
  fakeContextWindowManagers.set(conversationId, ctx.contextWindowManager);
603
682
  return ctx;
604
683
  }
@@ -807,9 +886,9 @@ describe("session-agent-loop overflow recovery (JARVIS-110)", () => {
807
886
  );
808
887
 
809
888
  // ── Test 2 ────────────────────────────────────────────────────────
810
- // When estimation says we're within budget but the provider rejects,
811
- // the post-run convergence loop should kick in and recover.
812
- // This test should PASS against current code (when no progress is made).
889
+ // When estimation says we're within budget but the provider rejects, the
890
+ // loop calibrates the estimator from the rejection and drives the reduction
891
+ // ladder on the next gate pass, recovering before the rerun.
813
892
  test("overflow recovery compacts below limit even when estimation underestimates", async () => {
814
893
  const events: ServerMessage[] = [];
815
894
  let reducerCalled = false;
@@ -819,7 +898,7 @@ describe("session-agent-loop overflow recovery (JARVIS-110)", () => {
819
898
  // up-front reduction.
820
899
  mockEstimateTokens = 185_000;
821
900
 
822
- // AND the post-run convergence reducer successfully compacts
901
+ // AND the reduction ladder successfully compacts on its first rung
823
902
  mockReducerStepFn = (msgs: Message[]) => {
824
903
  reducerCalled = true;
825
904
  return {
@@ -852,7 +931,10 @@ describe("session-agent-loop overflow recovery (JARVIS-110)", () => {
852
931
  // AND a provider that rejects the first call as too long (revealing the
853
932
  // real 242k count the estimator missed), then succeeds on the rerun.
854
933
  const { provider, calls } = createMockProvider([
855
- new Error("prompt is too long: 242201 tokens > 200000 maximum"),
934
+ new ContextOverflowError("prompt is too long", "mock-provider", {
935
+ actualTokens: 242_201,
936
+ maxTokens: 200_000,
937
+ }),
856
938
  textResponse("recovered"),
857
939
  ]);
858
940
 
@@ -1374,503 +1456,244 @@ describe("session-agent-loop overflow recovery (JARVIS-110)", () => {
1374
1456
  );
1375
1457
 
1376
1458
  // ── Test 8 ────────────────────────────────────────────────────────
1377
- // When mid-loop compaction exhausts maxAttempts but the agent loop
1378
- // still yields (yieldedForBudget remains true), the incomplete turn
1379
- // must escalate to the convergence loop instead of being silently
1380
- // treated as a completed turn.
1381
- test("exhausted mid-loop compaction attempts escalate to convergence loop", async () => {
1459
+ /**
1460
+ * Reactive recovery escalates the reduction ladder one rung per provider
1461
+ * rejection and ends the turn with `context_too_large` once the ladder is
1462
+ * exhausted and no auto-compress rung ran.
1463
+ */
1464
+ test("ladder escalation ends the turn with context_too_large when exhausted", async () => {
1465
+ // GIVEN an estimate below the mid-loop threshold, so only the provider's
1466
+ // rejection — not the proactive gate — drives recovery
1382
1467
  const events: ServerMessage[] = [];
1468
+ mockEstimateTokens = 100_000;
1383
1469
 
1384
- // Budget = 200_000 * 0.95 = 190_000
1385
- // Mid-loop threshold = 190_000 * 0.85 = 161_500
1386
- // Every estimate is above the threshold, so the first-call gate compacts
1387
- // before the first provider call and every checkpoint trips the yield.
1388
- mockEstimateTokens = 170_000;
1389
-
1390
- // The convergence reducer reduces tokens enough for the rerun to recover.
1391
- let convergenceReducerCalled = false;
1470
+ // AND a ladder that reduces on the first rung and reports exhaustion on
1471
+ // the second without ever applying the terminal auto-compress tier
1472
+ let reducerCallCount = 0;
1392
1473
  mockReducerStepFn = (msgs: Message[]) => {
1393
- convergenceReducerCalled = true;
1474
+ reducerCallCount++;
1475
+ const exhausted = reducerCallCount >= 2;
1394
1476
  return {
1395
1477
  messages: msgs,
1396
- tier: "forced_compaction",
1478
+ tier: exhausted ? "injection_downgrade" : "forced_compaction",
1397
1479
  state: {
1398
- appliedTiers: ["forced_compaction"],
1480
+ appliedTiers: exhausted
1481
+ ? [
1482
+ "forced_compaction",
1483
+ "tool_result_truncation",
1484
+ "media_stubbing",
1485
+ "injection_downgrade",
1486
+ ]
1487
+ : ["forced_compaction"],
1399
1488
  injectionMode: "full",
1400
- exhausted: true,
1489
+ exhausted,
1401
1490
  },
1402
- estimatedTokens: 80_000,
1491
+ estimatedTokens: exhausted ? 60_000 : 80_000,
1403
1492
  };
1404
1493
  };
1405
1494
 
1406
- // Every provider call returns a tool_use, so each loop run does a tool
1407
- // turn that trips the mid-loop budget gate. On the initial run the gate
1408
- // calls compaction (which surfaces `exhausted: true`); the convergence
1409
- // rerun runs without a compaction hook and yields "budget" directly.
1410
- // With the reducer exhausted, the convergence loop terminates with the
1411
- // turn still over budget and the orchestrator stamps `context_too_large`.
1495
+ // AND a provider that rejects every call with a context-overflow error
1412
1496
  const { provider, calls } = createMockProvider([
1413
- toolUseResponse("tu-1", "bash", { command: "ls" }),
1497
+ new ContextOverflowError(
1498
+ "context_length_exceeded: 250000 tokens > 200000 maximum",
1499
+ "mock-provider",
1500
+ { actualTokens: 250_000, maxTokens: 200_000 },
1501
+ ),
1414
1502
  ]);
1503
+ const ctx = makeCtx({ loopProvider: provider });
1415
1504
 
1416
- let compactionCallCount = 0;
1417
- const ctx = makeCtx({
1418
- loopProvider: provider,
1419
- loopTools: [
1420
- {
1421
- name: "bash",
1422
- description: "Run a shell command",
1423
- input_schema: {
1424
- type: "object",
1425
- properties: { command: { type: "string" } },
1426
- },
1427
- },
1428
- ],
1429
- toolExecutor: async () => ({ content: "output", isError: false }),
1430
- contextWindowManager: {
1431
- updateConfig: () => {},
1432
- shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
1433
- maybeCompact: async () => {
1434
- compactionCallCount++;
1435
- // Compaction's internal retry budget is exhausted — the
1436
- // compactor itself ran maxAttempts passes and still couldn't
1437
- // drop below the auto-threshold. `maybeCompact` surfaces this
1438
- // via `exhausted: true` so the loop yields "budget" and the
1439
- // orchestrator escalates straight to the convergence loop
1440
- // instead of looping on a stuck compactor.
1441
- return {
1442
- compacted: true,
1443
- exhausted: true,
1444
- messages: [
1445
- {
1446
- role: "user" as const,
1447
- content: [{ type: "text", text: "Hello" }],
1448
- },
1449
- ] as Message[],
1450
- compactedPersistedMessages: 5,
1451
- summaryText: "Compaction summary",
1452
- previousEstimatedInputTokens: 170_000,
1453
- estimatedInputTokens: 165_000, // barely reduced
1454
- maxInputTokens: 200_000,
1455
- thresholdTokens: 160_000,
1456
- compactedMessages: 10,
1457
- summaryCalls: 1,
1458
- summaryInputTokens: 500,
1459
- summaryOutputTokens: 200,
1460
- summaryModel: "mock-model",
1461
- };
1462
- },
1463
- } as unknown as Conversation["contextWindowManager"],
1464
- });
1465
-
1505
+ // WHEN the turn runs
1466
1506
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
1467
1507
 
1468
- // 1 initial auto-compact + 1 mid-loop compaction = 2 total. The
1469
- // first mid-loop call surfaces `exhausted: true`, so the
1470
- // orchestrator escalates immediately without retrying maybeCompact
1471
- // — the retry budget for the compactor itself lives inside
1472
- // `ContextWindowManager.maybeCompact`.
1473
- expect(compactionCallCount).toBe(2);
1474
-
1475
- // Provider calls: 1 initial tool turn (yields budget) + 1 convergence
1476
- // rerun that recovers. No mid-loop re-entries because the orchestrator
1477
- // broke out on `exhausted` before re-invoking the loop.
1478
- expect(calls.length).toBe(2);
1508
+ // THEN two ladder rungs ran (one per rejection before exhaustion) and the
1509
+ // exhausted-ladder rejection ended the turn — three provider calls total
1510
+ expect(reducerCallCount).toBe(2);
1511
+ expect(calls.length).toBe(3);
1479
1512
 
1480
- // After the compactor exhausted itself, the convergence loop
1481
- // should have been triggered (contextTooLargeDetected set to true)
1482
- expect(convergenceReducerCalled).toBe(true);
1513
+ // AND the loop emitted the terminal `context_too_large` exit, which the
1514
+ // daemon surfaced as a classified error and stamped onto the request log
1483
1515
  expect(setAgentLoopExitReasonOnLatestLogMock).toHaveBeenCalledWith(
1484
1516
  "test-conv",
1485
1517
  "context_too_large",
1486
1518
  );
1519
+ const errorEvent = events.find((e) => e.type === "conversation_error");
1520
+ expect(errorEvent).toBeDefined();
1521
+ if (errorEvent && "errorCategory" in errorEvent) {
1522
+ expect(errorEvent.errorCategory).toBe("context_too_large");
1523
+ }
1487
1524
  });
1488
1525
 
1489
1526
  // ── Test 8b ───────────────────────────────────────────────────────
1490
- // Counterpart to Test 8: when a mid-loop `maybeCompact` returns
1491
- // productive (`compacted: true`, no `exhausted` flag), the loop
1492
- // compacts in place and continues the run itself — it never yields
1493
- // "budget", so the orchestrator does not escalate to the convergence
1494
- // loop. Mid-loop iteration is now wholly internal to `AgentLoop.run`;
1495
- // the orchestrator only reacts to the binary `exhausted`/timeout
1496
- // signal carried back as a "budget" exit.
1497
- test("productive mid-loop compaction continues in place without escalating", async () => {
1527
+ /**
1528
+ * The common case: a single ladder rung reduces enough that the re-issued
1529
+ * provider call succeeds, so the loop continues the turn in place and ends
1530
+ * normally with no terminal overflow exit and no client-facing error.
1531
+ */
1532
+ test("single-rung recovery succeeds and the turn continues in place", async () => {
1533
+ // GIVEN an estimate below the mid-loop threshold
1498
1534
  const events: ServerMessage[] = [];
1535
+ mockEstimateTokens = 100_000;
1499
1536
 
1500
- // Budget = 200_000 * 0.95 = 190_000
1501
- // Mid-loop threshold = 190_000 * 0.85 = 161_500
1502
- // Every estimate is above the threshold: the first-call gate compacts
1503
- // before the first provider call, and each subsequent checkpoint trips
1504
- // the yield even after a successful compaction (each tool result inflates
1505
- // the context back past 85%).
1506
- mockEstimateTokens = 170_000;
1507
-
1508
- // A single tool round reaches one checkpoint; the in-loop budget gate
1509
- // trips there and compaction runs in place. The loop continues the run
1510
- // itself — the following provider call returns plain text and the turn
1511
- // completes — so the orchestrator never re-enters the convergence loop.
1537
+ // AND a ladder rung that reduces without reporting exhaustion
1538
+ let reducerCallCount = 0;
1539
+ mockReducerStepFn = (msgs: Message[]) => {
1540
+ reducerCallCount++;
1541
+ return {
1542
+ messages: msgs,
1543
+ tier: "forced_compaction",
1544
+ state: {
1545
+ appliedTiers: ["forced_compaction"],
1546
+ injectionMode: "full",
1547
+ exhausted: false,
1548
+ },
1549
+ estimatedTokens: 80_000,
1550
+ };
1551
+ };
1552
+
1553
+ // AND a provider that rejects once, then accepts the re-issued call
1512
1554
  const { provider, calls } = createMockProvider([
1513
- toolUseResponse("tu-1", "bash", { command: "ls" }),
1514
- textResponse("final answer"),
1555
+ new ContextOverflowError(
1556
+ "context_length_exceeded: 250000 tokens > 200000 maximum",
1557
+ "mock-provider",
1558
+ { actualTokens: 250_000, maxTokens: 200_000 },
1559
+ ),
1560
+ textResponse("recovered answer"),
1515
1561
  ]);
1562
+ const ctx = makeCtx({ loopProvider: provider });
1516
1563
 
1517
- // Compaction reports `estimatedInputTokens` well below the 161_500
1518
- // threshold — the "compaction is productive" signal (no `exhausted`
1519
- // flag) that lets the loop continue in place.
1520
- let compactionCallCount = 0;
1521
- const ctx = makeCtx({
1522
- loopProvider: provider,
1523
- loopTools: [
1524
- {
1525
- name: "bash",
1526
- description: "Run a shell command",
1527
- input_schema: {
1528
- type: "object",
1529
- properties: { command: { type: "string" } },
1530
- },
1531
- },
1532
- ],
1533
- toolExecutor: async () => ({ content: "output", isError: false }),
1534
- contextWindowManager: {
1535
- updateConfig: () => {},
1536
- shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
1537
- maybeCompact: async () => {
1538
- compactionCallCount++;
1539
- return {
1540
- compacted: true,
1541
- messages: [
1542
- {
1543
- role: "user" as const,
1544
- content: [{ type: "text", text: "Hello" }],
1545
- },
1546
- ] as Message[],
1547
- compactedPersistedMessages: 5,
1548
- summaryText: "Compaction summary",
1549
- previousEstimatedInputTokens: 170_000,
1550
- estimatedInputTokens: 100_000,
1551
- maxInputTokens: 200_000,
1552
- thresholdTokens: 160_000,
1553
- compactedMessages: 10,
1554
- summaryCalls: 1,
1555
- summaryInputTokens: 500,
1556
- summaryOutputTokens: 200,
1557
- summaryModel: "mock-model",
1558
- };
1559
- },
1560
- } as unknown as Conversation["contextWindowManager"],
1561
- });
1562
-
1564
+ // WHEN the turn runs
1563
1565
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
1564
1566
 
1565
- // 1 initial auto-compact + 1 productive mid-loop compaction.
1566
- expect(compactionCallCount).toBe(2);
1567
- // The loop continued in place after compacting: a tool turn followed by
1568
- // the post-compaction text turn, both within a single run.
1567
+ // THEN one rung ran and the retry succeeded — one rejection + one success
1568
+ expect(reducerCallCount).toBe(1);
1569
1569
  expect(calls.length).toBe(2);
1570
1570
 
1571
- // No escalation to the convergence loop because the mid-loop
1572
- // `maybeCompact` returned productive (no `exhausted` flag), and the turn
1573
- // completed normally.
1571
+ // AND the turn recovered in place: no terminal overflow exit, no error
1574
1572
  expect(setAgentLoopExitReasonOnLatestLogMock).not.toHaveBeenCalledWith(
1575
1573
  "test-conv",
1576
1574
  "context_too_large",
1577
1575
  );
1576
+ expect(setAgentLoopExitReasonOnLatestLogMock).not.toHaveBeenCalledWith(
1577
+ "test-conv",
1578
+ "budget_yield_unrecovered",
1579
+ );
1578
1580
  expect(events.find((e) => e.type === "conversation_error")).toBeUndefined();
1579
1581
  });
1580
1582
 
1581
1583
  // ── Test 9 ────────────────────────────────────────────────────────
1582
- // When the convergence loop reruns the agent loop and it still yields
1583
- // at checkpoint (yieldedForBudget), the loop must continue reducing
1584
- // through additional tiers instead of silently dropping the incomplete
1585
- // turn.
1586
- test("post-convergence yieldedForBudget continues reduction", async () => {
1584
+ /**
1585
+ * The ladder climbs through several rungs across successive rejections and
1586
+ * recovers once a later rung reduces the prompt enough for the provider to
1587
+ * accept it, completing the turn without a terminal overflow exit.
1588
+ */
1589
+ test("multi-rung escalation recovers when a later rung fits", async () => {
1590
+ // GIVEN an estimate below the mid-loop threshold
1587
1591
  const events: ServerMessage[] = [];
1592
+ mockEstimateTokens = 100_000;
1588
1593
 
1589
- // Budget = 200_000 * 0.95 = 190_000
1590
- // Mid-loop threshold = 190_000 * 0.85 = 161_500
1591
- let estimateCallCount = 0;
1592
- mockEstimateTokens = () => {
1593
- estimateCallCount++;
1594
- // Preflight: below budget
1595
- if (estimateCallCount === 1) return 100_000;
1596
- // Every checkpoint call: above threshold — always triggers yield
1597
- return 170_000;
1598
- };
1599
-
1600
- // Every provider call returns a tool_use, so each loop run does a tool
1601
- // turn that trips the mid-loop budget gate and yields "budget". The
1602
- // initial run's gate calls compaction (exhausted); the convergence
1603
- // reruns run without a compaction hook and yield directly.
1604
- const { provider, calls } = createMockProvider([
1605
- toolUseResponse("tu-1", "bash", { command: "ls" }),
1606
- ]);
1607
-
1608
- // Convergence reducer: first call returns non-exhausted, second returns exhausted
1594
+ // AND a ladder that reduces further on each successive rung
1609
1595
  let reducerCallCount = 0;
1610
1596
  mockReducerStepFn = (msgs: Message[]) => {
1611
1597
  reducerCallCount++;
1612
- if (reducerCallCount === 1) {
1613
- return {
1614
- messages: msgs,
1615
- tier: "forced_compaction",
1616
- state: {
1617
- appliedTiers: ["forced_compaction"],
1618
- injectionMode: "full",
1619
- exhausted: false,
1620
- },
1621
- estimatedTokens: 80_000,
1622
- };
1623
- }
1624
- // Second call: exhausted
1598
+ const tier =
1599
+ reducerCallCount === 1 ? "forced_compaction" : "tool_result_truncation";
1625
1600
  return {
1626
1601
  messages: msgs,
1627
- tier: "tool_result_truncation",
1602
+ tier,
1628
1603
  state: {
1629
- appliedTiers: ["forced_compaction", "tool_result_truncation"],
1604
+ appliedTiers:
1605
+ reducerCallCount === 1
1606
+ ? ["forced_compaction"]
1607
+ : ["forced_compaction", "tool_result_truncation"],
1630
1608
  injectionMode: "full",
1631
- exhausted: true,
1609
+ exhausted: false,
1632
1610
  },
1633
- estimatedTokens: 60_000,
1611
+ estimatedTokens: reducerCallCount === 1 ? 80_000 : 60_000,
1634
1612
  };
1635
1613
  };
1636
1614
 
1637
- const ctx = makeCtx({
1638
- loopProvider: provider,
1639
- loopTools: [
1640
- {
1641
- name: "bash",
1642
- description: "Run a shell command",
1643
- input_schema: {
1644
- type: "object",
1645
- properties: { command: { type: "string" } },
1646
- },
1647
- },
1648
- ],
1649
- toolExecutor: async () => ({ content: "output", isError: false }),
1650
- contextWindowManager: {
1651
- updateConfig: () => {},
1652
- shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
1653
- // Under the new architecture (Compaction Re-homing Arc, Bullet 1)
1654
- // the retry budget lives inside `ContextWindowManager._maybeCompact`,
1655
- // so a single daemon-level call represents the full manager retry
1656
- // sequence. Signal `exhausted: true` immediately to escalate the
1657
- // mid-loop to the convergence reducer.
1658
- maybeCompact: async () => ({
1659
- compacted: true,
1660
- messages: [
1661
- {
1662
- role: "user" as const,
1663
- content: [{ type: "text", text: "Hello" }],
1664
- },
1665
- ] as Message[],
1666
- compactedPersistedMessages: 5,
1667
- summaryText: "Compaction summary",
1668
- previousEstimatedInputTokens: 170_000,
1669
- estimatedInputTokens: 165_000,
1670
- maxInputTokens: 200_000,
1671
- thresholdTokens: 160_000,
1672
- compactedMessages: 10,
1673
- summaryCalls: 1,
1674
- summaryInputTokens: 500,
1675
- summaryOutputTokens: 200,
1676
- summaryModel: "mock-model",
1677
- exhausted: true,
1678
- }),
1679
- } as unknown as Conversation["contextWindowManager"],
1680
- });
1615
+ // AND a provider that rejects twice, then accepts the third call
1616
+ const { provider, calls } = createMockProvider([
1617
+ new ContextOverflowError(
1618
+ "context_length_exceeded: 250000 tokens > 200000 maximum",
1619
+ "mock-provider",
1620
+ { actualTokens: 250_000, maxTokens: 200_000 },
1621
+ ),
1622
+ new ContextOverflowError(
1623
+ "context_length_exceeded: 230000 tokens > 200000 maximum",
1624
+ "mock-provider",
1625
+ { actualTokens: 230_000, maxTokens: 200_000 },
1626
+ ),
1627
+ textResponse("recovered answer"),
1628
+ ]);
1629
+ const ctx = makeCtx({ loopProvider: provider });
1681
1630
 
1631
+ // WHEN the turn runs
1682
1632
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
1683
1633
 
1684
- // Reducer should have been called twice: once for first convergence tier,
1685
- // once more after yieldedForBudget triggered re-entry
1634
+ // THEN two rungs ran across two rejections and the third call recovered
1686
1635
  expect(reducerCallCount).toBe(2);
1687
-
1688
- // Provider calls: 1 initial run + 2 convergence reruns = 3 calls, each a
1689
- // tool turn that yields "budget". The mid-loop no longer drives
1690
- // daemon-level retries — the manager owns its retry budget and signals
1691
- // exhaustion via the `exhausted` flag.
1692
1636
  expect(calls.length).toBe(3);
1693
- expect(setAgentLoopExitReasonOnLatestLogMock).toHaveBeenCalledWith(
1637
+
1638
+ // AND no terminal overflow exit or error was surfaced
1639
+ expect(setAgentLoopExitReasonOnLatestLogMock).not.toHaveBeenCalledWith(
1694
1640
  "test-conv",
1695
1641
  "context_too_large",
1696
1642
  );
1643
+ expect(events.find((e) => e.type === "conversation_error")).toBeUndefined();
1697
1644
  });
1698
1645
 
1699
- // ── Test 9 ────────────────────────────────────────────────────────
1700
- // When the `auto_compress_latest_turn` rerun (the last layer of the
1701
- // overflow-recovery ladder) still yields at the mid-loop checkpoint,
1702
- // the turn cannot proceed. Before PR 1 of the Compaction Visibility
1703
- // workstream this terminated silently — no `agent_loop_exit_reason`,
1704
- // no client notice, no durable transcript row. Now the loop must:
1705
- // 1. emit a `conversation_error` event with code
1706
- // `BUDGET_YIELD_UNRECOVERED`,
1707
- // 2. persist a `role="assistant"` notice via the persistence
1708
- // pipeline (so reloads keep the message),
1709
- // 3. stamp `budget_yield_unrecovered` onto the latest llm_request_logs
1710
- // row.
1646
+ // ── Test 10 ───────────────────────────────────────────────────────
1647
+ /**
1648
+ * When the ladder applies its terminal `auto_compress_latest_turn` rung,
1649
+ * reports exhaustion, and the provider still rejects, the loop ends the turn
1650
+ * with `budget_yield_unrecovered`. The daemon then emits a classified
1651
+ * `BUDGET_YIELD_UNRECOVERED` error, persists a durable assistant notice, and
1652
+ * stamps the exit reason onto the latest llm_request_logs row.
1653
+ */
1711
1654
  test("budget_yield_unrecovered: classified error emitted, persisted, and stamped", async () => {
1655
+ // GIVEN an estimate below the mid-loop threshold
1712
1656
  const events: ServerMessage[] = [];
1657
+ mockEstimateTokens = 100_000;
1713
1658
 
1714
- // Every estimate after the very first preflight is above the mid-loop
1715
- // threshold (190_000 × 0.85 = 161_500). This makes every checkpoint
1716
- // yield, including the one inside the auto_compress rerun.
1717
- let estimateCallCount = 0;
1718
- mockEstimateTokens = () => {
1719
- estimateCallCount++;
1720
- if (estimateCallCount === 1) return 100_000;
1721
- return 170_000;
1722
- };
1723
-
1724
- // Convergence reducer becomes exhausted on the second tier so the
1725
- // loop escalates from convergence to the action-resolution block.
1659
+ // AND a ladder whose terminal rung applies `auto_compress_latest_turn` and
1660
+ // reports exhaustion — the signal that distinguishes
1661
+ // `budget_yield_unrecovered` from `context_too_large`
1726
1662
  let reducerCallCount = 0;
1727
1663
  mockReducerStepFn = (msgs: Message[]) => {
1728
1664
  reducerCallCount++;
1729
- const exhausted = reducerCallCount >= 2;
1665
+ const terminal = reducerCallCount >= 2;
1730
1666
  return {
1731
1667
  messages: msgs,
1732
- tier: exhausted ? "tool_result_truncation" : "forced_compaction",
1668
+ tier: terminal ? "auto_compress_latest_turn" : "forced_compaction",
1733
1669
  state: {
1734
- appliedTiers: exhausted
1735
- ? ["forced_compaction", "tool_result_truncation"]
1670
+ appliedTiers: terminal
1671
+ ? ["forced_compaction", "auto_compress_latest_turn"]
1736
1672
  : ["forced_compaction"],
1737
1673
  injectionMode: "full" as const,
1738
- exhausted,
1674
+ exhausted: terminal,
1739
1675
  },
1740
- estimatedTokens: exhausted ? 60_000 : 80_000,
1676
+ estimatedTokens: terminal ? 60_000 : 80_000,
1741
1677
  };
1742
1678
  };
1743
1679
 
1744
- // The overflow policy directs us into auto_compress_latest_turn so the
1745
- // emergency compaction + final agentLoop.run path executes.
1680
+ // AND an overflow policy that permits the terminal auto-compress rung
1746
1681
  mockOverflowAction = "auto_compress_latest_turn";
1747
1682
 
1748
- // Every provider call returns a tool_use, so each loop run does a tool
1749
- // turn that trips the mid-loop budget gate and yields "budget" —
1750
- // including the final auto_compress rerun.
1683
+ // AND a provider that rejects every call with a context-overflow error
1751
1684
  const { provider } = createMockProvider([
1752
- toolUseResponse("tu-1", "bash", { command: "ls" }),
1685
+ new ContextOverflowError(
1686
+ "context_length_exceeded: 250000 tokens > 200000 maximum",
1687
+ "mock-provider",
1688
+ { actualTokens: 250_000, maxTokens: 200_000 },
1689
+ ),
1753
1690
  ]);
1691
+ const ctx = makeCtx({ loopProvider: provider });
1754
1692
 
1755
- // Every forced `maybeCompact` is invoked through three distinct call
1756
- // sites, all owned by the loop's pre-call budget gate / recovery ladder:
1757
- // 1. First-call (turn-start) gate (`force: true`) — the loop now owns
1758
- // the turn-start compaction. It succeeds, but the mocked estimate
1759
- // stays above the mid-loop threshold, so the turn proceeds and the
1760
- // mid-loop gate still trips on the next iteration.
1761
- // 2. Mid-loop after the first tool turn (`force: true`) — must signal
1762
- // `exhausted: true` so the daemon escalates to the convergence
1763
- // reducer instead of looping forever.
1764
- // 3. auto_compress_latest_turn emergency compaction (`force: true`,
1765
- // `minKeepRecentUserTurns: 0`) — succeeds and drops tokens below
1766
- // threshold; the subsequent rerun yields again and is classified
1767
- // as BUDGET_YIELD_UNRECOVERED.
1768
- let forcedMaybeCompactCallCount = 0;
1769
- const ctx = makeCtx({
1770
- loopProvider: provider,
1771
- loopTools: [
1772
- {
1773
- name: "bash",
1774
- description: "Run a shell command",
1775
- input_schema: {
1776
- type: "object",
1777
- properties: { command: { type: "string" } },
1778
- },
1779
- },
1780
- ],
1781
- toolExecutor: async () => ({ content: "output", isError: false }),
1782
- contextWindowManager: {
1783
- updateConfig: () => {},
1784
- shouldCompact: () => ({ needed: false, estimatedTokens: 0 }),
1785
- maybeCompact: async (
1786
- _msgs: Message[],
1787
- _signal: AbortSignal,
1788
- opts?: { force?: boolean },
1789
- ) => {
1790
- // Only forced compactions drive this test; any non-forced probe is
1791
- // a no-op.
1792
- if (!opts?.force) {
1793
- return { compacted: false };
1794
- }
1795
- forcedMaybeCompactCallCount++;
1796
- if (forcedMaybeCompactCallCount === 1) {
1797
- // First-call (turn-start) gate: succeeds, but the mocked estimate
1798
- // stays above the threshold so the turn proceeds to the call and
1799
- // the mid-loop gate still trips next iteration.
1800
- return {
1801
- compacted: true,
1802
- messages: [
1803
- {
1804
- role: "user" as const,
1805
- content: [{ type: "text", text: "turn-start compacted" }],
1806
- },
1807
- ] as Message[],
1808
- compactedPersistedMessages: 5,
1809
- summaryText: "Turn-start summary",
1810
- previousEstimatedInputTokens: 170_000,
1811
- estimatedInputTokens: 165_000,
1812
- maxInputTokens: 200_000,
1813
- thresholdTokens: 160_000,
1814
- compactedMessages: 10,
1815
- summaryCalls: 1,
1816
- summaryInputTokens: 500,
1817
- summaryOutputTokens: 200,
1818
- summaryModel: "mock-model",
1819
- };
1820
- }
1821
- if (forcedMaybeCompactCallCount === 2) {
1822
- // Mid-loop call — the manager owns its own retry budget; signal
1823
- // exhaustion to escalate to convergence.
1824
- return {
1825
- compacted: true,
1826
- messages: [
1827
- {
1828
- role: "user" as const,
1829
- content: [{ type: "text", text: "mid-loop compacted" }],
1830
- },
1831
- ] as Message[],
1832
- compactedPersistedMessages: 5,
1833
- summaryText: "Mid-loop summary",
1834
- previousEstimatedInputTokens: 170_000,
1835
- estimatedInputTokens: 165_000,
1836
- maxInputTokens: 200_000,
1837
- thresholdTokens: 160_000,
1838
- compactedMessages: 10,
1839
- summaryCalls: 1,
1840
- summaryInputTokens: 500,
1841
- summaryOutputTokens: 200,
1842
- summaryModel: "mock-model",
1843
- exhausted: true,
1844
- };
1845
- }
1846
- // Emergency compaction call from auto_compress_latest_turn.
1847
- return {
1848
- compacted: true,
1849
- messages: [
1850
- {
1851
- role: "user" as const,
1852
- content: [{ type: "text", text: "compacted" }],
1853
- },
1854
- ] as Message[],
1855
- compactedPersistedMessages: 5,
1856
- summaryText: "Emergency summary",
1857
- previousEstimatedInputTokens: 170_000,
1858
- estimatedInputTokens: 90_000,
1859
- maxInputTokens: 200_000,
1860
- thresholdTokens: 160_000,
1861
- compactedMessages: 10,
1862
- summaryCalls: 1,
1863
- summaryInputTokens: 500,
1864
- summaryOutputTokens: 200,
1865
- summaryModel: "mock-model",
1866
- };
1867
- },
1868
- } as unknown as Conversation["contextWindowManager"],
1869
- });
1870
-
1693
+ // WHEN the turn runs
1871
1694
  await runAgentLoopImpl(ctx, "hello", "msg-1", (msg) => events.push(msg));
1872
1695
 
1873
- // The classified error is emitted to the client.
1696
+ // THEN the classified error is emitted to the client
1874
1697
  const errorEvents = events.filter((e) => e.type === "conversation_error");
1875
1698
  expect(errorEvents).toHaveLength(1);
1876
1699
  const errorEvent = errorEvents[0];
@@ -1882,17 +1705,23 @@ describe("session-agent-loop overflow recovery (JARVIS-110)", () => {
1882
1705
  throw new Error("conversation_error event missing `code` field");
1883
1706
  }
1884
1707
 
1885
- // The exit reason is stamped onto the latest llm_request_logs row.
1708
+ // AND the exit reason is stamped onto the latest llm_request_logs row
1886
1709
  expect(setAgentLoopExitReasonOnLatestLogMock).toHaveBeenCalledWith(
1887
1710
  "test-conv",
1888
1711
  "budget_yield_unrecovered",
1889
1712
  );
1890
-
1891
- // A `role="assistant"` notice is persisted via the persistence pipeline.
1892
- // The default persistence terminal calls
1893
- // `addMessage(conversationId, role, content, metadata, addOptions)` —
1894
- // we look for the call whose role positional arg is "assistant" and
1895
- // whose content positional arg mentions compaction.
1713
+ // AND it is stamped exactly once. The loop emits the terminal exit inline
1714
+ // as it breaks, then the wrapper records the synthetic yield row and stamps
1715
+ // again; both reaching the stamp would double-stamp two real rows. The
1716
+ // loop's inline emit must be suppressed so only the wrapper's post-row
1717
+ // stamp lands.
1718
+ expect(setAgentLoopExitReasonOnLatestLogMock).toHaveBeenCalledTimes(1);
1719
+
1720
+ // AND a `role="assistant"` notice is persisted via the persistence
1721
+ // pipeline. The default persistence terminal calls
1722
+ // `addMessage(conversationId, role, content, metadata, addOptions)`, so we
1723
+ // look for the call whose role positional arg is "assistant" and whose
1724
+ // content positional arg mentions compaction.
1896
1725
  const assistantPersistCall = addMessageMock.mock.calls.find((call) => {
1897
1726
  const role = call[1];
1898
1727
  const content = call[2];