@vellumai/assistant 0.11.8 → 0.11.9-staging.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (326) hide show
  1. package/ARCHITECTURE.md +2 -2
  2. package/Dockerfile +8 -48
  3. package/docker-entrypoint.sh +5 -1
  4. package/docker-kata-apt-env.sh +3 -0
  5. package/docker-kata-apt-shims.sh +127 -0
  6. package/docker-kata-apt-wrapper.sh +45 -0
  7. package/docker-kata-chroot-exec.sh +35 -0
  8. package/docker-kata-pip.sh +8 -2
  9. package/docs/architecture/memory.md +8 -4
  10. package/docs/guardian-request-flow.md +18 -14
  11. package/docs/trusted-contact-access.md +1 -7
  12. package/node_modules/@vellumai/avatar-catalog/src/catalog.ts +7 -1
  13. package/node_modules/@vellumai/avatar-catalog/src/index.ts +1 -1
  14. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/package.json +4 -1
  15. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/guardian-requests.ts +18 -0
  16. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/index.ts +1 -0
  17. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/platform-credential.ts +41 -0
  18. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/reactions.ts +60 -0
  19. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/package.json +4 -1
  20. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/guardian-requests.ts +18 -0
  21. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/index.ts +1 -0
  22. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/platform-credential.ts +41 -0
  23. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/reactions.ts +60 -0
  24. package/node_modules/@vellumai/gateway-client/src/__tests__/guardian-request-contract.test.ts +0 -18
  25. package/node_modules/@vellumai/gateway-client/src/__tests__/inbound-event-kind.test.ts +110 -5
  26. package/node_modules/@vellumai/gateway-client/src/guardian-request-contract.ts +5 -33
  27. package/node_modules/@vellumai/gateway-client/src/inbound-contract.ts +3 -0
  28. package/node_modules/@vellumai/gateway-client/src/inbound-event-kind.ts +100 -9
  29. package/node_modules/@vellumai/gateway-client/src/index.ts +1 -2
  30. package/node_modules/@vellumai/gateway-client/src/outbound-contract.ts +9 -0
  31. package/node_modules/@vellumai/service-contracts/package.json +4 -1
  32. package/node_modules/@vellumai/service-contracts/src/guardian-requests.ts +18 -0
  33. package/node_modules/@vellumai/service-contracts/src/index.ts +1 -0
  34. package/node_modules/@vellumai/service-contracts/src/platform-credential.ts +41 -0
  35. package/node_modules/@vellumai/service-contracts/src/reactions.ts +60 -0
  36. package/openapi.yaml +193 -4
  37. package/package.json +2 -2
  38. package/src/__tests__/access-request-card-view.test.ts +6 -5
  39. package/src/__tests__/access-request-seed-content-blocks.test.ts +5 -2
  40. package/src/__tests__/agent-loop-mutable-latest-user-message.test.ts +43 -82
  41. package/src/__tests__/always-loaded-tools-guard.test.ts +8 -2
  42. package/src/__tests__/attachment-stored-path-annotation.test.ts +80 -0
  43. package/src/__tests__/attachments-store.test.ts +27 -0
  44. package/src/__tests__/attachments.test.ts +43 -0
  45. package/src/__tests__/canned-reply-release.test.ts +48 -5
  46. package/src/__tests__/channel-delivery-store.test.ts +129 -18
  47. package/src/__tests__/channel-reply-delivery.test.ts +553 -93
  48. package/src/__tests__/channel-retry-sweep.test.ts +199 -0
  49. package/src/__tests__/client-os-metadata-persistence.test.ts +32 -3
  50. package/src/__tests__/conversation-error.test.ts +15 -0
  51. package/src/__tests__/conversation-load-history-repair.test.ts +118 -0
  52. package/src/__tests__/conversation-pairing.test.ts +141 -1
  53. package/src/__tests__/conversation-routes-disk-view.test.ts +28 -1
  54. package/src/__tests__/conversation-routes-slash-commands.test.ts +125 -0
  55. package/src/__tests__/conversation-runtime-assembly.test.ts +79 -0
  56. package/src/__tests__/conversation-store-ephemeral.test.ts +323 -10
  57. package/src/__tests__/conversation-sync-tags.test.ts +2 -37
  58. package/src/__tests__/credential-health-service.test.ts +98 -10
  59. package/src/__tests__/delete-propagation.test.ts +469 -0
  60. package/src/__tests__/dm-persistence.test.ts +16 -0
  61. package/src/__tests__/docker-kata-apt-shims.test.ts +135 -0
  62. package/src/__tests__/forbidden-legacy-symbols.test.ts +12 -0
  63. package/src/__tests__/gemini-provider.test.ts +46 -0
  64. package/src/__tests__/guardian-card-withdrawal.test.ts +0 -22
  65. package/src/__tests__/guardian-gateway-sim.ts +0 -23
  66. package/src/__tests__/guardian-question-mode.test.ts +108 -0
  67. package/src/__tests__/guardian-reply-router-answer-mode.test.ts +0 -2
  68. package/src/__tests__/guardian-routing-invariants.test.ts +26 -12
  69. package/src/__tests__/host-proxy-interface.test.ts +10 -0
  70. package/src/__tests__/inbound-slack-persistence.test.ts +16 -0
  71. package/src/__tests__/injector-v3-suppression.test.ts +43 -9
  72. package/src/__tests__/list-messages-page-latest.test.ts +127 -0
  73. package/src/__tests__/list-messages-system-card.test.ts +103 -0
  74. package/src/__tests__/media-resolve-image-validation.test.ts +77 -0
  75. package/src/__tests__/notification-decision-fallback.test.ts +199 -77
  76. package/src/__tests__/notification-decision-strategy.test.ts +198 -213
  77. package/src/__tests__/notification-discord-adapter.test.ts +25 -0
  78. package/src/__tests__/notification-slack-adapter.test.ts +133 -0
  79. package/src/__tests__/notification-telegram-adapter.test.ts +36 -0
  80. package/src/__tests__/openai-provider.test.ts +5 -5
  81. package/src/__tests__/outbound-slack-persistence.test.ts +85 -72
  82. package/src/__tests__/persist-user-message-set-processing-failure.test.ts +83 -65
  83. package/src/__tests__/platform-client-verify-credential.test.ts +100 -0
  84. package/src/__tests__/plugin-import-boundary-guard.test.ts +1 -1
  85. package/src/__tests__/pricing.test.ts +11 -0
  86. package/src/__tests__/process-message-display-content.test.ts +20 -9
  87. package/src/__tests__/processing-acquire-fenced-guard.test.ts +56 -0
  88. package/src/__tests__/provider-meta-persistence.test.ts +67 -0
  89. package/src/__tests__/reaction-persistence.test.ts +22 -198
  90. package/src/__tests__/run-conversation-turn-persistence.test.ts +5 -2
  91. package/src/__tests__/scripted-turn-metadata-persistence.test.ts +16 -0
  92. package/src/__tests__/skill-load-tool.test.ts +27 -0
  93. package/src/__tests__/skills.test.ts +34 -1
  94. package/src/__tests__/strip-memory-injections.test.ts +3 -4
  95. package/src/__tests__/terminal-tools.test.ts +9 -0
  96. package/src/__tests__/thread-backfill.test.ts +5 -3
  97. package/src/__tests__/unified-turn-context-location.test.ts +76 -0
  98. package/src/__tests__/voice-session-bridge.test.ts +18 -0
  99. package/src/__tests__/watch-retro-report-payload.test.ts +185 -0
  100. package/src/__tests__/watch-retro-tool-availability.test.ts +82 -0
  101. package/src/__tests__/workspace-migration-151-repair-renamed-fireworks-deepseek-pro-model-id.test.ts +235 -0
  102. package/src/__tests__/workspace-migration-152-repair-retired-fireworks-minimax-m2p7-model-id.test.ts +233 -0
  103. package/src/agent/attachments.ts +13 -2
  104. package/src/agent/loop.ts +3 -19
  105. package/src/api/README.md +9 -5
  106. package/src/api/index.ts +9 -8
  107. package/src/api/package.json +1 -0
  108. package/src/api/responses/conversation-message.ts +8 -0
  109. package/src/api/responses/home.ts +7 -18
  110. package/src/api/surfaces.ts +114 -7
  111. package/src/approvals/AGENTS.md +1 -1
  112. package/src/channels/__tests__/gateway-guardian-requests.test.ts +1 -23
  113. package/src/channels/__tests__/types.test.ts +22 -1
  114. package/src/channels/gateway-guardian-requests.ts +0 -22
  115. package/src/channels/types.ts +8 -6
  116. package/src/cli/__tests__/catalog-search-help.test.ts +18 -0
  117. package/src/cli/commands/channels/__tests__/channels.test.ts +26 -0
  118. package/src/cli/commands/channels/index.help.ts +19 -6
  119. package/src/cli/commands/channels/index.ts +6 -2
  120. package/src/cli/commands/db/__tests__/status.test.ts +22 -0
  121. package/src/cli/commands/db/index.help.ts +1 -1
  122. package/src/cli/commands/db/status.ts +172 -1
  123. package/src/cli/commands/platform/__tests__/connect.test.ts +29 -0
  124. package/src/cli/commands/platform/connect.ts +14 -5
  125. package/src/cli/commands/plugins.help.ts +6 -5
  126. package/src/cli/lib/__tests__/install-from-github.test.ts +0 -8
  127. package/src/cli/lib/__tests__/install-from-platform.test.ts +72 -0
  128. package/src/cli/lib/__tests__/plugin-catalog-local.test.ts +33 -0
  129. package/src/cli/lib/bundled-marketplace.json +14 -0
  130. package/src/cli/lib/install-from-github.ts +9 -5
  131. package/src/cli/lib/install-from-platform.ts +12 -1
  132. package/src/config/__tests__/assistant-initiated-threads-gate.test.ts +61 -0
  133. package/src/config/assistant-initiated-threads-gate.ts +50 -0
  134. package/src/config/bundled-skills/acp/SKILL.md +10 -3
  135. package/src/config/bundled-skills/schedule/SKILL.md +25 -11
  136. package/src/config/bundled-skills/schedule/references/SCRIPT_MODE_PATTERNS.md +3 -1
  137. package/src/config/call-site-defaults.ts +4 -1
  138. package/src/config/feature-flag-registry.json +21 -4
  139. package/src/config/schemas/memory-v3.ts +4 -3
  140. package/src/context/strip-injections.ts +17 -56
  141. package/src/conversations/__tests__/message-consolidation.test.ts +39 -0
  142. package/src/conversations/message-consolidation.ts +4 -3
  143. package/src/credential-health/credential-health-service.ts +130 -0
  144. package/src/daemon/__tests__/conversation-tool-setup.test.ts +31 -0
  145. package/src/daemon/conversation-agent-loop-handlers.ts +55 -59
  146. package/src/daemon/conversation-error.ts +2 -0
  147. package/src/daemon/conversation-messaging.ts +195 -47
  148. package/src/daemon/conversation-process.ts +20 -5
  149. package/src/daemon/conversation-runtime-assembly.ts +48 -39
  150. package/src/daemon/conversation-store.ts +159 -11
  151. package/src/daemon/conversation-tool-setup.ts +9 -2
  152. package/src/daemon/conversation.ts +290 -19
  153. package/src/daemon/dictation-text-processing.ts +2 -7
  154. package/src/daemon/handlers/config-channels.ts +33 -3
  155. package/src/daemon/handlers/config-model.test.ts +1 -0
  156. package/src/daemon/handlers/shared.ts +9 -0
  157. package/src/daemon/port-oversized-content.test.ts +117 -0
  158. package/src/daemon/port-oversized-content.ts +110 -0
  159. package/src/daemon/process-message.ts +87 -16
  160. package/src/daemon/reaction-record.test.ts +23 -11
  161. package/src/daemon/reaction-record.ts +6 -9
  162. package/src/documents/document-store.ts +1 -5
  163. package/src/home/conversation-starter-validation.ts +1 -5
  164. package/src/live-voice/__tests__/live-voice-photo.test.ts +910 -50
  165. package/src/live-voice/__tests__/live-voice-sight-frame-inline.test.ts +31 -25
  166. package/src/live-voice/__tests__/live-voice-sight-frame.test.ts +70 -1
  167. package/src/live-voice/live-voice-photo.ts +544 -61
  168. package/src/live-voice/live-voice-session.ts +6 -2
  169. package/src/live-voice/protocol.ts +20 -1
  170. package/src/messaging/provider-message-metadata.ts +40 -1
  171. package/src/messaging/providers/__tests__/transport-dispatch.test.ts +10 -2
  172. package/src/messaging/providers/channel-transport.ts +28 -0
  173. package/src/messaging/providers/discord/send.ts +3 -2
  174. package/src/messaging/providers/slack/api.ts +11 -1
  175. package/src/messaging/providers/slack/message-metadata.test.ts +133 -0
  176. package/src/messaging/providers/slack/message-metadata.ts +141 -1
  177. package/src/messaging/providers/slack/send.test.ts +58 -0
  178. package/src/messaging/providers/slack/send.ts +4 -4
  179. package/src/messaging/providers/slack/transport.ts +28 -2
  180. package/src/messaging/providers/telegram-bot/send.test.ts +159 -0
  181. package/src/messaging/providers/telegram-bot/send.ts +93 -0
  182. package/src/messaging/providers/telegram-bot/transport.ts +71 -0
  183. package/src/messaging/reaction-envelopes.test.ts +73 -0
  184. package/src/messaging/reaction-envelopes.ts +33 -5
  185. package/src/messaging/read-provider-metadata.ts +4 -3
  186. package/src/monitoring/recovery/__tests__/stranded-delivery-events.test.ts +208 -0
  187. package/src/monitoring/recovery/db.ts +35 -0
  188. package/src/monitoring/recovery/orphaned-channel-events.ts +3 -22
  189. package/src/monitoring/recovery/run-recovery.ts +6 -2
  190. package/src/monitoring/recovery/stale-processing.ts +3 -20
  191. package/src/monitoring/recovery/stranded-delivery-events.ts +68 -0
  192. package/src/notifications/AGENTS.md +2 -2
  193. package/src/notifications/README.md +2 -2
  194. package/src/notifications/__tests__/assistant-reply-producer.test.ts +11 -9
  195. package/src/notifications/__tests__/broadcaster.test.ts +35 -4
  196. package/src/notifications/__tests__/notification-utils.test.ts +31 -0
  197. package/src/notifications/access-request-copy.ts +126 -114
  198. package/src/notifications/adapters/discord.ts +9 -5
  199. package/src/notifications/adapters/shared.ts +17 -4
  200. package/src/notifications/adapters/slack.ts +14 -4
  201. package/src/notifications/adapters/telegram.ts +8 -4
  202. package/src/notifications/approval-card-data.ts +5 -2
  203. package/src/notifications/assistant-reply-producer.ts +7 -7
  204. package/src/notifications/broadcaster.ts +57 -25
  205. package/src/notifications/conversation-pairing.ts +68 -2
  206. package/src/notifications/copy-composer.ts +13 -32
  207. package/src/notifications/decision-engine.ts +56 -237
  208. package/src/notifications/guardian-delivery-recorder.ts +5 -9
  209. package/src/notifications/guardian-feed-projection.ts +0 -15
  210. package/src/notifications/guardian-question-mode.ts +147 -6
  211. package/src/notifications/notification-utils.ts +117 -1
  212. package/src/permissions/confirmation-guardian-request.ts +2 -2
  213. package/src/permissions/prompter.ts +1 -1
  214. package/src/persistence/conversation-crud.ts +93 -14
  215. package/src/persistence/conversation-queries.ts +144 -17
  216. package/src/persistence/conversation-types.ts +39 -5
  217. package/src/persistence/delivery-crud.ts +224 -28
  218. package/src/persistence/delivery-status.ts +21 -2
  219. package/src/persistence/migrations/374-channel-inbound-message-id-index.ts +26 -0
  220. package/src/persistence/migrations/375-create-channel-outbound-posts.ts +52 -0
  221. package/src/persistence/migrations/__tests__/375-create-channel-outbound-posts.test.ts +84 -0
  222. package/src/persistence/schema/conversations.ts +71 -2
  223. package/src/persistence/schema-contract.test.ts +78 -0
  224. package/src/persistence/schema-contract.ts +96 -0
  225. package/src/persistence/steps.ts +4 -0
  226. package/src/platform/client.ts +55 -0
  227. package/src/plugins/__tests__/mcp-servers.test.ts +46 -0
  228. package/src/plugins/defaults/injector-order.ts +2 -1
  229. package/src/plugins/defaults/memory/__tests__/memory-retrospective-accounting.test.ts +2 -2
  230. package/src/plugins/defaults/memory/context-search/sources/conversations.ts +13 -15
  231. package/src/plugins/defaults/memory/hooks/user-prompt-submit.ts +10 -0
  232. package/src/plugins/defaults/memory/injectors.ts +1 -1
  233. package/src/plugins/defaults/memory/memory-marker.ts +17 -7
  234. package/src/plugins/defaults/memory/substrate/__tests__/consolidation-prompt-flag-gating-guard.test.ts +1 -5
  235. package/src/plugins/defaults/memory/substrate/__tests__/skill-content.test.ts +17 -0
  236. package/src/plugins/defaults/memory/substrate/skill-content.ts +7 -0
  237. package/src/plugins/defaults/memory/v3/__tests__/carry-integration.test.ts +54 -41
  238. package/src/plugins/defaults/memory/v3/__tests__/injection.test.ts +3 -3
  239. package/src/plugins/defaults/memory/v3/__tests__/render-injection.test.ts +22 -0
  240. package/src/plugins/defaults/memory/v3/injector.ts +14 -15
  241. package/src/plugins/defaults/memory/v3/prune.test.ts +20 -2
  242. package/src/plugins/defaults/memory/v3/render-injection.ts +10 -1
  243. package/src/plugins/defaults/memory/v3/types.ts +15 -4
  244. package/src/plugins/defaults/turn-context/injectors.ts +3 -0
  245. package/src/plugins/defaults/turn-context/unified-turn-context.ts +24 -0
  246. package/src/plugins/mcp-servers.ts +14 -2
  247. package/src/providers/__tests__/dispatch-connection-routing.test.ts +50 -0
  248. package/src/providers/__tests__/registry-native-web-search.test.ts +49 -2
  249. package/src/providers/anthropic/client.ts +21 -24
  250. package/src/providers/call-site-routing.ts +5 -4
  251. package/src/providers/connection-resolution.ts +29 -1
  252. package/src/providers/content-block-size.test.ts +82 -0
  253. package/src/providers/content-block-size.ts +99 -0
  254. package/src/providers/file-block-text.test.ts +64 -0
  255. package/src/providers/file-block-text.ts +28 -0
  256. package/src/providers/gemini/client.ts +15 -9
  257. package/src/providers/inference/auth.ts +6 -6
  258. package/src/providers/media-resolve.ts +14 -0
  259. package/src/providers/model-catalog.ts +47 -34
  260. package/src/providers/openai/chat-completions-provider.ts +9 -25
  261. package/src/providers/openai/responses-provider.ts +11 -18
  262. package/src/providers/registry.ts +10 -7
  263. package/src/providers/routing-identity.ts +2 -1
  264. package/src/providers/types.ts +1 -4
  265. package/src/providers/vellum-model-routing.ts +3 -2
  266. package/src/runtime/AGENTS.md +40 -15
  267. package/src/runtime/approval-message-composer.ts +1 -6
  268. package/src/runtime/assistant-event-hub.ts +2 -39
  269. package/src/runtime/channel-approval-types.ts +1 -5
  270. package/src/runtime/channel-reply-delivery.ts +182 -115
  271. package/src/runtime/{slack-reply-session.test.ts → channel-reply-session.test.ts} +294 -156
  272. package/src/runtime/{slack-reply-session.ts → channel-reply-session.ts} +107 -86
  273. package/src/runtime/channel-retry-sweep.ts +102 -1
  274. package/src/runtime/finalize-event-delivery.ts +4 -4
  275. package/src/runtime/guardian-action-message-composer.ts +1 -4
  276. package/src/runtime/guardian-reply-router.ts +4 -75
  277. package/src/runtime/http-router.ts +1 -5
  278. package/src/runtime/question-request-guardian-bridge.ts +2 -3
  279. package/src/runtime/routes/__tests__/conversation-list-assistant-section.test.ts +359 -0
  280. package/src/runtime/routes/__tests__/dictation-command-mode.test.ts +120 -0
  281. package/src/runtime/routes/__tests__/sight-frame-routes.test.ts +487 -0
  282. package/src/runtime/routes/acp-routes.ts +1 -1
  283. package/src/runtime/routes/canned-reply-release.ts +19 -9
  284. package/src/runtime/routes/channel-route-shared.ts +0 -39
  285. package/src/runtime/routes/channel-verification-routes.ts +4 -1
  286. package/src/runtime/routes/conversation-list-routes.ts +44 -3
  287. package/src/runtime/routes/conversation-management-routes.ts +2 -1
  288. package/src/runtime/routes/conversation-routes.ts +138 -59
  289. package/src/runtime/routes/diagnostics-routes.ts +111 -60
  290. package/src/runtime/routes/guardian-action-routes.ts +10 -14
  291. package/src/runtime/routes/guardian-approval-interception.ts +0 -5
  292. package/src/runtime/routes/inbound-message-handler.ts +170 -48
  293. package/src/runtime/routes/inbound-stages/acl-enforcement.ts +4 -3
  294. package/src/runtime/routes/inbound-stages/background-dispatch.test.ts +169 -8
  295. package/src/runtime/routes/inbound-stages/background-dispatch.ts +82 -14
  296. package/src/runtime/routes/inbound-stages/guardian-reply-intercept.ts +23 -48
  297. package/src/runtime/routes/inbound-stages/reaction-intercept.test.ts +454 -94
  298. package/src/runtime/routes/inbound-stages/reaction-intercept.ts +309 -102
  299. package/src/runtime/routes/index.ts +2 -0
  300. package/src/runtime/routes/platform-routes.ts +99 -10
  301. package/src/runtime/routes/sight-frame-routes.ts +136 -0
  302. package/src/runtime/sync/resource-sync-events.ts +14 -37
  303. package/src/runtime/sync/sync-publisher.test.ts +5 -4
  304. package/src/tools/__tests__/tool-input-schemas.test.ts +2 -0
  305. package/src/tools/client-os.ts +11 -1
  306. package/src/tools/host-filesystem/edit.ts +2 -2
  307. package/src/tools/host-filesystem/read.ts +2 -2
  308. package/src/tools/host-filesystem/transfer.ts +2 -2
  309. package/src/tools/host-filesystem/write.ts +2 -2
  310. package/src/tools/host-terminal/host-shell.ts +2 -2
  311. package/src/tools/skills/load.ts +1 -1
  312. package/src/tools/terminal/safe-env.ts +4 -0
  313. package/src/tools/tool-input-schemas.ts +2 -0
  314. package/src/tools/tool-manifest.ts +2 -0
  315. package/src/tools/ui-surface/definitions.ts +4 -3
  316. package/src/tools/watch/watch-retro-report.ts +205 -0
  317. package/src/watch/__tests__/watch-retro.test.ts +268 -40
  318. package/src/watch/watch-retro.ts +190 -77
  319. package/src/workspace/migrations/151-repair-renamed-fireworks-deepseek-pro-model-id.ts +195 -0
  320. package/src/workspace/migrations/152-repair-retired-fireworks-minimax-m2p7-model-id.ts +198 -0
  321. package/src/workspace/migrations/__tests__/150-stt-flux-provider-to-model-family.test.ts +0 -10
  322. package/src/workspace/migrations/registry.ts +4 -0
  323. package/docker-kata-pip-chroot.sh +0 -22
  324. package/src/__tests__/slack-reaction-approvals.test.ts +0 -97
  325. package/src/__tests__/slack-reaction-guardian-approval.test.ts +0 -307
  326. package/src/api/events/conversation-list-invalidated.ts +0 -38
@@ -22,9 +22,15 @@
22
22
  * everything it recorded, and the timeline outlives the turn.
23
23
  */
24
24
 
25
+ import { randomUUID } from "node:crypto";
26
+
27
+ import {
28
+ type WatchRetroSurfaceData,
29
+ WatchRetroSurfaceDataSchema,
30
+ } from "../api/surfaces.js";
25
31
  import {
32
+ addMessage,
26
33
  getMessages,
27
- isStandaloneAssistantMessage,
28
34
  setConversationSurfaced,
29
35
  } from "../persistence/conversation-crud.js";
30
36
  import type { WakeOptions } from "../runtime/agent-wake.js";
@@ -48,22 +54,55 @@ const log = getLogger("watch-retro");
48
54
  const WATCH_RETRO_WAKE_SOURCE = "watch-retro";
49
55
 
50
56
  /**
51
- * What the retro asks for, and the order it asks in.
52
- *
53
- * **The ask comes first, and it is the shorter half.** What the user owes this
54
- * turn is a handful of answers; everything else is the assistant showing its
55
- * work. Leading with the report buries the one part that needs them under the
56
- * part that does not, and a reader who has to reach the bottom to find the
57
- * question has already been asked for more than the question was worth.
57
+ * What the retro reports, and the shape it reports in.
58
+ *
59
+ * **It is a card, not a turn of prose.** The turn ends by calling
60
+ * `watch_retro_report`, and this module turns that call into a `card` surface
61
+ * under the `watch_retro` template, which the client draws as a paged card: the
62
+ * record on the first page, one question per page after it.
63
+ *
64
+ * **The daemon appends the card; the model never reaches `ui_show`.** This wake
65
+ * is `clientless`, and `conversation-tool-setup` gates the whole `ui_surface`
66
+ * family on a client being present, so `ui_show` is absent from this turn's
67
+ * tool set rather than merely denied. A retrospective told to call it can only
68
+ * report that it cannot. `watch_retro_report` is an ordinary tool and passes
69
+ * that gate, and the surface is appended here, the way the memory
70
+ * retrospective's `skill_card` is appended by the daemon rather than requested
71
+ * by the model.
72
+ *
73
+ * **The append waits for the turn to end.** A `ui_surface` row written mid-turn
74
+ * can land between a persisted `tool_use` and its `tool_result`, an ordering
75
+ * strict OpenAI-compatible backends reject; the skill card defers around it,
76
+ * and running after `dispatch` resolves avoids it outright. The report survives
77
+ * the wait in the transcript, as the tool call's own input, so nothing is held
78
+ * in memory between the call and the append.
79
+ *
80
+ * **A template rather than a surface type of its own, so an older client still
81
+ * gets the report.** The macOS app ships its own renderer and floats its CLI to
82
+ * the npm `latest` tag (`clients/macos/src/main/cli-installer.ts`), so this
83
+ * assistant runs behind renderers that predate it. One that does not recognize
84
+ * a surface *type* renders an unsupported-surface notice and nothing else; one
85
+ * that does not recognize a card *template* still renders the card, falling
86
+ * back to `title`, `subtitle` and `body`. The card is the entire account of the
87
+ * session and the instructions below allow no prose beside it, so the degraded
88
+ * path has to carry the record rather than an error. That is what `body` is
89
+ * for, and why it repeats in prose what `templateData` carries in structure.
90
+ *
91
+ * **The record leads, and the paging is what allows that.** A question on its
92
+ * own page is not competing with the account for attention, the progress bar
93
+ * says how much is left, and the record is what the user needs in order to
94
+ * answer anything else. The two are one decision: the questions may sit behind
95
+ * the record only for as long as they have pages of their own. Prose has no
96
+ * second axis, so a report collapsed back into a single block has to put its
97
+ * questions first or bury them.
58
98
  *
59
99
  * **It asks about what it does not know, not about what it just wrote.** The
60
100
  * `skill-management` skill will not scaffold until four points are settled:
61
101
  * what the skill does, its trigger phrases, its major steps, and its
62
- * destructive step and done condition. Its checkpoint is explicit that the
63
- * ones to raise are the ones being guessed at. Re-asking all four regardless
64
- * turns the report into a questionnaire about itself, where the user confirms
65
- * a list of steps printed directly above the question asking whether those are
66
- * the steps.
102
+ * destructive step and done condition. The ones to raise are the ones being
103
+ * guessed at. Re-asking all four regardless turns the report into a
104
+ * questionnaire about itself, where the user confirms a list of steps printed
105
+ * on the page before.
67
106
  *
68
107
  * The trigger phrase is always one of the open ones, because it is the single
69
108
  * field the recording cannot supply: the timeline holds what they did, never
@@ -74,60 +113,69 @@ const WATCH_RETRO_WAKE_SOURCE = "watch-retro";
74
113
  * was seen. The recording establishes what someone did once; it establishes
75
114
  * nothing about whether they want it done again without being asked, and the
76
115
  * gap between those two is the whole risk of turning a demonstration into a
77
- * skill. `skill-management` will not scaffold until that step is settled
78
- * either, so an unasked one stalls the flow it was meant to feed.
79
- *
80
- * **The skill loads before the report is written, and nothing follows the
81
- * report.** This is about where the report lands on screen, not about the
82
- * order the work happens in. A client renders an assistant turn as its final
83
- * prose plus the intermediate work that led there, and the web transcript
84
- * collapses that intermediate half into an "Earlier activity" disclosure that
85
- * a settled turn shows closed (`finalResponseGroupIndex` in
86
- * `transcript-message-body.tsx`). Everything before the turn's last non-empty
87
- * text block is intermediate by that rule, including prose. Writing the report
88
- * first and loading the skill after it puts a `skill_load` and whatever the
89
- * model says once the skill is in hand *after* the report, which demotes the
90
- * report to intermediate work and hides it behind a collapsed row next to the
91
- * reasoning rows. The user presses stop on a session and is shown a folded-up
92
- * link where the account of it should be.
93
- *
94
- * So the load comes first and the report is last. `skill-management` opens on
95
- * "Ask before doing anything", so a turn that loads it and then asks is
96
- * following it rather than jumping the flow, and the report is the alignment
97
- * its first step calls for. The instruction that nothing follows the report is
98
- * as load-bearing as the ordering: one trailing line of sign-off is another
99
- * text block after it, and the report is intermediate again.
100
- *
101
- * **The report does not depend on the load succeeding.** `skill-management` is
102
- * a selector, not a fixed definition: `loadSkillCatalog` lets a managed or
116
+ * skill.
117
+ *
118
+ * **No question is a yes/no whose "no" teaches nothing.** "Priority came from
119
+ * the event count, under 100 is Medium?" packs an entire inferred rule into a
120
+ * binary: a "no" costs a round trip and comes back with nothing in it, and the
121
+ * user is now owed a follow-up they cannot see. Every question is a pick from
122
+ * named alternatives instead, with the model's own reading among them and
123
+ * marked as such. A yes/no survives only where the alternatives genuinely are
124
+ * yes and no, which in practice is the destructive gate and nothing else.
125
+ *
126
+ * The rule that falls out is the one worth keeping: if the alternatives cannot
127
+ * be enumerated, the question is not ready to be asked, and the honest move is
128
+ * to leave the step described rather than explained.
129
+ *
130
+ * **Every page is skippable, and every skip lands somewhere safe.** A skipped
131
+ * `fill` keeps its pre-filled suggestion, so it is an edit rather than a blank.
132
+ * A skipped `pick` takes the first option, which is the reading the recording
133
+ * already supports. A skipped `gate` takes the first option too, and on a gate
134
+ * that first option must be the cautious one. It is the single place where the
135
+ * default is deliberately not the model's guess.
136
+ *
137
+ * **The skill loads before the card is shown, and nothing follows the card.**
138
+ * `skill-management` opens on "Ask before doing anything", so a turn that loads
139
+ * it and then asks is following it rather than jumping the flow. Ending on the
140
+ * `ui_show` keeps the card as the last thing in the turn, where prose written
141
+ * after it would read as a sign-off nobody asked for.
142
+ *
143
+ * **The card does not depend on the load succeeding.** `skill-management` is a
144
+ * selector, not a fixed definition: `loadSkillCatalog` lets a managed or
103
145
  * workspace skill of the same id replace the bundled one, and
104
146
  * `resolveSkillSelector` hands back whichever entry won. So the load can come
105
147
  * back as someone else's skill, or as a refusal, and putting it ahead of the
106
- * report is what makes either one land before the user has been told anything.
107
- * A refusal is the likelier of the two here: this wake is `clientless`, and an
148
+ * card is what makes either one land before the user has been told anything. A
149
+ * refusal is the likelier of the two here: this wake is `clientless`, and an
108
150
  * inline-command load with no human present is denied outright
109
- * (`permissions/checker.ts`, `isDynamicSkillLoadInvocation`). The bundled skill
110
- * carries no such expansions, so the ordinary path is unaffected, but a shadow
111
- * that carries them is denied on arrival.
151
+ * (`permissions/checker.ts`, `isDynamicSkillLoadInvocation`).
112
152
  *
113
153
  * Neither outcome is allowed to become the retro. The session was recorded, the
114
154
  * timeline is already in this prompt, and the account of it is the one thing the
115
155
  * user is owed for having pressed stop; a permission error where that account
116
156
  * should be is the same empty thread the `surfaceConversation` guard exists to
117
- * prevent, just with prose in it. So the instructions write the report either
118
- * way and say the handoff did not happen, rather than reporting on the load.
157
+ * prevent. So the instructions show the card either way and say the handoff did
158
+ * not happen in the card's own coverage line, rather than reporting on the load.
119
159
  */
120
- const RETRO_INSTRUCTIONS = `Load the \`skill-management\` skill first, before you write anything, and follow it. Do not author or scaffold a skill until the four points that step names are settled: the report below is the alignment its first step calls for.
160
+ const RETRO_INSTRUCTIONS = `Load the \`skill-management\` skill first, before you do anything else, and follow it. Do not author or scaffold a skill yet: the report below is the alignment its first step calls for, and the answers come back before anything is written.
121
161
 
122
- Write the report below whether or not that load succeeds. If it fails, is refused, or comes back as something other than the skill-management flow, do not report on the load and do not retry it: write the report, and end it with one line saying you could not open the skill-authoring flow, so the user knows the handoff is the part that did not happen.
162
+ Then make exactly one \`watch_retro_report\` call. That call is the last thing you do this turn. Nothing follows it: no prose, no sign-off, no note about what you loaded, no further tool call. The report is drawn as a card in the conversation once this turn ends, so writing the same thing again in prose would show the user two of it. Report whether or not the skill loaded; if it did not, add one short sentence to \`coverage\` saying you could not open the skill-authoring flow, so the user knows the handoff is the part that did not happen.
123
163
 
124
- Then write back to the user in two sections, in this order, both as level-2 headings. That report is the last thing you do this turn, and nothing follows it: no sign-off, no note about what you loaded, no further tool call. Their answers come back as an ordinary reply and the flow picks up from there.
164
+ The payload:
125
165
 
126
- First, "What I need from you". The questions you cannot answer from the recording, numbered, most consequential first. Each one concrete enough to answer in a sentence, and each one about something you are genuinely guessing at: a value you could not read, a choice whose rule you could not infer, a step you only saw the result of. Always ask what they would say to start this task, in their own words, because the recording cannot tell you that. Ask about the done condition if it is unclear. Always confirm any destructive or irreversible step, even one the recording showed plainly: watching someone do a thing once is not agreement to have it done again unattended, and this is the one place the rule below does not apply. Otherwise do not ask them to confirm something the recording already showed you.
166
+ - \`task\`: the task, named the way the user would name it. Six words at most, and no trailing clause explaining it: it is a card title, not a sentence.
167
+ - \`purpose\`: what it is for, in under twelve words. Skip it when the task already says it; a line restating the title is worse than no line.
168
+ - \`steps\`: the steps in order, as short imperative fragments: "Open the Sentry issue", not "You opened the Sentry issue from the alert email". Three to eight of them. Concrete enough to follow, carrying no purpose of their own.
169
+ - \`eyebrow\`: what the session cost, in the teaching's own words, e.g. "Taught in 4 min" or "Taught in 4 min, 11 screens". Never "watched": the user taught you this, they did not perform for you.
170
+ - \`questions\`: at most three, most consequential first. Fewer is better, and none is a valid answer if the recording settled everything. Each question is \`{ id, kind, prompt }\`: \`prompt\` is the question worded the way you would ask it out loud, and \`id\` is a handle no other question on this card uses. A \`pick\` or a \`gate\` adds \`options\`, each of them \`{ id, label }\` and optionally a \`note\`: \`label\` is the answer as the user reads it, and \`id\` is that option's own handle. Use those names exactly. A question's text is \`prompt\` and never \`question\` or \`text\`; an option's text is \`label\` and never \`value\` or \`title\`. Anything sent under another name is dropped on the way to the card, and the page it belonged to is lost.
127
171
 
128
- Second, "What I saw". Open with one sentence naming the task and what it is for, on its own and not as a list item. Then the steps in order beneath it, one line each and concrete enough to follow, carrying no purpose of their own. This is the record your questions sit on top of, so state it rather than asking about it.
172
+ Every question is answerable in one tap and every one is skippable, so ask only about what you are genuinely guessing at: a value you could not read, a choice whose rule you could not infer, a step you only saw the result of. Do not ask the user to confirm something the recording already showed you.
129
173
 
130
- Open on the first heading. No preamble, no announcing what you are about to do, no narrating which skills you are loading. Correct your reading against whatever they tell you. If they decide this is not worth keeping, say so and stop.`;
174
+ - \`kind: "fill"\` is a single text field, and there is at most one of them: what they would say to start this task, in their own words. Always ask it, because the recording cannot tell you. Put your best guess in \`suggestion\` so skipping keeps a working phrase instead of leaving it blank.
175
+ - \`kind: "pick"\` is two to four named alternatives. The first option is the default and must be the reading the recording supports; mark it with a \`note\` saying so. Never ask a yes/no whose "no" tells you nothing: "was the rule X?" wastes the question, where "what decides this?" with X first among the options gets an answer either way. If you cannot name the alternatives, you do not understand the gap well enough to ask about it, so leave the step described and ask nothing.
176
+ - \`kind: "gate"\` is for a destructive or irreversible step, and it is asked however plainly the step was seen. Not "did you do this" (you watched them), but whether you may do it unattended. The first option must be the cautious one ("Ask me first"), because a skipped question takes it.
177
+
178
+ Ask about the done condition only if it is genuinely unclear, and as a \`pick\`.`;
131
179
 
132
180
  /**
133
181
  * Told to the model whenever the render was bounded, naming the bound that
@@ -154,6 +202,12 @@ function coverageNotice(render: WatchTimelineRender): string {
154
202
  return `This is a partial recording. The session logged ${render.totalEntries} entries and the timeline below carries only the ${render.entries.length} most recent of them, so the first ${dropped} are missing entirely. Treat the beginning of the task as something to ask about rather than something to state. What is here may also be cut short in places. Say plainly what you could not read instead of filling it in.`;
155
203
  }
156
204
 
205
+ /** The card template the retro reports through. */
206
+ const WATCH_RETRO_TEMPLATE = "watch_retro";
207
+
208
+ /** The tool a retrospective hands its report to. */
209
+ const WATCH_RETRO_TOOL_NAME = "watch_retro_report";
210
+
157
211
  /** The element the recording is fenced in. */
158
212
  const TIMELINE_TAG = "watch-timeline";
159
213
 
@@ -374,7 +428,15 @@ async function dispatchWatchRetro(
374
428
  if (!dispatched.invoked) {
375
429
  return { status: "failed", reason: dispatched.reason ?? "unknown" };
376
430
  }
377
- if (!hasReport(summary.conversationId, priorMessageIds)) {
431
+ // The card is what the user is owed, so a turn that made no usable report
432
+ // call has produced nothing regardless of what else it wrote. Appending is
433
+ // also the test: there is no separate "did it report" check that could
434
+ // disagree with whether a card actually landed.
435
+ const surfaceId = await appendRetroCard(
436
+ summary.conversationId,
437
+ priorMessageIds,
438
+ );
439
+ if (surfaceId === null) {
378
440
  return { status: "failed", reason: "no_report" };
379
441
  }
380
442
 
@@ -405,36 +467,87 @@ function messageIds(conversationId: string): ReadonlySet<string> {
405
467
  }
406
468
 
407
469
  /**
408
- * Whether the turn left the user something to read.
409
- *
410
- * Asked of the conversation rather than of the dispatch result, because a wake
411
- * reports invocation and a report is a stronger thing. `inspectWakeOutput`
412
- * counts a `tool_use` block as output, so a retro whose first act is loading
413
- * the `skill-management` skill has already "produced output" before it has
414
- * said anything, and a run that then stops or errors still returns
415
- * `invoked: true`. Surfacing on that gives the user a thread of tool plumbing
416
- * with no report and no question in it.
417
- *
418
- * Standalone assistant rows are not reports either: that is the shape a
419
- * provider error takes, which persists as an assistant message and returns
420
- * normally, and a system card is machinery rather than an account of the
421
- * session.
470
+ * Turn the turn's `watch_retro_report` call into the card the user is shown.
471
+ *
472
+ * **Read back out of history rather than handed over.** The tool records and
473
+ * returns; nothing is kept in memory between the call and this append, so a
474
+ * crash in between loses no report that was actually made. The turn's own
475
+ * `tool_use` block is the record, and its `input` is the payload.
476
+ *
477
+ * **The newest call wins.** A model that corrects itself calls again rather
478
+ * than editing, so the last call in the turn is the one it stands behind.
479
+ *
480
+ * **Parsed again here.** A tool result is a promise about validation, not about
481
+ * what was persisted, and the block this reads has been through the provider
482
+ * and the message store since. The card's own schema is what the renderer
483
+ * trusts, so it is what the payload is held to before a row is written.
484
+ *
485
+ * Returns the appended surface id, or null when the turn made no usable call.
422
486
  */
423
- function hasReport(
487
+ async function appendRetroCard(
424
488
  conversationId: string,
425
489
  priorMessageIds: ReadonlySet<string>,
426
- ): boolean {
427
- return getMessages(conversationId).some((message) => {
490
+ ): Promise<string | null> {
491
+ let payload: WatchRetroSurfaceData | null = null;
492
+ for (const message of getMessages(conversationId)) {
428
493
  if (priorMessageIds.has(message.id) || message.role !== "assistant") {
429
- return false;
494
+ continue;
430
495
  }
431
- if (isStandaloneAssistantMessage(message.role, message.metadata)) {
432
- return false;
496
+ for (const block of message.content) {
497
+ if (block.type !== "tool_use" || block.name !== WATCH_RETRO_TOOL_NAME) {
498
+ continue;
499
+ }
500
+ const parsed = WatchRetroSurfaceDataSchema.safeParse(block.input);
501
+ // A call whose payload cannot be drawn is not a report. The schema is
502
+ // tolerant, so this only rejects what is not an object at all; the task
503
+ // check below is what rejects an empty one.
504
+ if (parsed.success && parsed.data.task.trim().length > 0) {
505
+ payload = parsed.data;
506
+ }
433
507
  }
434
- return message.content.some(
435
- (block) => block.type === "text" && block.text.trim().length > 0,
436
- );
437
- });
508
+ }
509
+ if (payload === null) {
510
+ return null;
511
+ }
512
+
513
+ const surfaceId = `${WATCH_RETRO_TEMPLATE}-${randomUUID()}`;
514
+ const steps = payload.steps;
515
+ // `title`, `subtitle` and `body` are what a renderer too old to know the
516
+ // template draws, and they are the whole report for that reader. Derived here
517
+ // rather than asked of the model, so the degraded view cannot drift from the
518
+ // structured one or be forgotten. Questions stay out of it: that reader has
519
+ // no way to answer them.
520
+ const body = steps.map((step, index) => `${index + 1}. ${step}`).join("\n");
521
+ const surfaceBlock = {
522
+ type: "ui_surface",
523
+ surfaceId,
524
+ surfaceType: "card",
525
+ title: payload.task,
526
+ display: "inline",
527
+ data: {
528
+ title: payload.task,
529
+ ...(payload.purpose ? { subtitle: payload.purpose } : {}),
530
+ body,
531
+ template: WATCH_RETRO_TEMPLATE,
532
+ templateData: payload,
533
+ },
534
+ };
535
+ // Plain-text sibling, the approval-card pattern: providers drop `ui_surface`
536
+ // when serializing history, so without this the model's next turn would have
537
+ // no idea what it just showed the user, and the CLI, search and channel
538
+ // replies would render the session as nothing at all.
539
+ const fallbackBlock = {
540
+ type: "text",
541
+ text: `Here is what I saw: ${payload.task}${body ? `\n\n${body}` : ""}`,
542
+ _surfaceFallback: true,
543
+ };
544
+ await addMessage(
545
+ conversationId,
546
+ "assistant",
547
+ JSON.stringify([surfaceBlock, fallbackBlock]),
548
+ { skipIndexing: true, clientMessageId: surfaceId },
549
+ );
550
+ return surfaceId;
438
551
  }
439
552
 
440
553
  /**
@@ -0,0 +1,195 @@
1
+ import { existsSync, readFileSync, renameSync, writeFileSync } from "node:fs";
2
+ import { join } from "node:path";
3
+ import { Database } from "bun:sqlite";
4
+
5
+ import type { WorkspaceMigration } from "./types.js";
6
+
7
+ /**
8
+ * Repair the renamed undated Fireworks DeepSeek V4 Pro model ID in
9
+ * workspace LLM config.
10
+ *
11
+ * Fireworks serves DeepSeek V4 Pro only under the dated official release
12
+ * ID `accounts/fireworks/models/deepseek-v4-pro-0813`; the undated
13
+ * `accounts/fireworks/models/deepseek-v4-pro` preview has no serverless
14
+ * deployment (the model page states "Serverless: Not supported"), so every
15
+ * serverless call to it fails. Existing configs can still pin the undated
16
+ * ID in `llm.default`, `llm.callSites.*`, and `llm.profiles.*`.
17
+ *
18
+ * Repair those leaves only on an exact stale match, replacing with the
19
+ * dated ID.
20
+ *
21
+ * Provider guard: the stale ID belongs to the `fireworks` provider and also
22
+ * appears in managed profiles stamped `provider: "vellum"` (which route
23
+ * Fireworks-account model IDs through the managed proxy). Under the entries
24
+ * model (migration 145) `provider` can also hold a `provider_connections`
25
+ * entry name whose row kind drives dispatch. A fragment is repaired when
26
+ * its `provider` is `"fireworks"`, `"vellum"`, absent, or an entry name
27
+ * whose row kind is one of those: every fireworks-kind route serves the
28
+ * dated ID and only the dated ID. Any other provider is left untouched: an
29
+ * `openai-compatible` endpoint may legitimately serve a model by the stale
30
+ * name.
31
+ */
32
+ export const repairRenamedFireworksDeepseekProModelIdMigration: WorkspaceMigration =
33
+ {
34
+ id: "151-repair-renamed-fireworks-deepseek-pro-model-id",
35
+ description:
36
+ "Repair renamed Fireworks accounts/fireworks/models/deepseek-v4-pro model ID in workspace LLM config",
37
+ run(workspaceDir: string): void {
38
+ const configPath = join(workspaceDir, "config.json");
39
+ if (!existsSync(configPath)) {
40
+ return;
41
+ }
42
+
43
+ // Read outside the parse catch: a transient filesystem error (EIO,
44
+ // EACCES) must reach the runner so the migration retries, while
45
+ // malformed JSON is a permanent state this migration cannot repair.
46
+ const rawText = readFileSync(configPath, "utf-8");
47
+
48
+ let config: Record<string, unknown>;
49
+ try {
50
+ const raw = JSON.parse(rawText);
51
+ if (!raw || typeof raw !== "object" || Array.isArray(raw)) {
52
+ return;
53
+ }
54
+ config = raw as Record<string, unknown>;
55
+ } catch {
56
+ return;
57
+ }
58
+
59
+ const llm = readObject(config.llm);
60
+ if (llm === null) {
61
+ return;
62
+ }
63
+
64
+ // Entry rows load lazily: only a stale fragment whose provider is
65
+ // neither a repairable vendor nor absent needs them. An unreadable DB
66
+ // then fails the run (retried next boot) rather than checkpointing a
67
+ // pass that skips entry-bound profiles.
68
+ let rows: Map<string, string> | null | undefined;
69
+ const isRepairableProvider = (provider: unknown): boolean => {
70
+ if (provider === undefined) {
71
+ return true;
72
+ }
73
+ if (typeof provider !== "string") {
74
+ return false;
75
+ }
76
+ if (REPAIRABLE_PROVIDERS.has(provider)) {
77
+ return true;
78
+ }
79
+ if (rows === undefined) {
80
+ rows = readConnectionRows(workspaceDir);
81
+ }
82
+ if (rows === null) {
83
+ throw new Error(
84
+ "provider_connections is not readable; retrying the model-ID repair on the next run",
85
+ );
86
+ }
87
+ const kind = rows.get(provider);
88
+ return kind !== undefined && REPAIRABLE_PROVIDERS.has(kind);
89
+ };
90
+
91
+ let changed = false;
92
+
93
+ changed =
94
+ repairFragment(readObject(llm.default), isRepairableProvider) ||
95
+ changed;
96
+
97
+ const callSites = readObject(llm.callSites);
98
+ if (callSites !== null) {
99
+ for (const rawConfig of Object.values(callSites)) {
100
+ changed =
101
+ repairFragment(readObject(rawConfig), isRepairableProvider) ||
102
+ changed;
103
+ }
104
+ }
105
+
106
+ const profiles = readObject(llm.profiles);
107
+ if (profiles !== null) {
108
+ for (const rawProfile of Object.values(profiles)) {
109
+ changed =
110
+ repairFragment(readObject(rawProfile), isRepairableProvider) ||
111
+ changed;
112
+ }
113
+ }
114
+
115
+ if (!changed) {
116
+ return;
117
+ }
118
+
119
+ // Write-then-rename so an interrupted write cannot leave config.json
120
+ // truncated: a torn in-place write would parse as invalid JSON on the
121
+ // retry, which the catch above treats as "nothing to do", letting the
122
+ // runner checkpoint the migration as completed against a corrupt file.
123
+ const tmpPath = `${configPath}.migration-151.tmp`;
124
+ writeFileSync(tmpPath, JSON.stringify(config, null, 2) + "\n");
125
+ renameSync(tmpPath, configPath);
126
+ },
127
+ // The exact-match rewrite is idempotent, so a transient failure (full
128
+ // disk, I/O error) is safe to retry on later startups.
129
+ retryFailedCheckpoint: true,
130
+ down(_workspaceDir: string): void {
131
+ // Forward-only: reintroducing the undated model ID would break
132
+ // Fireworks calls.
133
+ },
134
+ };
135
+
136
+ // ---------------------------------------------------------------------------
137
+ // Helpers: self-contained per workspace migrations AGENTS.md
138
+ // ---------------------------------------------------------------------------
139
+
140
+ const STALE_MODEL_ID = "accounts/fireworks/models/deepseek-v4-pro";
141
+ const REPLACEMENT_MODEL_ID = "accounts/fireworks/models/deepseek-v4-pro-0813";
142
+ const REPAIRABLE_PROVIDERS = new Set(["fireworks", "vellum"]);
143
+
144
+ function repairFragment(
145
+ fragment: Record<string, unknown> | null,
146
+ isRepairableProvider: (provider: unknown) => boolean,
147
+ ): boolean {
148
+ if (fragment === null) {
149
+ return false;
150
+ }
151
+ if (fragment.model !== STALE_MODEL_ID) {
152
+ return false;
153
+ }
154
+ if (!isRepairableProvider(fragment.provider)) {
155
+ return false;
156
+ }
157
+ fragment.model = REPLACEMENT_MODEL_ID;
158
+ return true;
159
+ }
160
+
161
+ /**
162
+ * Connection name -> provider kind, or null when the DB or table is not
163
+ * readable. The caller fails the run on null: entry-name providers must be
164
+ * judged against real rows, never guessed. An absent DB file is a real
165
+ * state (no rows, so every entry name is dangling and stays untouched).
166
+ */
167
+ function readConnectionRows(workspaceDir: string): Map<string, string> | null {
168
+ const dbPath = join(workspaceDir, "data", "db", "assistant.db");
169
+ if (!existsSync(dbPath)) {
170
+ return new Map();
171
+ }
172
+ let db: Database;
173
+ try {
174
+ db = new Database(dbPath);
175
+ } catch {
176
+ return null;
177
+ }
178
+ try {
179
+ const rows = db
180
+ .query(`SELECT name, provider FROM provider_connections`)
181
+ .all() as Array<{ name: string; provider: string }>;
182
+ return new Map(rows.map((r) => [r.name, r.provider]));
183
+ } catch {
184
+ return null;
185
+ } finally {
186
+ db.close();
187
+ }
188
+ }
189
+
190
+ function readObject(value: unknown): Record<string, unknown> | null {
191
+ if (value === null || typeof value !== "object" || Array.isArray(value)) {
192
+ return null;
193
+ }
194
+ return value as Record<string, unknown>;
195
+ }