@vellumai/assistant 0.11.8 → 0.11.9-staging.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (318) hide show
  1. package/ARCHITECTURE.md +2 -2
  2. package/Dockerfile +8 -48
  3. package/docker-entrypoint.sh +5 -1
  4. package/docker-kata-apt-env.sh +3 -0
  5. package/docker-kata-apt-shims.sh +127 -0
  6. package/docker-kata-apt-wrapper.sh +45 -0
  7. package/docker-kata-chroot-exec.sh +35 -0
  8. package/docker-kata-pip.sh +8 -2
  9. package/docs/architecture/memory.md +8 -4
  10. package/docs/guardian-request-flow.md +18 -14
  11. package/docs/trusted-contact-access.md +1 -7
  12. package/node_modules/@vellumai/avatar-catalog/src/catalog.ts +7 -1
  13. package/node_modules/@vellumai/avatar-catalog/src/index.ts +1 -1
  14. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/package.json +4 -1
  15. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/guardian-requests.ts +18 -0
  16. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/index.ts +1 -0
  17. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/platform-credential.ts +41 -0
  18. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/reactions.ts +60 -0
  19. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/package.json +4 -1
  20. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/guardian-requests.ts +18 -0
  21. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/index.ts +1 -0
  22. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/platform-credential.ts +41 -0
  23. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/reactions.ts +60 -0
  24. package/node_modules/@vellumai/gateway-client/src/__tests__/guardian-request-contract.test.ts +0 -18
  25. package/node_modules/@vellumai/gateway-client/src/__tests__/inbound-event-kind.test.ts +110 -5
  26. package/node_modules/@vellumai/gateway-client/src/guardian-request-contract.ts +5 -33
  27. package/node_modules/@vellumai/gateway-client/src/inbound-contract.ts +3 -0
  28. package/node_modules/@vellumai/gateway-client/src/inbound-event-kind.ts +100 -9
  29. package/node_modules/@vellumai/gateway-client/src/index.ts +1 -2
  30. package/node_modules/@vellumai/gateway-client/src/outbound-contract.ts +9 -0
  31. package/node_modules/@vellumai/service-contracts/package.json +4 -1
  32. package/node_modules/@vellumai/service-contracts/src/guardian-requests.ts +18 -0
  33. package/node_modules/@vellumai/service-contracts/src/index.ts +1 -0
  34. package/node_modules/@vellumai/service-contracts/src/platform-credential.ts +41 -0
  35. package/node_modules/@vellumai/service-contracts/src/reactions.ts +60 -0
  36. package/openapi.yaml +193 -4
  37. package/package.json +2 -2
  38. package/src/__tests__/access-request-card-view.test.ts +6 -5
  39. package/src/__tests__/access-request-seed-content-blocks.test.ts +5 -2
  40. package/src/__tests__/agent-loop-mutable-latest-user-message.test.ts +43 -82
  41. package/src/__tests__/always-loaded-tools-guard.test.ts +8 -2
  42. package/src/__tests__/attachment-stored-path-annotation.test.ts +80 -0
  43. package/src/__tests__/attachments-store.test.ts +27 -0
  44. package/src/__tests__/attachments.test.ts +43 -0
  45. package/src/__tests__/canned-reply-release.test.ts +48 -5
  46. package/src/__tests__/channel-delivery-store.test.ts +129 -18
  47. package/src/__tests__/channel-reply-delivery.test.ts +553 -93
  48. package/src/__tests__/channel-retry-sweep.test.ts +199 -0
  49. package/src/__tests__/client-os-metadata-persistence.test.ts +32 -3
  50. package/src/__tests__/conversation-error.test.ts +15 -0
  51. package/src/__tests__/conversation-load-history-repair.test.ts +118 -0
  52. package/src/__tests__/conversation-pairing.test.ts +141 -1
  53. package/src/__tests__/conversation-routes-disk-view.test.ts +28 -1
  54. package/src/__tests__/conversation-routes-slash-commands.test.ts +125 -0
  55. package/src/__tests__/conversation-runtime-assembly.test.ts +79 -0
  56. package/src/__tests__/conversation-store-ephemeral.test.ts +323 -10
  57. package/src/__tests__/conversation-sync-tags.test.ts +2 -37
  58. package/src/__tests__/credential-health-service.test.ts +98 -10
  59. package/src/__tests__/delete-propagation.test.ts +469 -0
  60. package/src/__tests__/dm-persistence.test.ts +16 -0
  61. package/src/__tests__/docker-kata-apt-shims.test.ts +135 -0
  62. package/src/__tests__/forbidden-legacy-symbols.test.ts +12 -0
  63. package/src/__tests__/guardian-card-withdrawal.test.ts +0 -22
  64. package/src/__tests__/guardian-gateway-sim.ts +0 -23
  65. package/src/__tests__/guardian-question-mode.test.ts +108 -0
  66. package/src/__tests__/guardian-reply-router-answer-mode.test.ts +0 -2
  67. package/src/__tests__/guardian-routing-invariants.test.ts +26 -12
  68. package/src/__tests__/host-proxy-interface.test.ts +10 -0
  69. package/src/__tests__/inbound-slack-persistence.test.ts +16 -0
  70. package/src/__tests__/injector-v3-suppression.test.ts +43 -9
  71. package/src/__tests__/list-messages-page-latest.test.ts +127 -0
  72. package/src/__tests__/list-messages-system-card.test.ts +103 -0
  73. package/src/__tests__/media-resolve-image-validation.test.ts +77 -0
  74. package/src/__tests__/notification-decision-fallback.test.ts +199 -77
  75. package/src/__tests__/notification-decision-strategy.test.ts +198 -213
  76. package/src/__tests__/notification-discord-adapter.test.ts +25 -0
  77. package/src/__tests__/notification-slack-adapter.test.ts +133 -0
  78. package/src/__tests__/notification-telegram-adapter.test.ts +36 -0
  79. package/src/__tests__/openai-provider.test.ts +5 -5
  80. package/src/__tests__/outbound-slack-persistence.test.ts +85 -72
  81. package/src/__tests__/persist-user-message-set-processing-failure.test.ts +83 -65
  82. package/src/__tests__/platform-client-verify-credential.test.ts +100 -0
  83. package/src/__tests__/plugin-import-boundary-guard.test.ts +1 -1
  84. package/src/__tests__/process-message-display-content.test.ts +20 -9
  85. package/src/__tests__/processing-acquire-fenced-guard.test.ts +56 -0
  86. package/src/__tests__/provider-meta-persistence.test.ts +67 -0
  87. package/src/__tests__/reaction-persistence.test.ts +22 -198
  88. package/src/__tests__/run-conversation-turn-persistence.test.ts +5 -2
  89. package/src/__tests__/scripted-turn-metadata-persistence.test.ts +16 -0
  90. package/src/__tests__/skill-load-tool.test.ts +27 -0
  91. package/src/__tests__/skills.test.ts +34 -1
  92. package/src/__tests__/strip-memory-injections.test.ts +3 -4
  93. package/src/__tests__/terminal-tools.test.ts +9 -0
  94. package/src/__tests__/thread-backfill.test.ts +5 -3
  95. package/src/__tests__/unified-turn-context-location.test.ts +76 -0
  96. package/src/__tests__/voice-session-bridge.test.ts +18 -0
  97. package/src/__tests__/watch-retro-report-payload.test.ts +185 -0
  98. package/src/__tests__/watch-retro-tool-availability.test.ts +82 -0
  99. package/src/__tests__/workspace-migration-151-repair-renamed-fireworks-deepseek-pro-model-id.test.ts +235 -0
  100. package/src/__tests__/workspace-migration-152-repair-retired-fireworks-minimax-m2p7-model-id.test.ts +233 -0
  101. package/src/agent/attachments.ts +13 -2
  102. package/src/agent/loop.ts +3 -19
  103. package/src/api/README.md +9 -5
  104. package/src/api/index.ts +9 -8
  105. package/src/api/package.json +1 -0
  106. package/src/api/responses/conversation-message.ts +8 -0
  107. package/src/api/responses/home.ts +7 -18
  108. package/src/api/surfaces.ts +114 -7
  109. package/src/approvals/AGENTS.md +1 -1
  110. package/src/channels/__tests__/gateway-guardian-requests.test.ts +1 -23
  111. package/src/channels/__tests__/types.test.ts +22 -1
  112. package/src/channels/gateway-guardian-requests.ts +0 -22
  113. package/src/channels/types.ts +8 -6
  114. package/src/cli/__tests__/catalog-search-help.test.ts +12 -0
  115. package/src/cli/commands/channels/__tests__/channels.test.ts +26 -0
  116. package/src/cli/commands/channels/index.help.ts +15 -1
  117. package/src/cli/commands/channels/index.ts +6 -2
  118. package/src/cli/commands/db/__tests__/status.test.ts +22 -0
  119. package/src/cli/commands/db/index.help.ts +1 -1
  120. package/src/cli/commands/db/status.ts +172 -1
  121. package/src/cli/commands/platform/__tests__/connect.test.ts +29 -0
  122. package/src/cli/commands/platform/connect.ts +14 -5
  123. package/src/cli/commands/plugins.help.ts +6 -5
  124. package/src/cli/lib/__tests__/install-from-github.test.ts +0 -8
  125. package/src/cli/lib/__tests__/install-from-platform.test.ts +72 -0
  126. package/src/cli/lib/__tests__/plugin-catalog-local.test.ts +33 -0
  127. package/src/cli/lib/bundled-marketplace.json +14 -0
  128. package/src/cli/lib/install-from-github.ts +9 -5
  129. package/src/cli/lib/install-from-platform.ts +12 -1
  130. package/src/config/__tests__/assistant-initiated-threads-gate.test.ts +61 -0
  131. package/src/config/assistant-initiated-threads-gate.ts +50 -0
  132. package/src/config/bundled-skills/acp/SKILL.md +10 -3
  133. package/src/config/bundled-skills/schedule/SKILL.md +25 -11
  134. package/src/config/bundled-skills/schedule/references/SCRIPT_MODE_PATTERNS.md +3 -1
  135. package/src/config/call-site-defaults.ts +4 -1
  136. package/src/config/feature-flag-registry.json +21 -4
  137. package/src/config/schemas/memory-v3.ts +4 -3
  138. package/src/context/strip-injections.ts +17 -56
  139. package/src/conversations/__tests__/message-consolidation.test.ts +39 -0
  140. package/src/conversations/message-consolidation.ts +4 -3
  141. package/src/credential-health/credential-health-service.ts +130 -0
  142. package/src/daemon/__tests__/conversation-tool-setup.test.ts +31 -0
  143. package/src/daemon/conversation-agent-loop-handlers.ts +55 -59
  144. package/src/daemon/conversation-error.ts +2 -0
  145. package/src/daemon/conversation-messaging.ts +195 -47
  146. package/src/daemon/conversation-process.ts +20 -5
  147. package/src/daemon/conversation-runtime-assembly.ts +48 -39
  148. package/src/daemon/conversation-store.ts +159 -11
  149. package/src/daemon/conversation-tool-setup.ts +9 -2
  150. package/src/daemon/conversation.ts +290 -19
  151. package/src/daemon/dictation-text-processing.ts +2 -7
  152. package/src/daemon/handlers/config-channels.ts +33 -3
  153. package/src/daemon/handlers/shared.ts +9 -0
  154. package/src/daemon/port-oversized-content.test.ts +117 -0
  155. package/src/daemon/port-oversized-content.ts +110 -0
  156. package/src/daemon/process-message.ts +87 -16
  157. package/src/daemon/reaction-record.test.ts +23 -11
  158. package/src/daemon/reaction-record.ts +6 -9
  159. package/src/documents/document-store.ts +1 -5
  160. package/src/home/conversation-starter-validation.ts +1 -5
  161. package/src/live-voice/__tests__/live-voice-photo.test.ts +910 -50
  162. package/src/live-voice/__tests__/live-voice-sight-frame-inline.test.ts +31 -25
  163. package/src/live-voice/__tests__/live-voice-sight-frame.test.ts +70 -1
  164. package/src/live-voice/live-voice-photo.ts +544 -61
  165. package/src/live-voice/live-voice-session.ts +6 -2
  166. package/src/live-voice/protocol.ts +20 -1
  167. package/src/messaging/provider-message-metadata.ts +40 -1
  168. package/src/messaging/providers/__tests__/transport-dispatch.test.ts +10 -2
  169. package/src/messaging/providers/channel-transport.ts +28 -0
  170. package/src/messaging/providers/discord/send.ts +3 -2
  171. package/src/messaging/providers/slack/api.ts +11 -1
  172. package/src/messaging/providers/slack/message-metadata.test.ts +133 -0
  173. package/src/messaging/providers/slack/message-metadata.ts +141 -1
  174. package/src/messaging/providers/slack/send.test.ts +58 -0
  175. package/src/messaging/providers/slack/send.ts +4 -4
  176. package/src/messaging/providers/slack/transport.ts +28 -2
  177. package/src/messaging/providers/telegram-bot/send.test.ts +159 -0
  178. package/src/messaging/providers/telegram-bot/send.ts +93 -0
  179. package/src/messaging/providers/telegram-bot/transport.ts +71 -0
  180. package/src/messaging/reaction-envelopes.test.ts +73 -0
  181. package/src/messaging/reaction-envelopes.ts +33 -5
  182. package/src/messaging/read-provider-metadata.ts +4 -3
  183. package/src/monitoring/recovery/__tests__/stranded-delivery-events.test.ts +208 -0
  184. package/src/monitoring/recovery/db.ts +35 -0
  185. package/src/monitoring/recovery/orphaned-channel-events.ts +3 -22
  186. package/src/monitoring/recovery/run-recovery.ts +6 -2
  187. package/src/monitoring/recovery/stale-processing.ts +3 -20
  188. package/src/monitoring/recovery/stranded-delivery-events.ts +68 -0
  189. package/src/notifications/AGENTS.md +2 -2
  190. package/src/notifications/README.md +2 -2
  191. package/src/notifications/__tests__/assistant-reply-producer.test.ts +11 -9
  192. package/src/notifications/__tests__/broadcaster.test.ts +35 -4
  193. package/src/notifications/__tests__/notification-utils.test.ts +31 -0
  194. package/src/notifications/access-request-copy.ts +126 -114
  195. package/src/notifications/adapters/discord.ts +9 -5
  196. package/src/notifications/adapters/shared.ts +17 -4
  197. package/src/notifications/adapters/slack.ts +14 -4
  198. package/src/notifications/adapters/telegram.ts +8 -4
  199. package/src/notifications/approval-card-data.ts +5 -2
  200. package/src/notifications/assistant-reply-producer.ts +7 -7
  201. package/src/notifications/broadcaster.ts +57 -25
  202. package/src/notifications/conversation-pairing.ts +68 -2
  203. package/src/notifications/copy-composer.ts +13 -32
  204. package/src/notifications/decision-engine.ts +56 -237
  205. package/src/notifications/guardian-delivery-recorder.ts +5 -9
  206. package/src/notifications/guardian-feed-projection.ts +0 -15
  207. package/src/notifications/guardian-question-mode.ts +147 -6
  208. package/src/notifications/notification-utils.ts +117 -1
  209. package/src/permissions/confirmation-guardian-request.ts +2 -2
  210. package/src/permissions/prompter.ts +1 -1
  211. package/src/persistence/conversation-crud.ts +93 -14
  212. package/src/persistence/conversation-queries.ts +144 -17
  213. package/src/persistence/conversation-types.ts +39 -5
  214. package/src/persistence/delivery-crud.ts +224 -28
  215. package/src/persistence/delivery-status.ts +21 -2
  216. package/src/persistence/migrations/374-channel-inbound-message-id-index.ts +26 -0
  217. package/src/persistence/migrations/375-create-channel-outbound-posts.ts +52 -0
  218. package/src/persistence/migrations/__tests__/375-create-channel-outbound-posts.test.ts +84 -0
  219. package/src/persistence/schema/conversations.ts +71 -2
  220. package/src/persistence/schema-contract.test.ts +78 -0
  221. package/src/persistence/schema-contract.ts +96 -0
  222. package/src/persistence/steps.ts +4 -0
  223. package/src/platform/client.ts +55 -0
  224. package/src/plugins/__tests__/mcp-servers.test.ts +46 -0
  225. package/src/plugins/defaults/injector-order.ts +2 -1
  226. package/src/plugins/defaults/memory/__tests__/memory-retrospective-accounting.test.ts +2 -2
  227. package/src/plugins/defaults/memory/context-search/sources/conversations.ts +13 -15
  228. package/src/plugins/defaults/memory/hooks/user-prompt-submit.ts +10 -0
  229. package/src/plugins/defaults/memory/injectors.ts +1 -1
  230. package/src/plugins/defaults/memory/memory-marker.ts +17 -7
  231. package/src/plugins/defaults/memory/substrate/__tests__/consolidation-prompt-flag-gating-guard.test.ts +1 -5
  232. package/src/plugins/defaults/memory/v3/__tests__/carry-integration.test.ts +54 -41
  233. package/src/plugins/defaults/memory/v3/__tests__/injection.test.ts +3 -3
  234. package/src/plugins/defaults/memory/v3/injector.ts +14 -15
  235. package/src/plugins/defaults/memory/v3/types.ts +15 -4
  236. package/src/plugins/defaults/turn-context/injectors.ts +3 -0
  237. package/src/plugins/defaults/turn-context/unified-turn-context.ts +24 -0
  238. package/src/plugins/mcp-servers.ts +14 -2
  239. package/src/providers/__tests__/dispatch-connection-routing.test.ts +50 -0
  240. package/src/providers/__tests__/registry-native-web-search.test.ts +49 -2
  241. package/src/providers/anthropic/client.ts +21 -24
  242. package/src/providers/call-site-routing.ts +5 -4
  243. package/src/providers/connection-resolution.ts +29 -1
  244. package/src/providers/content-block-size.test.ts +82 -0
  245. package/src/providers/content-block-size.ts +99 -0
  246. package/src/providers/file-block-text.test.ts +64 -0
  247. package/src/providers/file-block-text.ts +28 -0
  248. package/src/providers/gemini/client.ts +12 -7
  249. package/src/providers/inference/auth.ts +6 -6
  250. package/src/providers/media-resolve.ts +14 -0
  251. package/src/providers/model-catalog.ts +31 -34
  252. package/src/providers/openai/chat-completions-provider.ts +9 -25
  253. package/src/providers/openai/responses-provider.ts +11 -18
  254. package/src/providers/registry.ts +10 -7
  255. package/src/providers/routing-identity.ts +2 -1
  256. package/src/providers/types.ts +1 -4
  257. package/src/providers/vellum-model-routing.ts +3 -2
  258. package/src/runtime/AGENTS.md +40 -15
  259. package/src/runtime/approval-message-composer.ts +1 -6
  260. package/src/runtime/assistant-event-hub.ts +2 -39
  261. package/src/runtime/channel-approval-types.ts +1 -5
  262. package/src/runtime/channel-reply-delivery.ts +182 -115
  263. package/src/runtime/{slack-reply-session.test.ts → channel-reply-session.test.ts} +294 -156
  264. package/src/runtime/{slack-reply-session.ts → channel-reply-session.ts} +107 -86
  265. package/src/runtime/channel-retry-sweep.ts +102 -1
  266. package/src/runtime/finalize-event-delivery.ts +4 -4
  267. package/src/runtime/guardian-action-message-composer.ts +1 -4
  268. package/src/runtime/guardian-reply-router.ts +4 -75
  269. package/src/runtime/http-router.ts +1 -5
  270. package/src/runtime/question-request-guardian-bridge.ts +2 -3
  271. package/src/runtime/routes/__tests__/conversation-list-assistant-section.test.ts +359 -0
  272. package/src/runtime/routes/__tests__/dictation-command-mode.test.ts +120 -0
  273. package/src/runtime/routes/__tests__/sight-frame-routes.test.ts +487 -0
  274. package/src/runtime/routes/acp-routes.ts +1 -1
  275. package/src/runtime/routes/canned-reply-release.ts +19 -9
  276. package/src/runtime/routes/channel-route-shared.ts +0 -39
  277. package/src/runtime/routes/channel-verification-routes.ts +4 -1
  278. package/src/runtime/routes/conversation-list-routes.ts +44 -3
  279. package/src/runtime/routes/conversation-management-routes.ts +2 -1
  280. package/src/runtime/routes/conversation-routes.ts +138 -59
  281. package/src/runtime/routes/diagnostics-routes.ts +111 -60
  282. package/src/runtime/routes/guardian-action-routes.ts +10 -14
  283. package/src/runtime/routes/guardian-approval-interception.ts +0 -5
  284. package/src/runtime/routes/inbound-message-handler.ts +170 -48
  285. package/src/runtime/routes/inbound-stages/acl-enforcement.ts +4 -3
  286. package/src/runtime/routes/inbound-stages/background-dispatch.test.ts +169 -8
  287. package/src/runtime/routes/inbound-stages/background-dispatch.ts +82 -14
  288. package/src/runtime/routes/inbound-stages/guardian-reply-intercept.ts +23 -48
  289. package/src/runtime/routes/inbound-stages/reaction-intercept.test.ts +454 -94
  290. package/src/runtime/routes/inbound-stages/reaction-intercept.ts +309 -102
  291. package/src/runtime/routes/index.ts +2 -0
  292. package/src/runtime/routes/platform-routes.ts +99 -10
  293. package/src/runtime/routes/sight-frame-routes.ts +136 -0
  294. package/src/runtime/sync/resource-sync-events.ts +14 -37
  295. package/src/runtime/sync/sync-publisher.test.ts +5 -4
  296. package/src/tools/__tests__/tool-input-schemas.test.ts +2 -0
  297. package/src/tools/client-os.ts +11 -1
  298. package/src/tools/host-filesystem/edit.ts +2 -2
  299. package/src/tools/host-filesystem/read.ts +2 -2
  300. package/src/tools/host-filesystem/transfer.ts +2 -2
  301. package/src/tools/host-filesystem/write.ts +2 -2
  302. package/src/tools/host-terminal/host-shell.ts +2 -2
  303. package/src/tools/skills/load.ts +1 -1
  304. package/src/tools/terminal/safe-env.ts +4 -0
  305. package/src/tools/tool-input-schemas.ts +2 -0
  306. package/src/tools/tool-manifest.ts +2 -0
  307. package/src/tools/ui-surface/definitions.ts +4 -3
  308. package/src/tools/watch/watch-retro-report.ts +205 -0
  309. package/src/watch/__tests__/watch-retro.test.ts +268 -40
  310. package/src/watch/watch-retro.ts +190 -77
  311. package/src/workspace/migrations/151-repair-renamed-fireworks-deepseek-pro-model-id.ts +195 -0
  312. package/src/workspace/migrations/152-repair-retired-fireworks-minimax-m2p7-model-id.ts +198 -0
  313. package/src/workspace/migrations/__tests__/150-stt-flux-provider-to-model-family.test.ts +0 -10
  314. package/src/workspace/migrations/registry.ts +4 -0
  315. package/docker-kata-pip-chroot.sh +0 -22
  316. package/src/__tests__/slack-reaction-approvals.test.ts +0 -97
  317. package/src/__tests__/slack-reaction-guardian-approval.test.ts +0 -307
  318. package/src/api/events/conversation-list-invalidated.ts +0 -38
@@ -0,0 +1,205 @@
1
+ import { z } from "zod";
2
+
3
+ import { RiskLevel } from "../../permissions/types.js";
4
+ import {
5
+ invalidToolInputResult,
6
+ toToolInputSchema,
7
+ } from "../shared/zod-tool-schema.js";
8
+ import type {
9
+ ToolContext,
10
+ ToolDefinition,
11
+ ToolExecutionResult,
12
+ } from "../types.js";
13
+
14
+ /**
15
+ * How a watch retrospective hands its report to the daemon.
16
+ *
17
+ * **The card cannot come from `ui_show`.** A retrospective runs as a
18
+ * `clientless` wake, which pins the turn non-interactive, and
19
+ * `conversation-tool-setup` gates the whole `ui_surface` tool family on a
20
+ * client being present (`return channelCapabilities?.supportsDynamicUi ??
21
+ * !hasNoClient`). So `ui_show` is not merely denied in this turn, it is absent
22
+ * from the tool set, and a retrospective told to call it can only report that
23
+ * it cannot. This tool is an ordinary one and passes that gate untouched.
24
+ *
25
+ * **It records; it does not render.** The executor validates the payload and
26
+ * returns, and nothing here writes to the conversation. What makes the card is
27
+ * `watch-retro.ts` reading this call back out of the turn's own history once
28
+ * the turn has finished, and appending the `ui_surface` block itself. Two
29
+ * reasons for the split. A surface appended from inside the turn can land
30
+ * between a persisted `tool_use` and its `tool_result`, which is the ordering
31
+ * hazard the memory retrospective's skill card defers around
32
+ * (`memory-retrospective-skill-card.ts`); waiting until the turn is over avoids
33
+ * it rather than detecting it. And the report is then held in the one place
34
+ * that survives a crash between the call and the append: the transcript.
35
+ *
36
+ * **Validation is the model-facing half.** The payload is the card's own
37
+ * shape, so one the renderer could not draw is refused here, while the model
38
+ * still has a turn left to correct it. The daemon parses it again before
39
+ * appending, since a tool result is not a promise about what was persisted.
40
+ *
41
+ * **Questions are validated by shape, and the shape is not advertised.** The
42
+ * surface schema is tolerant and strips what it does not recognize, so a
43
+ * question sent as `question`/`value` instead of `prompt`/`label` parses clean
44
+ * and draws a page with no text on it. Checking the shape here turns that into
45
+ * a rejection naming the field, which the model has a turn left to act on.
46
+ * Advertising it too would be the surer teacher, but the always-loaded payload
47
+ * is within a few hundred bytes of the budget
48
+ * (`browser-skill-baseline-tool-payload`) and the question shape is around 450
49
+ * of them, spent on every request by every caller for a tool one flow uses.
50
+ * The divergence is safe in a way the general rule in `zod-tool-schema.ts`
51
+ * warns it is not: a model reaching this tool has been sent here by
52
+ * `RETRO_INSTRUCTIONS`, which names every field, so there is no caller that
53
+ * sees the terse schema without the contract.
54
+ */
55
+
56
+ /** One alternative on a `pick` or a `gate`. */
57
+ const questionOptionSchema = z.object({
58
+ id: z.string().min(1),
59
+ label: z.string().min(1),
60
+ note: z.string().optional().catch(undefined),
61
+ });
62
+
63
+ /**
64
+ * One question, held to what the card can actually page through.
65
+ *
66
+ * A `pick` or a `gate` with fewer than two options is one button, which is not
67
+ * a question, and the renderer drops it. More than four is the questionnaire
68
+ * the paging exists to avoid, on a page that then scrolls. Rejecting either
69
+ * here means the model hears about the gap it could not name instead of the
70
+ * user reading a page that is missing or too long to tap through.
71
+ */
72
+ const questionSchema = z
73
+ .object({
74
+ id: z.string().min(1),
75
+ kind: z.enum(["fill", "pick", "gate"]),
76
+ prompt: z.string().min(1),
77
+ eyebrow: z.string().optional().catch(undefined),
78
+ suggestion: z.string().optional().catch(undefined),
79
+ // No `.catch`, for the same reason `questions` has none, and for one more:
80
+ // a swallowed option array reads as an empty one, so a mistyped `label`
81
+ // would come back as the count complaint below. That answer sends the
82
+ // model to add options it already sent.
83
+ options: z.array(questionOptionSchema).optional(),
84
+ })
85
+ .refine(
86
+ (question) => {
87
+ if (question.kind === "fill") {
88
+ return true;
89
+ }
90
+ const count = question.options?.length ?? 0;
91
+ return count >= 2 && count <= 4;
92
+ },
93
+ {
94
+ error: 'a "pick" or "gate" needs two to four options',
95
+ path: ["options"],
96
+ },
97
+ );
98
+
99
+ export const watchRetroReportInputSchema = z.looseObject({
100
+ task: z.string(),
101
+ purpose: z.string().optional().catch(undefined),
102
+ steps: z.array(z.string()).optional().catch(undefined),
103
+ eyebrow: z.string().optional().catch(undefined),
104
+ coverage: z.string().optional().catch(undefined),
105
+ // No `.catch` here, unlike the decoration around it. A swallowed questions
106
+ // array is the failure this schema exists to report: the card would draw
107
+ // without the pages the model meant to ask on, and nothing would say so.
108
+ questions: z.array(questionSchema).optional(),
109
+ });
110
+
111
+ /**
112
+ * The same payload with the question shape left off, which is what the model
113
+ * is shown. Derived from the validating schema rather than written out beside
114
+ * it, so the fields around `questions` cannot drift from what is enforced.
115
+ * See the note above on why this one field is not advertised.
116
+ */
117
+ const watchRetroReportAdvertisedSchema = watchRetroReportInputSchema.extend({
118
+ questions: z.array(z.unknown()).optional(),
119
+ });
120
+
121
+ /**
122
+ * The first id used by more than one question, or null when they are distinct.
123
+ *
124
+ * Checked here rather than in the schema because the renderer's response to a
125
+ * repeat is to drop the later question, not to fail: the id is the key an
126
+ * answer is held under, so two questions sharing one would submit whichever
127
+ * answer landed last for both.
128
+ */
129
+ function firstDuplicateQuestionId(
130
+ questions: readonly { id: string }[] | undefined,
131
+ ): string | null {
132
+ const seen = new Set<string>();
133
+ for (const question of questions ?? []) {
134
+ if (seen.has(question.id)) {
135
+ return question.id;
136
+ }
137
+ seen.add(question.id);
138
+ }
139
+ return null;
140
+ }
141
+
142
+ export async function executeWatchRetroReport(
143
+ input: Record<string, unknown>,
144
+ _context: ToolContext,
145
+ ): Promise<ToolExecutionResult> {
146
+ const parsed = watchRetroReportInputSchema.safeParse(input);
147
+ if (!parsed.success) {
148
+ return invalidToolInputResult("watch_retro_report", parsed.error);
149
+ }
150
+ const { task, steps, questions } = parsed.data;
151
+ if (!task || task.trim().length === 0) {
152
+ return {
153
+ content: '"task" is required: name the task the session recorded.',
154
+ isError: true,
155
+ };
156
+ }
157
+ // A record with no steps is not a record. The questions are optional and the
158
+ // rest is decoration, but a card whose first page is a bare title tells the
159
+ // user nothing about the session they just finished.
160
+ if (!steps || steps.length === 0) {
161
+ return {
162
+ content:
163
+ '"steps" is required: list what you saw, in order, as short imperative fragments.',
164
+ isError: true,
165
+ };
166
+ }
167
+ // Last, because the record is the part a card cannot do without and a report
168
+ // missing it should hear about that first.
169
+ const duplicateId = firstDuplicateQuestionId(questions);
170
+ if (duplicateId !== null) {
171
+ return {
172
+ content: `Two questions share the id "${duplicateId}". An id is the handle an answer comes back under, so the card keeps the first question and drops the rest. Give each one its own.`,
173
+ isError: true,
174
+ };
175
+ }
176
+ return {
177
+ content: JSON.stringify({ recorded: true, steps: steps.length }),
178
+ isError: false,
179
+ };
180
+ }
181
+
182
+ export const watchRetroReportTool = {
183
+ name: "watch_retro_report",
184
+ // Deliberately terse, field descriptions included: this tool is always
185
+ // registered, so every byte here is spent on every request. The contract is
186
+ // carried by the retrospective prompt, its only caller, and enforced by the
187
+ // validating schema above, which is stricter than what this advertises.
188
+ // `browser-skill-baseline-tool-payload` holds the total to a budget.
189
+ description:
190
+ "Report what a Watch (teach mode) session recorded. Only from a watch retrospective, which gives the payload shape.",
191
+ category: "ui-surface",
192
+ executionTarget: "sandbox",
193
+ defaultRiskLevel: RiskLevel.Low,
194
+
195
+ input_schema: toToolInputSchema(watchRetroReportAdvertisedSchema, {
196
+ advertiseRequired: ["task", "steps"],
197
+ }),
198
+
199
+ async execute(
200
+ input: Record<string, unknown>,
201
+ context: ToolContext,
202
+ ): Promise<ToolExecutionResult> {
203
+ return executeWatchRetroReport(input, context);
204
+ },
205
+ } satisfies ToolDefinition;
@@ -5,9 +5,12 @@ import {
5
5
  addMessage,
6
6
  createConversation,
7
7
  getConversation,
8
+ getMessages,
8
9
  PROVIDER_ERROR_MESSAGE_KIND,
9
10
  } from "../../persistence/conversation-crud.js";
10
11
  import { initializeDb } from "../../persistence/db-init.js";
12
+ import { toToolInputSchema } from "../../tools/shared/zod-tool-schema.js";
13
+ import { watchRetroReportInputSchema } from "../../tools/watch/watch-retro-report.js";
11
14
  import {
12
15
  buildRetroWakeOptions,
13
16
  buildWatchRetroPrompt,
@@ -88,8 +91,12 @@ function recordObservation(axTree: string): {
88
91
  * tool before stopping looks like from the conversation's side.
89
92
  */
90
93
  function recordingDispatch(
91
- reply: string | null = "Here is what I understood.",
94
+ reply: string | null = null,
92
95
  metadata?: Record<string, unknown>,
96
+ report: Record<string, unknown> | null = {
97
+ task: "Filing the receipt",
98
+ steps: ["Open the inbox", "Save the attachment"],
99
+ },
93
100
  ) {
94
101
  const calls: { conversationId: string; prompt: string }[] = [];
95
102
  const dispatch = async (
@@ -97,8 +104,20 @@ function recordingDispatch(
97
104
  prompt: string,
98
105
  ): Promise<WatchRetroDispatchResult> => {
99
106
  calls.push({ conversationId, prompt });
107
+ const blocks: Record<string, unknown>[] = [];
100
108
  if (reply !== null) {
101
- await addMessage(conversationId, "assistant", reply, {
109
+ blocks.push({ type: "text", text: reply });
110
+ }
111
+ if (report !== null) {
112
+ blocks.push({
113
+ type: "tool_use",
114
+ id: `toolu_${randomUUID()}`,
115
+ name: "watch_retro_report",
116
+ input: report,
117
+ });
118
+ }
119
+ if (blocks.length > 0) {
120
+ await addMessage(conversationId, "assistant", JSON.stringify(blocks), {
102
121
  skipIndexing: true,
103
122
  ...(metadata ? { metadata } : {}),
104
123
  });
@@ -108,6 +127,38 @@ function recordingDispatch(
108
127
  return { calls, dispatch };
109
128
  }
110
129
 
130
+ /** A turn that ran and left nothing behind at all. */
131
+ function emptyDispatch() {
132
+ return recordingDispatch(null, undefined, null);
133
+ }
134
+
135
+ /** A turn that ran and wrote prose but never called the report tool. */
136
+ function proseOnlyDispatch(reply = "Here is what I understood.") {
137
+ return recordingDispatch(reply, undefined, null);
138
+ }
139
+
140
+ /**
141
+ * The required property names of the object schema reached by walking `path`,
142
+ * descending through array `items` wherever the path crosses one.
143
+ */
144
+ function requiredFieldsUnder(
145
+ schema: Record<string, unknown>,
146
+ path: readonly string[],
147
+ ): string[] {
148
+ let node: Record<string, unknown> | undefined = schema;
149
+ for (const segment of path) {
150
+ const properties = node?.properties as
151
+ | Record<string, Record<string, unknown>>
152
+ | undefined;
153
+ node = properties?.[segment];
154
+ while (node?.items) {
155
+ node = node.items as Record<string, unknown>;
156
+ }
157
+ }
158
+ const required = node?.required;
159
+ return Array.isArray(required) ? (required as string[]) : [];
160
+ }
161
+
111
162
  describe("watch retrospective", () => {
112
163
  test("dispatches exactly one turn carrying the session's timeline", async () => {
113
164
  const summary = recordSession([
@@ -151,7 +202,7 @@ describe("watch retrospective", () => {
151
202
  // that loads `skill-management` and then stops or errors looks invoked and
152
203
  // has said nothing.
153
204
  const result = await runWatchRetro(summary, {
154
- dispatch: recordingDispatch(null).dispatch,
205
+ dispatch: emptyDispatch().dispatch,
155
206
  });
156
207
 
157
208
  expect(result).toEqual({ status: "failed", reason: "no_report" });
@@ -164,9 +215,11 @@ describe("watch retrospective", () => {
164
215
  // A failed LLM call persists an assistant row and returns normally, so the
165
216
  // thread has assistant text in it and still holds no account of anything.
166
217
  const result = await runWatchRetro(summary, {
167
- dispatch: recordingDispatch("The model call failed.", {
168
- messageKind: PROVIDER_ERROR_MESSAGE_KIND,
169
- }).dispatch,
218
+ dispatch: recordingDispatch(
219
+ "The model call failed.",
220
+ { messageKind: PROVIDER_ERROR_MESSAGE_KIND },
221
+ null,
222
+ ).dispatch,
170
223
  });
171
224
 
172
225
  expect(result).toEqual({ status: "failed", reason: "no_report" });
@@ -186,7 +239,7 @@ describe("watch retrospective", () => {
186
239
  );
187
240
 
188
241
  const result = await runWatchRetro(summary, {
189
- dispatch: recordingDispatch(null).dispatch,
242
+ dispatch: emptyDispatch().dispatch,
190
243
  });
191
244
 
192
245
  expect(result).toEqual({ status: "failed", reason: "no_report" });
@@ -197,7 +250,7 @@ describe("watch retrospective", () => {
197
250
  const summary = recordSession(["filing the receipt"]);
198
251
 
199
252
  const result = await runWatchRetro(summary, {
200
- dispatch: recordingDispatch(" \n ").dispatch,
253
+ dispatch: proseOnlyDispatch(" \n ").dispatch,
201
254
  });
202
255
 
203
256
  expect(result).toEqual({ status: "failed", reason: "no_report" });
@@ -225,52 +278,95 @@ describe("watch retrospective", () => {
225
278
 
226
279
  const { prompt } = calls[0]!;
227
280
  expect(prompt).toContain("skill-management");
228
- // The ask leads, and the record follows it.
229
- expect(prompt).toContain("What I need from you");
230
- expect(prompt).toContain("What I saw");
231
- expect(prompt.indexOf("What I need from you")).toBeLessThan(
232
- prompt.indexOf("What I saw"),
281
+ // One card, and the record is its first page. Order carries no priority
282
+ // once each question owns a page, so the record leads without costing the
283
+ // questions anything.
284
+ expect(prompt).toContain("`watch_retro_report`");
285
+ expect(prompt).toContain("`steps`");
286
+ expect(prompt.indexOf("`steps`")).toBeLessThan(
287
+ prompt.indexOf("`questions`"),
233
288
  );
234
289
  // The one field the recording cannot supply is always asked for.
235
- expect(prompt).toContain("what they would say to start this task");
236
- // And the report is not turned back into a questionnaire about itself.
237
290
  expect(prompt).toContain(
238
- "do not ask them to confirm something the recording already showed you.",
291
+ "what they would say to start this task, in their own words",
292
+ );
293
+ // And the card is not turned back into a questionnaire about itself.
294
+ expect(prompt).toContain(
295
+ "Do not ask the user to confirm something the recording already showed you.",
239
296
  );
240
- expect(prompt).toContain("No preamble");
241
- // A destructive step is confirmed however plainly it was recorded. The
297
+ // A destructive step is asked about however plainly it was recorded. The
242
298
  // recording establishes what someone did once and nothing about whether
243
- // they want it repeated unattended, so it is the one exception to the
244
- // rule above.
299
+ // they want it repeated unattended, so it is the one exception to the rule
300
+ // above.
301
+ expect(prompt).toContain("asked however plainly the step was seen");
302
+ });
303
+
304
+ test("caps the questions and keeps every one of them answerable in a tap", async () => {
305
+ const summary = recordSession(["renaming the export"]);
306
+ const { calls, dispatch } = recordingDispatch();
307
+
308
+ await runWatchRetro(summary, { dispatch });
309
+
310
+ const { prompt } = calls[0]!;
311
+ expect(prompt).toContain("at most three");
312
+ // No question may be a yes/no whose "no" carries nothing back: a binary
313
+ // that packs an inferred rule into it spends the question and returns
314
+ // nothing, leaving the user owed a follow-up they cannot see.
315
+ expect(prompt).toContain('Never ask a yes/no whose "no" tells you nothing');
245
316
  expect(prompt).toContain(
246
- "Always confirm any destructive or irreversible step, even one the recording showed plainly",
317
+ "If you cannot name the alternatives, you do not understand the gap well enough to ask about it",
247
318
  );
248
319
  });
249
320
 
250
- test("puts the report last in the turn, after the skill it hands off to", async () => {
321
+ test("makes every skip land on a safe answer", async () => {
322
+ const summary = recordSession(["archiving the thread"]);
323
+ const { calls, dispatch } = recordingDispatch();
324
+
325
+ await runWatchRetro(summary, { dispatch });
326
+
327
+ const { prompt } = calls[0]!;
328
+ // Every page is skippable, so every page's default is what the model
329
+ // actually gets from a user who taps through. A `fill` keeps its
330
+ // suggestion; a `pick` keeps the reading the recording supports.
331
+ expect(prompt).toContain(
332
+ "Put your best guess in `suggestion` so skipping keeps a working phrase",
333
+ );
334
+ expect(prompt).toContain(
335
+ "The first option is the default and must be the reading the recording supports",
336
+ );
337
+ // The gate is the one place the default is deliberately not the guess.
338
+ // Watching someone do a destructive thing once says nothing about whether
339
+ // they want it repeated with nobody looking, so a skipped gate must land
340
+ // on the cautious answer rather than on what was recorded.
341
+ expect(prompt).toContain(
342
+ 'The first option must be the cautious one ("Ask me first"), because a skipped question takes it.',
343
+ );
344
+ });
345
+
346
+ test("puts the card last in the turn, after the skill it hands off to", async () => {
251
347
  const summary = recordSession(["renaming the export"]);
252
348
  const { calls, dispatch } = recordingDispatch();
253
349
 
254
350
  await runWatchRetro(summary, { dispatch });
255
351
 
256
352
  const { prompt } = calls[0]!;
257
- // The report is what the user is shown when a session ends, so it has to
258
- // be the turn's last prose. A client renders the final text block as the
259
- // response and folds everything before it into collapsed intermediate
260
- // work, so a `skill_load` issued after the report demotes the report to
261
- // an "Earlier activity" row the user has to unfold to read.
353
+ // The card is what the user is shown when a session ends, so nothing may
354
+ // follow it. Prose written after it reads as a sign-off nobody asked for,
355
+ // and a client folds everything before a turn's last text block into
356
+ // collapsed intermediate work.
262
357
  expect(prompt).toContain("Load the `skill-management` skill first");
263
358
  expect(
264
359
  prompt.indexOf("Load the `skill-management` skill first"),
265
- ).toBeLessThan(prompt.indexOf("What I need from you"));
266
- expect(prompt).toContain("That report is the last thing you do this turn");
267
- // A sign-off is another text block after the report, which puts the report
268
- // back on the wrong side of the same rule, so it is refused by name.
360
+ ).toBeLessThan(prompt.indexOf("`watch_retro_report`"));
361
+ expect(prompt).toContain("That call is the last thing you do this turn");
269
362
  expect(prompt).toContain("no sign-off");
270
363
  expect(prompt).toContain("no further tool call");
364
+ // One card, not a card per question: a second `ui_show` would replace the
365
+ // paging with a stack of surfaces, which is the shape this replaced.
366
+ expect(prompt).toContain("exactly one `watch_retro_report` call");
271
367
  });
272
368
 
273
- test("writes the report even when the skill it hands off to will not load", async () => {
369
+ test("shows the card even when the skill it hands off to will not load", async () => {
274
370
  const summary = recordSession(["filing the receipt"]);
275
371
  const { calls, dispatch } = recordingDispatch();
276
372
 
@@ -279,14 +375,11 @@ describe("watch retrospective", () => {
279
375
  const { prompt } = calls[0]!;
280
376
  // `skill-management` is a selector, and a managed or workspace skill of the
281
377
  // same id replaces the bundled one in the catalog. Putting the load ahead
282
- // of the report is what lets a shadow, or the refusal a clientless wake
378
+ // of the card is what lets a shadow, or the refusal a clientless wake
283
379
  // gives an inline-command load, land before the user has been told
284
380
  // anything. Neither may become the retro: the session is recorded and the
285
381
  // account of it is what the user is owed.
286
- expect(prompt).toContain(
287
- "Write the report below whether or not that load succeeds",
288
- );
289
- expect(prompt).toContain("do not report on the load and do not retry it");
382
+ expect(prompt).toContain("Report whether or not the skill loaded");
290
383
  // The failure is still named, so a missing handoff does not read as a
291
384
  // report that simply chose not to ask about authoring.
292
385
  expect(prompt).toContain("could not open the skill-authoring flow");
@@ -299,14 +392,116 @@ describe("watch retrospective", () => {
299
392
  await runWatchRetro(summary, { dispatch });
300
393
 
301
394
  const { prompt } = calls[0]!;
302
- expect(prompt).toContain(
303
- "Do not author or scaffold a skill until the four points that step names are settled",
304
- );
395
+ expect(prompt).toContain("Do not author or scaffold a skill yet");
305
396
  // The retro delegates authoring to the skill-management flow rather than
306
397
  // naming the tool that writes a skill, so nothing here can reach it.
307
398
  expect(prompt).not.toContain("scaffold_managed_skill");
308
399
  });
309
400
 
401
+ test("draws the card from the turn's report call", async () => {
402
+ const summary = recordSession(["filing the receipt"]);
403
+ const { dispatch } = recordingDispatch();
404
+
405
+ const result = await runWatchRetro(summary, { dispatch });
406
+
407
+ expect(result).toEqual({
408
+ status: "dispatched",
409
+ conversationId: summary.conversationId,
410
+ });
411
+ expect(getConversation(summary.conversationId)!.surfacedAt).not.toBeNull();
412
+
413
+ // The card is appended by the daemon, not by the turn: nothing the model
414
+ // can call renders a surface in a clientless wake.
415
+ const appended = getMessages(summary.conversationId).at(-1)!;
416
+ const surface = appended.content.find(
417
+ (block) => block.type === "ui_surface",
418
+ ) as { surfaceType: string; data: Record<string, unknown> } | undefined;
419
+ expect(surface?.surfaceType).toBe("card");
420
+ expect(surface?.data.template).toBe("watch_retro");
421
+ expect((surface?.data.templateData as { task: string }).task).toBe(
422
+ "Filing the receipt",
423
+ );
424
+
425
+ // `title`, `subtitle` and `body` are the whole report for a renderer too
426
+ // old to know the template, so they are derived here rather than left to
427
+ // the model to remember.
428
+ expect(surface?.data.title).toBe("Filing the receipt");
429
+ expect(surface?.data.body).toContain("1. Open the inbox");
430
+
431
+ // Providers drop `ui_surface` when serializing history, so without the
432
+ // fallback sibling the model's next turn would not know what it showed.
433
+ const fallback = appended.content.find(
434
+ (block) =>
435
+ block.type === "text" &&
436
+ (block as { _surfaceFallback?: boolean })._surfaceFallback === true,
437
+ );
438
+ expect(fallback).toBeDefined();
439
+ });
440
+
441
+ test("a turn that wrote prose but never reported is not a report", async () => {
442
+ const summary = recordSession(["filing the receipt"]);
443
+
444
+ // Prose is no longer the report, so a turn that narrates the session and
445
+ // never calls the tool has produced nothing the user can be shown. Left
446
+ // counting, it would surface a thread with an account in it that no card
447
+ // backs and no answer can be given to.
448
+ const result = await runWatchRetro(summary, {
449
+ dispatch: proseOnlyDispatch("Here is what I saw. You filed a receipt.")
450
+ .dispatch,
451
+ });
452
+
453
+ expect(result).toEqual({ status: "failed", reason: "no_report" });
454
+ expect(getConversation(summary.conversationId)!.surfacedAt).toBeNull();
455
+ });
456
+
457
+ test("a report call with no task is not a report", async () => {
458
+ const summary = recordSession(["filing the receipt"]);
459
+ const { dispatch } = recordingDispatch(null, undefined, {
460
+ task: " ",
461
+ steps: ["Open the inbox"],
462
+ });
463
+
464
+ // The payload schema is tolerant by design, so an empty task parses. A
465
+ // card whose first page has no title is not an account of anything, and
466
+ // the append is the report test, so it has to reject here.
467
+ const result = await runWatchRetro(summary, { dispatch });
468
+ expect(result).toEqual({ status: "failed", reason: "no_report" });
469
+ });
470
+
471
+ test("the newest report call wins", async () => {
472
+ const summary = recordSession(["filing the receipt"]);
473
+ const conversationId = summary.conversationId;
474
+ const dispatch = async (): Promise<WatchRetroDispatchResult> => {
475
+ // A model that corrects itself calls again rather than editing.
476
+ for (const task of ["First reading", "Corrected reading"]) {
477
+ await addMessage(
478
+ conversationId,
479
+ "assistant",
480
+ JSON.stringify([
481
+ {
482
+ type: "tool_use",
483
+ id: `toolu_${randomUUID()}`,
484
+ name: "watch_retro_report",
485
+ input: { task, steps: ["Open the inbox"] },
486
+ },
487
+ ]),
488
+ { skipIndexing: true },
489
+ );
490
+ }
491
+ return { invoked: true };
492
+ };
493
+
494
+ await runWatchRetro(summary, { dispatch });
495
+
496
+ const appended = getMessages(conversationId).at(-1)!;
497
+ const surface = appended.content.find(
498
+ (block) => block.type === "ui_surface",
499
+ ) as { data: Record<string, unknown> } | undefined;
500
+ expect((surface?.data.templateData as { task: string }).task).toBe(
501
+ "Corrected reading",
502
+ );
503
+ });
504
+
310
505
  test("says so when the render was bounded", async () => {
311
506
  const narrations = Array.from(
312
507
  { length: DEFAULT_MAX_ENTRIES + 5 },
@@ -709,4 +904,37 @@ describe("watch retrospective", () => {
709
904
  expect(prompt).toContain("<ax-tree>");
710
905
  expect(prompt).toContain("</ax-tree>");
711
906
  });
907
+
908
+ /**
909
+ * The instructions are the only place the model learns what to put in a
910
+ * question, and a field they leave unnamed is one it has to invent a name
911
+ * for. The surface schema strips what it does not recognize, so an invented
912
+ * name draws a page with no text on it rather than failing anywhere.
913
+ *
914
+ * Read out of the tool's validating schema instead of listed here, so a
915
+ * field added to the payload cannot be added without a line about it. The
916
+ * validating one rather than the advertised one, because the question shape
917
+ * is deliberately not advertised: the prompt is where the model gets it, so
918
+ * the prompt is what has to be complete.
919
+ */
920
+ test("the instructions name every field a question requires", async () => {
921
+ // A timeline of two words, so a field name found below is one the
922
+ // instructions wrote rather than one the recording happened to contain.
923
+ const { sessionId } = recordObservation("Window: Editor");
924
+ const prompt = buildWatchRetroPrompt(renderWatchTimeline(sessionId));
925
+
926
+ const schema = toToolInputSchema(watchRetroReportInputSchema);
927
+ const questionShape = requiredFieldsUnder(schema, ["questions"]);
928
+ const optionShape = requiredFieldsUnder(schema, ["questions", "options"]);
929
+
930
+ // The two the dev-QA payload came back without, named so the guard reads
931
+ // as the thing it is defending rather than as an empty loop if the shape
932
+ // lookup ever returns nothing.
933
+ expect(questionShape).toContain("prompt");
934
+ expect(optionShape).toContain("label");
935
+
936
+ for (const field of [...questionShape, ...optionShape]) {
937
+ expect(prompt).toMatch(new RegExp(`\\b${field}\\b`));
938
+ }
939
+ });
712
940
  });