@vellumai/assistant 0.11.3 → 0.11.4-staging.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (324) hide show
  1. package/ARCHITECTURE.md +11 -6
  2. package/docs/architecture/memory.md +11 -0
  3. package/docs/architecture/turn-actor.md +70 -0
  4. package/docs/flux-turn-detection-spike.md +243 -0
  5. package/docs/stt-provider-onboarding.md +3 -1
  6. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/channels.ts +11 -0
  7. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/channels.ts +11 -0
  8. package/node_modules/@vellumai/gateway-client/src/admission-policy-contract.ts +34 -0
  9. package/node_modules/@vellumai/gateway-client/src/index.ts +2 -0
  10. package/node_modules/@vellumai/service-contracts/src/channels.ts +11 -0
  11. package/openapi.yaml +140 -38
  12. package/package.json +1 -1
  13. package/scripts/voice-ttft-spike.ts +3 -3
  14. package/src/__tests__/app-compiler.test.ts +38 -3
  15. package/src/__tests__/attachments-store.test.ts +21 -12
  16. package/src/__tests__/byok-default-profile-ensure.test.ts +2 -0
  17. package/src/__tests__/call-setup-flow-name-capture.test.ts +0 -1
  18. package/src/__tests__/call-site-routing-provider.test.ts +1 -1
  19. package/src/__tests__/channel-availability-routes.test.ts +14 -1
  20. package/src/__tests__/channel-capabilities-dedupe.test.ts +214 -0
  21. package/src/__tests__/channel-delivery-store.test.ts +14 -14
  22. package/src/__tests__/config-loader-backfill.test.ts +3 -3
  23. package/src/__tests__/config-schema.test.ts +25 -10
  24. package/src/__tests__/conversation-agent-loop-inference-profile.test.ts +8 -11
  25. package/src/__tests__/conversation-agent-loop-overflow.test.ts +8 -11
  26. package/src/__tests__/conversation-agent-loop.test.ts +28 -20
  27. package/src/__tests__/conversation-attention-store.test.ts +63 -0
  28. package/src/__tests__/conversation-delete-schedule-cleanup.test.ts +0 -4
  29. package/src/__tests__/conversation-fork-crud.test.ts +69 -0
  30. package/src/__tests__/conversation-fork-referential.test.ts +67 -0
  31. package/src/__tests__/conversation-fork-retrospective.test.ts +24 -0
  32. package/src/__tests__/conversation-notifiers-provenance.test.ts +59 -0
  33. package/src/__tests__/conversation-queue.test.ts +55 -4
  34. package/src/__tests__/conversation-runtime-assembly.test.ts +134 -102
  35. package/src/__tests__/conversation-runtime-workspace.test.ts +14 -10
  36. package/src/__tests__/credential-prompt-route.test.ts +7 -10
  37. package/src/__tests__/custom-profile-ensure.test.ts +5 -1
  38. package/src/__tests__/discord-access-request-privacy.test.ts +5 -1
  39. package/src/__tests__/discord-requester-notice-privacy.test.ts +3 -3
  40. package/src/__tests__/document-append-idempotency.test.ts +233 -0
  41. package/src/__tests__/edit-propagation.test.ts +0 -7
  42. package/src/__tests__/helpers/mock-actor-context.ts +49 -0
  43. package/src/__tests__/helpers/mock-conversation.ts +13 -1
  44. package/src/__tests__/injector-chain.test.ts +63 -41
  45. package/src/__tests__/injector-disk-pressure.test.ts +11 -23
  46. package/src/__tests__/llm-context-resolution.test.ts +73 -1
  47. package/src/__tests__/llm-schema.test.ts +5 -2
  48. package/src/__tests__/mcp-list-plugin-servers.test.ts +250 -0
  49. package/src/__tests__/memory-retrieval-hook.test.ts +6 -5
  50. package/src/__tests__/messages-read-boundary-guard.test.ts +134 -0
  51. package/src/__tests__/mtime-cache.test.ts +1 -1
  52. package/src/__tests__/non-member-access-request.test.ts +0 -20
  53. package/src/__tests__/outbound-slack-persistence.test.ts +40 -1
  54. package/src/__tests__/plugin-import-boundary-guard.test.ts +5 -0
  55. package/src/__tests__/plugin-secret-pattern-contribution.test.ts +1 -1
  56. package/src/__tests__/post-compaction-reinjection-idempotency.test.ts +14 -7
  57. package/src/__tests__/provider-commit-message-generator.test.ts +20 -0
  58. package/src/__tests__/run-conversation-turn-persistence.test.ts +434 -105
  59. package/src/__tests__/scoped-approval-grants.test.ts +11 -6
  60. package/src/__tests__/secret-ingress-channel.test.ts +0 -1
  61. package/src/__tests__/skills.test.ts +32 -0
  62. package/src/__tests__/slack-edit-ordering-characterization.test.ts +0 -1
  63. package/src/__tests__/subagent-call-site-routing.test.ts +31 -19
  64. package/src/__tests__/subagent-spawn-and-await.test.ts +14 -10
  65. package/src/__tests__/turn-events-store.test.ts +43 -0
  66. package/src/__tests__/user-plugin-loader.test.ts +1 -1
  67. package/src/__tests__/visible-app-context.test.ts +16 -9
  68. package/src/__tests__/worker-entrypoint-guards.test.ts +54 -0
  69. package/src/__tests__/worker-plugin-surface.test.ts +77 -0
  70. package/src/__tests__/workspace-migration-142-consolidate-voice-front-door.test.ts +158 -0
  71. package/src/__tests__/workspace-migration-143-repair-deprecated-codex-model-id.test.ts +133 -0
  72. package/src/__tests__/workspace-migration-144-convert-stranded-subscription-openai-profiles.test.ts +316 -0
  73. package/src/__tests__/workspace-migration-145-collapse-profile-bindings-to-entries.test.ts +325 -0
  74. package/src/acp/__tests__/acp-claude-oauth.test.ts +10 -2
  75. package/src/acp/__tests__/auth-required.test.ts +161 -0
  76. package/src/acp/acp-claude-oauth.ts +19 -2
  77. package/src/acp/agent-process.test.ts +100 -0
  78. package/src/acp/agent-process.ts +29 -26
  79. package/src/acp/auth-required.ts +102 -0
  80. package/src/acp/session-manager.test.ts +119 -0
  81. package/src/acp/session-manager.ts +68 -2
  82. package/src/api/events/acp-auth-required.ts +55 -0
  83. package/src/api/index.ts +7 -0
  84. package/src/apps/app-store.ts +3 -0
  85. package/src/bundler/package-resolver.ts +2 -30
  86. package/src/calls/__tests__/voice-session-bridge.test.ts +173 -1
  87. package/src/calls/__tests__/voice-triage-escalate.test.ts +94 -0
  88. package/src/calls/call-controller.ts +9 -2
  89. package/src/calls/call-setup-flow.ts +0 -1
  90. package/src/calls/media-stream-stt-session.ts +15 -0
  91. package/src/calls/voice-session-bridge.ts +71 -16
  92. package/src/calls/voice-triage-escalate.ts +104 -2
  93. package/src/channels/__tests__/plugin-channel-declarations.test.ts +161 -0
  94. package/src/channels/config.ts +13 -0
  95. package/src/channels/plugin-channel-declarations.ts +108 -0
  96. package/src/channels/types.ts +30 -0
  97. package/src/cli/AGENTS.md +5 -2
  98. package/src/cli/commands/credentials.help.ts +2 -2
  99. package/src/cli/commands/inference-providers.ts +1 -1
  100. package/src/cli/commands/mcp.help.ts +13 -4
  101. package/src/cli/commands/mcp.ts +9 -0
  102. package/src/cli/commands/memory/__tests__/memory-v3.test.ts +128 -5
  103. package/src/cli/commands/memory/index.help.ts +43 -1
  104. package/src/cli/commands/memory/memory-v3.ts +64 -0
  105. package/src/cli/commands/stt.help.ts +27 -2
  106. package/src/cli/lib/__tests__/upgrade-plugin.test.ts +39 -0
  107. package/src/cli/lib/bundled-marketplace.json +13 -0
  108. package/src/cli/lib/upgrade-plugin.ts +42 -0
  109. package/src/config/__tests__/default-profile-catalog.test.ts +34 -2
  110. package/src/config/__tests__/default-provider.test.ts +6 -1
  111. package/src/config/__tests__/profile-materialization.test.ts +75 -19
  112. package/src/config/bundled-skills/acp/SKILL.md +6 -7
  113. package/src/config/bundled-skills/document-editor/SKILL.md +2 -2
  114. package/src/config/bundled-skills/document-editor/TOOLS.json +2 -2
  115. package/src/config/bundled-skills/media-processing/services/preprocess.ts +14 -4
  116. package/src/config/bundled-skills/settings/TOOLS.json +3 -3
  117. package/src/config/bundled-skills/transcribe/tools/transcribe-media.test.ts +22 -1
  118. package/src/config/bundled-skills/transcribe/tools/transcribe-media.ts +9 -2
  119. package/src/config/call-site-defaults.ts +4 -5
  120. package/src/config/default-profile-catalog.ts +82 -11
  121. package/src/config/default-profile-names.ts +4 -1
  122. package/src/config/default-provider-resolution.ts +4 -0
  123. package/src/config/llm-context-resolution.ts +11 -3
  124. package/src/config/llm-resolver.ts +28 -1
  125. package/src/config/profile-materialization.ts +70 -22
  126. package/src/config/schemas/__tests__/live-voice.test.ts +107 -4
  127. package/src/config/schemas/call-site-catalog.ts +4 -4
  128. package/src/config/schemas/live-voice.ts +57 -23
  129. package/src/config/schemas/llm.ts +59 -32
  130. package/src/config/schemas/mcp.ts +23 -0
  131. package/src/config/schemas/plugin-updates.ts +6 -2
  132. package/src/config/schemas/stt.ts +1 -0
  133. package/src/context/outbound-sanitize.ts +96 -1
  134. package/src/daemon/__tests__/plugin-mcp-reconcile.test.ts +82 -0
  135. package/src/daemon/conversation-agent-loop-handlers.ts +15 -10
  136. package/src/daemon/conversation-agent-loop.ts +17 -6
  137. package/src/daemon/conversation-messaging.ts +5 -1
  138. package/src/daemon/conversation-notifiers.ts +9 -1
  139. package/src/daemon/conversation-process.ts +9 -6
  140. package/src/daemon/conversation-runtime-assembly.ts +3 -4
  141. package/src/daemon/conversation-tool-setup.ts +1 -2
  142. package/src/daemon/conversation.ts +48 -0
  143. package/src/daemon/mcp-reload-service.ts +36 -6
  144. package/src/daemon/process-message.ts +13 -3
  145. package/src/daemon/providers-setup.ts +6 -3
  146. package/src/daemon/trust-context-types.ts +29 -0
  147. package/src/daemon/wake-conversation-ops.ts +3 -2
  148. package/src/documents/document-store.ts +138 -5
  149. package/src/hooks/hook-loader.ts +3 -3
  150. package/src/hooks/registry.ts +50 -6
  151. package/src/inbound/__tests__/oauth-callback-url.test.ts +83 -0
  152. package/src/inbound/oauth-callback-url.ts +61 -0
  153. package/src/live-voice/__tests__/live-voice-agent-turn.test.ts +1 -104
  154. package/src/live-voice/__tests__/live-voice-events.test.ts +7 -8
  155. package/src/live-voice/__tests__/live-voice-flux-turn-end.test.ts +932 -0
  156. package/src/live-voice/__tests__/live-voice-metrics.test.ts +115 -8
  157. package/src/live-voice/__tests__/live-voice-photo.test.ts +100 -0
  158. package/src/live-voice/__tests__/live-voice-progress.test.ts +60 -194
  159. package/src/live-voice/__tests__/live-voice-stt.test.ts +14 -0
  160. package/src/live-voice/__tests__/live-voice-triage-escalate.test.ts +29 -0
  161. package/src/live-voice/__tests__/live-voice-tts-session.test.ts +0 -483
  162. package/src/live-voice/__tests__/live-voice-vad.test.ts +0 -16
  163. package/src/live-voice/__tests__/progress-narration.test.ts +214 -0
  164. package/src/live-voice/live-voice-archive.ts +2 -0
  165. package/src/live-voice/live-voice-metrics.ts +57 -32
  166. package/src/live-voice/live-voice-photo.ts +1 -2
  167. package/src/live-voice/live-voice-session.ts +535 -314
  168. package/src/live-voice/progress-narration.ts +277 -0
  169. package/src/live-voice/protocol.ts +21 -1
  170. package/src/mcp/__tests__/effective-config.test.ts +238 -0
  171. package/src/mcp/__tests__/mcp-auth-orchestrator.test.ts +0 -1
  172. package/src/mcp/__tests__/mcp-oauth-client-registration.test.ts +200 -0
  173. package/src/mcp/__tests__/mcp-oauth-provider.test.ts +9 -9
  174. package/src/mcp/__tests__/plugin-server-credential-isolation.test.ts +95 -0
  175. package/src/mcp/client.ts +16 -11
  176. package/src/mcp/effective-config.ts +113 -0
  177. package/src/mcp/manager.ts +11 -6
  178. package/src/mcp/mcp-auth-orchestrator.ts +13 -22
  179. package/src/mcp/mcp-oauth-provider.ts +205 -240
  180. package/src/monitoring/__tests__/plugin-auto-update.test.ts +166 -3
  181. package/src/monitoring/plugin-auto-update.ts +128 -24
  182. package/src/notifications/signal.ts +1 -0
  183. package/src/permissions/confirmation-guardian-request.test.ts +15 -11
  184. package/src/permissions/confirmation-guardian-request.ts +2 -2
  185. package/src/permissions/question-guardian-request.test.ts +14 -6
  186. package/src/permissions/question-guardian-request.ts +1 -2
  187. package/src/persistence/attachments-store.ts +8 -1
  188. package/src/persistence/bookmark-crud.ts +3 -7
  189. package/src/persistence/conversation-attention-store.ts +16 -45
  190. package/src/persistence/conversation-crud.ts +33 -4
  191. package/src/persistence/conversation-lineage.ts +9 -0
  192. package/src/persistence/conversation-queries.ts +108 -41
  193. package/src/persistence/delivery-crud.ts +38 -29
  194. package/src/persistence/external-conversation-store.ts +32 -4
  195. package/src/persistence/llm-request-log-store.ts +4 -10
  196. package/src/persistence/llm-usage-store.ts +8 -3
  197. package/src/persistence/message-reads.test.ts +197 -0
  198. package/src/persistence/message-reads.ts +211 -0
  199. package/src/persistence/migrations/366-chatgpt-subscription-row-identity.test.ts +120 -0
  200. package/src/persistence/migrations/366-chatgpt-subscription-row-identity.ts +62 -0
  201. package/src/persistence/real-user-turn-filter.ts +27 -3
  202. package/src/persistence/steps.ts +9 -0
  203. package/src/plugin-api/__tests__/oauth-callback-url-export.test.ts +29 -0
  204. package/src/plugin-api/conversation-turn.ts +168 -5
  205. package/src/plugin-api/index.ts +21 -5
  206. package/src/plugins/__tests__/mcp-servers.test.ts +371 -0
  207. package/src/plugins/defaults/memory/AGENTS.md +4 -0
  208. package/src/plugins/defaults/memory/__tests__/buffer-format.test.ts +204 -0
  209. package/src/plugins/defaults/memory/__tests__/memory-retrospective-accounting.test.ts +72 -0
  210. package/src/plugins/defaults/memory/__tests__/memory-retrospective-job.test.ts +4 -1
  211. package/src/plugins/defaults/memory/__tests__/memory-retrospective-provider-path.test.ts +4 -1
  212. package/src/plugins/defaults/memory/buffer-format.ts +165 -0
  213. package/src/plugins/defaults/memory/context-search/sources/conversations.ts +6 -0
  214. package/src/plugins/defaults/memory/graph/image-ref-utils.ts +3 -0
  215. package/src/plugins/defaults/memory/graph/tool-handlers.ts +1 -30
  216. package/src/plugins/defaults/memory/graph-topology/pending-buffer.test.ts +34 -0
  217. package/src/plugins/defaults/memory/graph-topology/pending-buffer.ts +8 -12
  218. package/src/plugins/defaults/memory/hooks/post-compact.ts +1 -4
  219. package/src/plugins/defaults/memory/indexer.ts +3 -1
  220. package/src/plugins/defaults/memory/memory-retrospective-accounting.ts +19 -7
  221. package/src/plugins/defaults/memory/src/__tests__/memory-v3-gate-stats.test.ts +281 -0
  222. package/src/plugins/defaults/memory/src/memory-v3-routes.ts +207 -0
  223. package/src/plugins/defaults/memory/substrate/__tests__/consolidation-job.test.ts +33 -0
  224. package/src/plugins/defaults/memory/substrate/__tests__/static-context.test.ts +199 -2
  225. package/src/plugins/defaults/memory/substrate/consolidation-job.ts +25 -18
  226. package/src/plugins/defaults/memory/substrate/skill-content.ts +8 -1
  227. package/src/plugins/defaults/memory/substrate/static-context.ts +160 -4
  228. package/src/plugins/defaults/memory/substrate/sweep-job.ts +2 -4
  229. package/src/plugins/defaults/memory/v1/graph/extraction.ts +3 -1
  230. package/src/plugins/defaults/memory/v3/prune.ts +2 -0
  231. package/src/plugins/defaults/memory/v3/selection-log-store.ts +2 -0
  232. package/src/plugins/defaults/memory/worker.ts +6 -3
  233. package/src/plugins/external-plugin-loader.ts +47 -0
  234. package/src/plugins/mcp-servers.ts +361 -0
  235. package/src/plugins/mtime-cache.ts +23 -49
  236. package/src/plugins/worker-plugin-surface.ts +33 -0
  237. package/src/providers/__tests__/connection-model-compat.test.ts +1 -1
  238. package/src/providers/__tests__/dispatch-connection-routing.test.ts +214 -2
  239. package/src/providers/__tests__/preflight-resolved-config.test.ts +57 -0
  240. package/src/providers/__tests__/retry-callsite.test.ts +5 -2
  241. package/src/providers/call-site-routing.ts +30 -3
  242. package/src/providers/connection-resolution.ts +194 -11
  243. package/src/providers/inference/auth.ts +6 -0
  244. package/src/providers/inference/connection-availability.ts +24 -2
  245. package/src/providers/inference/connections.ts +2 -0
  246. package/src/providers/model-intents.ts +26 -6
  247. package/src/providers/openai/codex-models.ts +2 -1
  248. package/src/providers/provider-send-message.ts +32 -3
  249. package/src/providers/speech-to-text/__tests__/deepgram-flux-frames.test.ts +433 -0
  250. package/src/providers/speech-to-text/__tests__/deepgram-flux-realtime.test.ts +620 -0
  251. package/src/providers/speech-to-text/__tests__/provider-catalog.test.ts +34 -0
  252. package/src/providers/speech-to-text/__tests__/resolve.test.ts +285 -6
  253. package/src/providers/speech-to-text/deepgram-flux-frames.ts +395 -0
  254. package/src/providers/speech-to-text/deepgram-flux-realtime.ts +719 -0
  255. package/src/providers/speech-to-text/provider-catalog.ts +99 -8
  256. package/src/providers/speech-to-text/resolve.ts +25 -2
  257. package/src/routes/worker.ts +17 -5
  258. package/src/runtime/access-request-helper.ts +9 -12
  259. package/src/runtime/agent-wake.ts +3 -3
  260. package/src/runtime/pre-first-message-gate.ts +4 -0
  261. package/src/runtime/routes/__tests__/acp-claude-auth-routes.test.ts +12 -4
  262. package/src/runtime/routes/__tests__/conversation-list-routes.test.ts +219 -1
  263. package/src/runtime/routes/__tests__/conversation-query-routes.test.ts +52 -0
  264. package/src/runtime/routes/__tests__/default-provider-routes.test.ts +61 -0
  265. package/src/runtime/routes/__tests__/inference-profiles-routes.test.ts +44 -0
  266. package/src/runtime/routes/__tests__/inference-provider-connection-routes.test.ts +102 -1
  267. package/src/runtime/routes/__tests__/plugins-routes.test.ts +44 -0
  268. package/src/runtime/routes/__tests__/stt-routes.test.ts +25 -0
  269. package/src/runtime/routes/__tests__/user-route-dispatcher.test.ts +62 -1
  270. package/src/runtime/routes/channel-availability-routes.ts +32 -14
  271. package/src/runtime/routes/channel-route-shared.ts +0 -6
  272. package/src/runtime/routes/chatgpt-subscription-auth-routes.ts +6 -6
  273. package/src/runtime/routes/conversation-list-routes.ts +112 -1
  274. package/src/runtime/routes/conversation-query-routes.ts +40 -27
  275. package/src/runtime/routes/credential-prompt-routes.ts +4 -7
  276. package/src/runtime/routes/default-provider-routes.ts +15 -0
  277. package/src/runtime/routes/inbound-message-handler.ts +17 -41
  278. package/src/runtime/routes/inbound-stages/acl-enforcement.test.ts +0 -1
  279. package/src/runtime/routes/inbound-stages/acl-enforcement.ts +0 -9
  280. package/src/runtime/routes/inbound-stages/admission-policy.ts +1 -17
  281. package/src/runtime/routes/inbound-stages/bootstrap-intercept.test.ts +0 -1
  282. package/src/runtime/routes/inbound-stages/bootstrap-intercept.ts +2 -3
  283. package/src/runtime/routes/inbound-stages/edit-intercept.ts +1 -3
  284. package/src/runtime/routes/inbound-stages/guardian-reply-intercept.test.ts +0 -1
  285. package/src/runtime/routes/inbound-stages/guardian-reply-intercept.ts +3 -4
  286. package/src/runtime/routes/inbound-stages/reaction-intercept.test.ts +0 -1
  287. package/src/runtime/routes/inbound-stages/reaction-intercept.ts +11 -20
  288. package/src/runtime/routes/inbound-stages/secret-ingress-check.ts +2 -3
  289. package/src/runtime/routes/inference-profiles-routes.ts +20 -11
  290. package/src/runtime/routes/inference-provider-connection-routes.ts +77 -15
  291. package/src/runtime/routes/log-export-routes.ts +3 -0
  292. package/src/runtime/routes/mcp-auth-routes.ts +148 -57
  293. package/src/runtime/routes/plugins-routes.ts +21 -3
  294. package/src/runtime/routes/stt-routes.ts +31 -25
  295. package/src/runtime/routes/surface-conversation-resolver.ts +3 -0
  296. package/src/runtime/routes/user-route-dispatcher.ts +39 -14
  297. package/src/runtime/routes/user-route-import.ts +108 -0
  298. package/src/schedule/worker.ts +6 -0
  299. package/src/security/oauth2.ts +6 -22
  300. package/src/stt/__tests__/daemon-batch-transcriber.test.ts +22 -0
  301. package/src/stt/__tests__/types.test.ts +94 -0
  302. package/src/stt/daemon-batch-transcriber.ts +10 -0
  303. package/src/stt/stt-stream-session.ts +8 -4
  304. package/src/stt/types.ts +103 -0
  305. package/src/subagent/manager.ts +1 -3
  306. package/src/subagent/types.ts +7 -6
  307. package/src/tools/acp/spawn.test.ts +97 -0
  308. package/src/tools/acp/spawn.ts +32 -0
  309. package/src/tools/document/document-tool.ts +12 -3
  310. package/src/tools/registry.ts +2 -1
  311. package/src/tools/workflows/run-workflow.ts +1 -2
  312. package/src/tts/__tests__/reasoning-tag-filter.test.ts +63 -0
  313. package/src/tts/reasoning-tag-filter.ts +110 -0
  314. package/src/workspace/byok-default-profile-ensure.ts +61 -18
  315. package/src/workspace/custom-profile-ensure.ts +4 -24
  316. package/src/workspace/migrations/142-consolidate-voice-front-door.ts +70 -0
  317. package/src/workspace/migrations/143-repair-deprecated-codex-model-id.ts +134 -0
  318. package/src/workspace/migrations/144-convert-stranded-subscription-openai-profiles.ts +265 -0
  319. package/src/workspace/migrations/145-collapse-profile-bindings-to-entries.ts +328 -0
  320. package/src/workspace/migrations/__tests__/141-stt-english-default-to-multilingual.test.ts +0 -10
  321. package/src/workspace/migrations/registry.ts +8 -0
  322. package/src/workspace/provider-commit-message-generator.ts +7 -5
  323. package/src/live-voice/__tests__/front-decision.test.ts +0 -645
  324. package/src/live-voice/front-decision.ts +0 -476
package/ARCHITECTURE.md CHANGED
@@ -624,16 +624,21 @@ To add a new daemon batch STT provider, follow the full checklist in `docs/stt-p
624
624
 
625
625
  Real-time conversation chat message capture on macOS uses a WebSocket-based streaming STT path. When the configured `services.stt` provider supports conversation streaming (determined by the `conversationStreamingMode` field in the provider catalog), native clients open a WebSocket session through the gateway to the daemon's `/v1/stt/stream` endpoint. The daemon resolves a `StreamingTranscriber` for the configured provider and streams partial/final transcript events back to the client in real time.
626
626
 
627
- Two provider adapters are supported, each implementing the `StreamingTranscriber` interface from `src/stt/types.ts`:
627
+ Each provider that advertises a conversation streaming mode in the catalog ships an adapter implementing the `StreamingTranscriber` interface from `src/stt/types.ts`. The catalog (`src/providers/speech-to-text/provider-catalog.ts`) is the source of truth for which providers are streaming-capable; `resolveStreamingTranscriber()` maps each to its adapter:
628
628
 
629
- | Provider | Adapter | Mode | Mechanism |
630
- | ----------------- | ----------------------------------------------------------- | ------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
631
- | **Deepgram** | `src/providers/speech-to-text/deepgram-realtime.ts` | `realtime-ws` | Opens a WebSocket to Deepgram's `/v1/listen` endpoint, forwards raw PCM audio, normalizes Deepgram's `is_final`/`speech_final` semantics into `partial`/`final` events. Uses model `nova-2`. |
632
- | **Google Gemini** | `src/providers/speech-to-text/google-gemini-live-stream.ts` | `realtime-ws` | Opens a bidirectional streaming session against Gemini's Live API (`ai.live.connect`), forwards PCM audio frames, and normalizes `serverContent.inputTranscription` events into `partial`/`final` events. Uses model `gemini-2.5-flash-native-audio-latest`. |
629
+ | Provider | Adapter | Mode | Mechanism |
630
+ | ------------------ | ----------------------------------------------------------- | ------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
631
+ | **Deepgram** | `src/providers/speech-to-text/deepgram-realtime.ts` | `realtime-ws` | Opens a WebSocket to Deepgram's `/v1/listen` endpoint, forwards raw PCM audio, normalizes Deepgram's `is_final`/`speech_final` semantics into `partial`/`final` events. Uses model `nova-2`. |
632
+ | **Deepgram Flux** | `src/providers/speech-to-text/deepgram-flux-realtime.ts` | `realtime-ws` | Opens a WebSocket to Deepgram's `/v2/listen` conversational endpoint. The model decides turn boundaries, so the adapter also emits `turn-start`, `eager-turn-end`, `turn-resumed`, and `turn-end`. Streaming only, no batch endpoint. Uses model `flux-general-en`. |
633
+ | **Google Gemini** | `src/providers/speech-to-text/google-gemini-live-stream.ts` | `realtime-ws` | Opens a bidirectional streaming session against Gemini's Live API (`ai.live.connect`), forwards PCM audio frames, and normalizes `serverContent.inputTranscription` events into `partial`/`final` events. Uses model `gemini-2.5-flash-native-audio-latest`. |
634
+ | **Vellum managed** | `src/providers/speech-to-text/vellum-managed-realtime.ts` | `realtime-ws` | Wraps the Deepgram adapter and dials the gateway speech relay (`/v1/speech/stt/stream`) instead of Deepgram directly; both relay legs speak Deepgram's wire protocol, and the model is pinned server-side. |
635
+ | **xAI** | `src/providers/speech-to-text/xai-realtime.ts` | `realtime-ws` | Opens a WebSocket to `wss://api.x.ai/v1/stt` and normalizes xAI's `transcript.partial`/`transcript.done` payloads into `partial`/`final` events. |
636
+ | **OpenAI Whisper** | `src/providers/speech-to-text/openai-whisper-stream.ts` | `incremental-batch` | Whisper exposes no streaming endpoint, so the shared `incremental-batch-stream.ts` strategy re-transcribes an accumulating buffer and diffs successive results into `partial`/`final` events. |
633
637
 
634
638
  **Provider-specific behavior differences:**
635
639
 
636
640
  - **Deepgram (`realtime-ws`)**: True WebSocket streaming with sub-second partial latency. Emits `partial` events for `is_final: false` frames and `final` events for `is_final: true` frames. Supports backpressure (drops audio frames when `bufferedAmount > 1 MiB`). Sends `CloseStream` message on stop with a 5-second grace period for the provider to flush remaining finals. Inactivity timeout: 30 seconds (provider-side hang detection). Connect timeout: 10 seconds. Auth errors map to close codes 1008/4001; rate limits to 1013.
641
+ - **Deepgram Flux (`realtime-ws`)**: The model owns turn detection, so the adapter carries no endpointing heuristics and exposes no `finalizeUtterance`: callers feature-detect it and fall back to `stop()`. It emits the four turn-boundary events alongside transcripts, where `eager-turn-end` is a speculative end-of-turn that a later `turn-resumed` retracts or a `turn-end` confirms. Streaming only: batch callers get a diagnostic naming `deepgram` as the batch-capable provider on the same credential.
637
642
  - **Google Gemini (`realtime-ws`)**: WebSocket-backed Live API session. Partials are emitted as Gemini streams `inputTranscription.text` fragments; a `final` is emitted when the server signals `generationComplete` or `turnComplete`. On `stop()`, the adapter sends `audioStreamEnd: true` and waits up to a 5-second grace window for the server to flush remaining transcription before force-closing. Inactivity timeout: 30 seconds. Connect timeout: 10 seconds. Close codes 1008/4001 map to `auth`; 1013 maps to `rate-limit`; other codes map to `provider-error`. The model's own text turn is suppressed via a silent system instruction so we only pay for transcription.
638
643
 
639
644
  **Session lifecycle (daemon side):**
@@ -643,7 +648,7 @@ Two provider adapters are supported, each implementing the `StreamingTranscriber
643
648
  3. The transcriber's `start()` method opens the provider session.
644
649
  4. A `ready` event (with `provider` field) is sent to the client, signaling that audio frames are accepted.
645
650
  5. Client sends `audio` frames (binary WebSocket frames or base64-encoded JSON) and a `stop` event when recording ends.
646
- 6. The transcriber emits `partial` and `final` events, forwarded to the client as JSON frames with monotonic `seq` numbers.
651
+ 6. The transcriber emits transcript events, plus turn-boundary events (`turn-start`, `eager-turn-end`, `turn-resumed`, `turn-end`) from providers that detect them, forwarded to the client as JSON frames with monotonic `seq` numbers.
647
652
  7. The session closes deterministically on: client disconnect, `stop` event followed by provider `closed`, idle timeout (60 seconds), or runtime shutdown.
648
653
 
649
654
  **Session lifecycle (client side):**
@@ -75,6 +75,17 @@ graph LR
75
75
  - `handleRemember` (`graph/tool-handlers.ts`) appends timestamped bullets to
76
76
  `memory/buffer.md` + the daily archive whenever memory is enabled. Facts may
77
77
  carry `[[slug]]` page hints that consolidation reads first when filing.
78
+ - The buffer entry format itself is owned by `buffer-format.ts` at the plugin
79
+ root: the writer (`formatRememberEntry`) plus the single matcher every reader
80
+ uses. A fact may span several lines, so the entry and the line are different
81
+ units, and the readers below (consolidation's cutoff, the injected `<info>`
82
+ Buffer cap, the Memory tab's pending nodes) all recognize entries through
83
+ that one matcher rather than their own. An entry opens with a timestamped
84
+ bullet at column 0 and its body is indented under it, which is what makes the
85
+ format round-trip: the delimiter is the column-0 bullet shape, so nesting the
86
+ body keeps fact content from imitating one. Entries written before that
87
+ nesting existed still parse, since an unindented body line that is not itself
88
+ entry-shaped is read as a continuation.
78
89
  - **Consolidation** (`substrate/consolidation-job.ts`) is a background
79
90
  agent conversation that files buffer entries into concept pages, rewrites
80
91
  the aggregate views, and trims the buffer. Scheduling
@@ -0,0 +1,70 @@
1
+ # Turn Actor
2
+
3
+ Which actor a piece of code means when it reads trust from a conversation.
4
+
5
+ ## Two different questions
6
+
7
+ A `Conversation` is long-lived and several actors can send into it. Code that reads trust is asking one of two things, and they have different answers:
8
+
9
+ - **The acting actor.** Who this turn is executing for. Governs authorization, and is recorded as the provenance of anything the turn persists. Undefined between turns.
10
+ - **The resting actor.** Who the conversation belongs to when no turn is running. Used to hydrate a new turn, to scope history, and by routes answering questions about a conversation rather than about a turn.
11
+
12
+ They coincide most of the time, which is why one field answering both went unnoticed. They diverge whenever another actor sends while a turn is in flight, or when a turn runs on a conversation an earlier actor last touched.
13
+
14
+ ## Read through the accessors
15
+
16
+ ```ts
17
+ conversation.getTurnTrust(); // who this turn is for; undefined if unrecorded
18
+ conversation.getTrustContext(); // who the conversation belongs to
19
+ conversation.getTurnOrRestingTrust(); // the turn's actor, else the owner
20
+ ```
21
+
22
+ The names follow the existing convention on `Conversation`: `getTurn*` for
23
+ per-turn values (`getTurnActorPrincipalId`, `getTurnChannelContext`), plain
24
+ `get*` for conversation-level ones (`getAuthContext`).
25
+
26
+ Do not read `trustContext` or `currentTurnTrustContext` directly. Call sites handed a conversation-shaped context rather than the class (handler deps, the messaging context) use the structural counterparts `turnOrRestingTrust(ctx)` / `restingTrust(ctx)` from `trust-context-types.ts`, which read the same fields and carry the same names. The accessors exist so that every call site states which question it is asking; a raw field read states nothing, and the wrong answer is silent.
27
+
28
+ **Use `getTurnTrust()`** for authorization decisions and for routing a reply back to the requester: cases where substituting the conversation's owner would be wrong rather than approximate, so `undefined` must surface and be handled.
29
+
30
+ **Use `getTurnOrRestingTrust()`** for provenance stamped onto persisted rows (see `provenanceFromTrustContext`). Provenance is read back by the memory indexer, which runs extraction only for guardian rows, and by the transcript assembly, which wraps non-guardian user content before it reaches the model. The turn's actor is correct when a turn is running; rows persisted with no turn in flight (wake notices) must keep the owner's class, because stamping `"unknown"` there silently stops memory extraction for the owner's own flows.
31
+
32
+ **Use `getTrustContext()`** when there is no turn: HTTP routes reporting conversation state, hydration, and persistence of conversation-level options. Also when a caller deliberately wants the conversation's owner rather than whoever is currently acting; that intent should be obvious from the call site, and if it is not, it is probably the wrong accessor.
33
+
34
+ ## When the acting actor is unknown
35
+
36
+ `getTurnTrust()` returns `undefined`. It does not substitute the owner, because
37
+ a caller asking who is acting should not silently receive someone else.
38
+
39
+ Callers that can accept the owner as a stand-in ask for that by name:
40
+
41
+ ```ts
42
+ conversation.getTurnOrRestingTrust();
43
+ ```
44
+
45
+ That fallback is load-bearing, not transitional politeness: a deferred wake
46
+ fires with no inbound actor, and refusing it an answer denies every sensitive
47
+ tool in the resumed turn (LUM-2929). The substitution lives in one named
48
+ method, so it is greppable, and removable in one place once every entry point
49
+ records a turn actor.
50
+
51
+ Callers for which the owner would be wrong rather than approximate, such as
52
+ authorization, call `getTurnTrust()` alone and handle `undefined`. Provenance
53
+ is deliberately not in that set: its readers treat an absent class more
54
+ permissively than `"unknown"`, so failing closed there fails open downstream.
55
+
56
+ ## Writers that stamp for a run
57
+
58
+ `agent-wake` and `voice-session-bridge` set the conversation's trust before a run
59
+ and restore the prior value afterwards, each guarding the restore so a turn that
60
+ started in between is not clobbered. They are supplying the acting actor for
61
+ their run, and are covered by this contract.
62
+
63
+ `call-controller` keeps its own `trustContext` on its own object and never reads
64
+ the conversation's. It is outside this contract.
65
+
66
+ ## Why this exists
67
+
68
+ Trust was read as `currentTurnTrustContext ?? trustContext` at each consumer, so every call site independently chose which actor it got, invisibly. Between June and August 2026 that produced seven separate fixes, alternating between a turn running as the wrong actor and a turn running as nobody, depending on whether the conversation was still resident in cache. A per-caller `preferTurnSnapshot` flag was added so one consumer could opt into the turn value, which is the same choice made explicit for one call site.
69
+
70
+ Naming the two questions is what makes the wrong answer visible at the point it is chosen.
@@ -0,0 +1,243 @@
1
+ # Deepgram Flux Turn Detection: Local Spike Runbook
2
+
3
+ How to enable Deepgram Flux end-of-turn in live voice on your own machine, run it against the existing front-door path, and read a latency comparison that is actually honest.
4
+
5
+ Flux is a spike. `liveVoice.flux.turnEnd.enabled` defaults to `false`, the existing local-VAD-plus-front-door hold machinery is untouched and still the default, and it is still the fallback when Flux goes quiet. Nothing here is a rollout.
6
+
7
+ ---
8
+
9
+ ## Read this before you read any number
10
+
11
+ **`endpointCommitLatencyMs` is the headline, and it is the only endpoint field that is like-for-like across the two arms.** It measures the local VAD speech-stop mark to the moment the turn committed. Take its median per arm and compare the two medians. There is no arithmetic to do and no source-specific semantics to remember.
12
+
13
+ It is stamped in `releaseUtterance` (`live-voice-session.ts`), which every committed turn passes through whichever decider released it, from the same speech-stop anchor on both. So the two arms are one population measured over one span. It is absent only on a turn that never committed, and in push-to-talk, which this spike does not run.
14
+
15
+ It is also the only endpoint number that the Flux socket teardown stays out of: `releaseUtterance` stamps it before it stops the transcriber. `roundTripMs`, `llmFirstDeltaMs`, and `totalMs` all carry that teardown on a Flux arm, and so does what the caller hears. Read "The per-turn socket teardown, and why it is load-bearing" in section 3 before you compare any of those three against a `deepgram` run.
16
+
17
+ ### Why the other endpoint fields are not the comparison
18
+
19
+ `endpointDecisionMaxLatencyMs`, `endpointHoldCount`, and `endpointDecisionSource` all still exist and are all still worth reading. But `endpointDecisionMaxLatencyMs` is **not** comparable between arms, in two independent ways, and comparing it flatters Flux by roughly a full second with nothing in the output to tell you.
20
+
21
+ **It spans different things.**
22
+
23
+ | Path | What `endpointDecisionMaxLatencyMs` actually spans |
24
+ | -------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
25
+ | `endpointDecisionSource: "provider"` | Local VAD speech-stop mark to the `turn-end` event. This is the **whole** end-of-turn latency. |
26
+ | `endpointDecisionSource: "front-door"` | Speculative dispatch to the hold verdict, i.e. the endpoint-decider LLM roundtrip **only**. The trailing-silence wait that had to elapse before that dispatch is not in the number, and neither is the hold extension that follows it. |
27
+
28
+ Source: `handleProviderTurnEnd` passes `this.msSinceLocalSpeechStop()` (`live-voice-session.ts`), while `holdSpeculativeTurn` passes `Date.now() - turn.speculativeDispatchedAtMs`, and the dispatch happens at the silence boundary.
29
+
30
+ **It samples different populations.** `markEndpointDecision` is called from exactly two places: the Flux commit, and the hold verdict. A front-door turn that committed straight through the boundary emits **no** `endpoint_decision` at all, and its metrics frame carries none of the three endpoint fields. So the front-door sample is "turns the model judged mid-thought", the slow tail by construction, while the Flux sample is every turn. Two denominators that cannot be reconciled after the fact is exactly why `endpointCommitLatencyMs` exists.
31
+
32
+ Read `endpointDecisionMaxLatencyMs` as a **breakdown of** the headline, not against the other arm: on a Flux turn it isolates how much of the commit latency was Flux's own decision, and on a held front-door turn it isolates the decider LLM roundtrip.
33
+
34
+ ### What the front door actually costs
35
+
36
+ You do not need this to read the headline. You need it when a front-door number surprises you.
37
+
38
+ - On an **unheld** turn the added end-of-turn cost is `silenceThresholdMs` alone. The speculative leg dispatched at the boundary _is_ the assistant turn, so if the model does not return the hold token, generation has been running since the boundary and the decider roundtrip is not added latency. That is the common case, and the real bar Flux has to beat.
39
+ - On a **held** turn, add `endpointExtensionMs` (default **1500**) per hold, capped at `endpointMaxExtensions` (default **2**), so up to 3000ms on top.
40
+ - `silenceThresholdMs` is the trailing-silence wait the boundary cost. Its value is, in precedence order: the client's explicit "pause before reply" setting sent on the start frame, else `liveVoice.vad.silenceThresholdMs` (schema default **1200**, `config/schemas/live-voice.ts`), else the in-code `DEFAULT_SILENCE_THRESHOLD_MS` of **800** (`live-voice-session.ts`). In a real daemon run the factory always seeds the config value, so **1200 is the number in force** unless you moved the web client's pause slider or overrode the config key. The 800 constant governs only a session built with no `liveVoice` config at all, which in practice means tests.
41
+
42
+ ### The cross-check that needs no correction
43
+
44
+ The per-turn timestamps in the metrics snapshot are absolute and come from one clock, and both paths stamp `utteranceEndAtMs` at the moment the turn commits (`markUtteranceReleased` runs inside `releaseUtterance`, which both paths call). `speechStartAtMs` is stamped at local VAD onset and is first-wins, so it survives a pause the front door held across.
45
+
46
+ ```
47
+ utteranceEndAtMs - speechStartAtMs
48
+ ```
49
+
50
+ Say the **same scripted sentence** on both arms, and the difference between the arms in that figure is the endpointing cost. It is an independent path to the same answer as `endpointCommitLatencyMs`: that field anchors at the speech-stop mark and this one at the VAD speech onset, so a run where the two disagree means the mic was picking up something the script did not say. Use it to sanity-check the headline.
51
+
52
+ ---
53
+
54
+ ## 1. Credentials
55
+
56
+ Flux shares the existing Deepgram credential. `credentialProvider: "deepgram"` in `providers/speech-to-text/provider-catalog.ts`, so there is **no new key to obtain and nowhere new to put it**. If `deepgram` already transcribes for you, Flux already has what it needs.
57
+
58
+ If it does not, set the Deepgram key the ordinary way (client Settings, Speech-to-text card) and pick either Deepgram entry; both write the same `deepgram` credential.
59
+
60
+ ## 2. Enable Flux
61
+
62
+ > **Setting `services.stt.provider` to `deepgram-flux` turns off batch transcription for the whole workspace, not just live voice.** `services.stt.provider` is the single source of truth for every STT route, and Flux is the only provider in the catalog with no `daemon-batch` boundary. For as long as it is set, these all stop working: voice-message and inbound-attachment transcription, the `transcribe` skill, the `media-processing` skill's audio segments, `POST /v1/stt/transcribe`, and phone-call transcription. Each reports a message naming Flux as the cause rather than failing silently, but they do not fall back to `deepgram` on their own. **Set `services.stt.provider` back to `deepgram` when the spike is over**, and do not run the spike on an assistant that is also taking calls or handling voice messages.
63
+
64
+ Two keys in `config.json`, which lives at `$VELLUM_WORKSPACE_DIR/config.json` (default `~/.vellum/workspace/config.json`):
65
+
66
+ ```json
67
+ {
68
+ "services": {
69
+ "stt": {
70
+ "provider": "deepgram-flux"
71
+ }
72
+ },
73
+ "liveVoice": {
74
+ "flux": {
75
+ "turnEnd": {
76
+ "enabled": true
77
+ }
78
+ }
79
+ }
80
+ }
81
+ ```
82
+
83
+ Restart the daemon so the config is reloaded, then start a live-voice session **hands-free**. Push-to-talk is deliberately excluded: the latch requires a server-VAD turn detector, because in PTT the client's release already is the boundary and there is nothing for Flux to decide.
84
+
85
+ The rest of `liveVoice.flux` is optional and defaulted (`config/schemas/live-voice.ts`):
86
+
87
+ | Key | Default | Range | Notes |
88
+ | ------------------- | ----------------- | ------------ | --------------------------------------------------------------------------------------------------------------------------------------- |
89
+ | `model` | `flux-general-en` | any string | English only in this spike. |
90
+ | `eotThreshold` | `0.7` | 0.5 to 0.9 | Lower commits sooner and cuts speakers off more; higher adds latency. |
91
+ | `eagerEotThreshold` | _unset_ | 0.3 to 0.9 | Leave it unset. See "Known gaps". |
92
+ | `eotTimeoutMs` | `5000` | 500 to 60000 | Silence after which Flux force-ends a turn it never got confident about. Lower it to 1500 to 2000 for a measurement run; see section 4. |
93
+
94
+ ## 3. Run the A/B
95
+
96
+ **Flip `turnEnd.enabled` between runs and leave `services.stt.provider` on `deepgram-flux` in both arms.** That holds the STT engine, the model, the socket, and the transcriber lifecycle constant, so the only thing that changes is which signal commits the turn.
97
+
98
+ - **Arm A (control):** `provider: "deepgram-flux"`, `turnEnd.enabled: false`. Flux transcribes; the four turn-detection events are ignored; the local silence boundary and the front-door hold path run exactly as they do today.
99
+ - **Arm B (treatment):** `provider: "deepgram-flux"`, `turnEnd.enabled: true`.
100
+
101
+ **Do not A/B by switching the provider between `deepgram` and `deepgram-flux`.** That confounds two changes at once:
102
+
103
+ 1. It swaps the STT model. `deepgram` runs `nova-2` (`DEFAULT_MODEL`, `deepgram-realtime.ts`), so any transcript-quality or first-partial difference lands in your latency numbers.
104
+ 2. It swaps the transcriber lifecycle. `deepgram` implements `finalizeUtterance`, so the session adopts it as a persistent stream shared across the whole session (`sharedTranscriber`) and never tears it down between turns. Flux implements no `finalizeUtterance`, so every utterance owns its own `/v2/listen` socket and every release closes it. That per-turn socket churn is inside `roundTripMs` and `totalMs` on the Flux side and absent on the `deepgram` side, so a naive provider-swap A/B attributes it to turn detection. The next section is what it costs and why it cannot be removed.
105
+
106
+ Say the same scripted set of utterances in both arms. Include at least a few deliberate mid-sentence thinking pauses, because that is the case the hold path exists for and the case Flux has to not regress.
107
+
108
+ ### The per-turn socket teardown, and why it is load-bearing
109
+
110
+ The Flux adapter implements no `finalizeUtterance`. That is forced by Flux's wire protocol, not a statement that Flux owns the turn boundary, and adding the method as a no-op would break transcript correctness.
111
+
112
+ `parseFluxFrame` emits `final` only on `EndOfTurn`, and that is the adapter's sole source of `final`. Flux offers no mid-stream flush, so `CloseStream` is the only message that makes it answer for a turn still in progress, which is exactly what `stop()` sends. A no-op `finalizeUtterance` would report `finalized` without flushing anything, so a turn released on a caller-side boundary would dispatch on an empty transcript while its real text arrived afterwards and was dropped as a late final segment. With `turnEnd.enabled` at its default `false` that is every turn, because the release always comes from the local silence path. With the latch on it is still every fail-open fallback, every max-duration force-end, and every barge-in, all of which release with a Flux turn open.
113
+
114
+ So the release path is: `stop()` sends `CloseStream`, the adapter waits for Deepgram's close frame under a `CLOSE_GRACE_MS` ceiling (**5000ms**, `deepgram-flux-realtime.ts`), and then emits `closed`. The cycle only reaches `transcriber_closed` on that `closed` event, and `startAssistantTurnIfReady` refuses to dispatch before it. Two consequences for the numbers:
115
+
116
+ - **`endpointCommitLatencyMs` does not contain the teardown.** `releaseUtterance` stamps the commit latency and `utteranceEndAtMs` before it calls `stop()`, so the headline comparison is clean on both arms.
117
+ - **`roundTripMs`, `llmFirstDeltaMs`, and `totalMs` do contain it,** and so does the silence the caller sits through. On a Flux arm the assistant turn cannot start until the socket has closed. A `deepgram` run pays none of that: the shared stream stays open and the cycle reaches `transcriber_closed` on the `finalized` event instead.
118
+
119
+ The re-dial is off the end-of-turn path but is not free either. `rearmAfterTurn` opens the next `/v2/listen` socket once the assistant turn finishes, and speech arriving during that handshake is held in the VAD pre-roll buffer, so its cost lands on the next utterance's first partial rather than on its commit.
120
+
121
+ ### What this A/B still does not hold constant
122
+
123
+ With Flux as the provider in both arms, the transcript **final** arrives at different moments. In arm B the `EndOfTurn` frame emits `final` immediately before `turn-end`, so the final is already in hand at commit. In arm A the local boundary fires first and release calls `stop()`, which sends `CloseStream` and waits for Flux to flush. So `sttMs` (the `utteranceEnd` to `finalTranscript` span) is not comparable between arms. It is not the headline number, but do not read a regression into it.
124
+
125
+ ## 4. Read the numbers
126
+
127
+ The daemon sends a `metrics` frame over the live-voice WebSocket on `turn_completed`, `turn_cancelled`, and `session_ended`. The endpoint fields are flattened onto that frame by `getLiveVoiceMetricsAggregateFields`:
128
+
129
+ | Field | Meaning |
130
+ | ------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------- |
131
+ | `endpointCommitLatencyMs` | **The headline.** Local VAD speech-stop mark to the commit. Present on every committed turn on both arms. Compare medians across arms. |
132
+ | `endpointDecisionSource` | `"provider"` or `"front-door"`. Present only when an endpoint decision was recorded. |
133
+ | `endpointDecisionMaxLatencyMs` | Worst single decision latency in the turn. A breakdown of the headline, never a cross-arm comparison. See the first section. |
134
+ | `endpointHoldCount` | Hold verdicts in the turn. Always 0 on a Flux-committed turn. |
135
+
136
+ The last three are **absent** unless a decision was recorded, which is deliberate: the absence is itself the signal, and it is what makes them useless as a cross-arm comparison. `endpointCommitLatencyMs` is absent only on a turn that never committed.
137
+
138
+ **How to see them.** The web client stores the whole frame but its `console.debug("[live-voice] turn latency", ...)` line does not print the endpoint fields. Read them from the raw frame instead: DevTools, Network, WS, select the live-voice socket, filter frames for `"type":"metrics"`, and read the `turn_completed` frames. The per-turn `metrics.recentTurns[]` array in the same frame carries the `timestamps` you need for the cross-check in the first section.
139
+
140
+ ### Reading an arm-B turn
141
+
142
+ | What you see | What happened |
143
+ | -------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------- |
144
+ | `endpointDecisionSource: "provider"` | Flux committed the turn. This is the measurement you came for. |
145
+ | No endpoint fields at all | The fail-open deadline fired, the utterance fell back to the silence path, and the front door released it without holding. Not a Flux sample. |
146
+ | `endpointDecisionSource: "front-door"` | Same fallback, but the front door then held. Not a Flux sample either. |
147
+
148
+ Fallbacks are logged at `warn`, so they are visible at the default log level:
149
+
150
+ ```
151
+ No provider end-of-turn is coming; falling back to the silence boundary for this utterance
152
+ ```
153
+
154
+ The deadline is `liveVoice.flux.eotTimeoutMs` plus a 1000ms margin (`PROVIDER_TURN_END_FALLBACK_MARGIN_MS`), measured from the local speech-stop mark. If you see these routinely, you are measuring the fallback path, not Flux.
155
+
156
+ #### Lower `eotTimeoutMs` for the run
157
+
158
+ With the shipped defaults the fail-open budget is `eotTimeoutMs` (**5000**) plus `PROVIDER_TURN_END_FALLBACK_MARGIN_MS` (**1000**), so **6000ms from the local speech-stop mark**. The local silence boundary fires at ~1200ms and then hands the turn to Flux, so a stalled turn is roughly **4.8 seconds of dead air** before the utterance replays onto the hold path and anyone answers.
159
+
160
+ That is the worst case by design (the budget has to clear Flux's own force-end), but it is a bad property to carry through a measurement run. It makes a stall expensive to sit through, which biases you toward not reproducing one, and it means arm B's slow turns are dominated by a timeout rather than by turn detection.
161
+
162
+ Set `liveVoice.flux.eotTimeoutMs` to **1500 to 2000** for the duration of the run. The budget drops to 2500 to 3000ms, a stall costs about 1.3 to 1.8 seconds past the silence boundary instead of 4.8, and arm B stays representative of what Flux does rather than of what the deadline does. Put it back before drawing any conclusion about the shipped configuration: it also changes how long Flux itself waits before force-ending a turn it never got confident about.
163
+
164
+ ### Prefer medians, and do not chase a single bad sample
165
+
166
+ `isStaleProviderTurnEnd` decides on Flux's own `turn_index`, which `parseFluxFrame` carries onto `turn-start` and `turn-end`. The cycle records the newest turn Flux opened, so a delayed end-of-turn for an older index is recognized as stale and dropped, even though the caller has resumed speaking since. An end-of-turn for the turn still in progress is never stale: the mid-thought pause is what Flux's turn model exists to judge, its verdict covers the resumed speech, and the newest speech-stop mark is the right anchor for it.
167
+
168
+ The local VAD generation counter is only the fallback, for an event that carries no turn number. There the outlier mode survives: if the caller resumes and stops again before a delayed `turn-end` lands, the second boundary re-stamps `turnBoundaryGeneration` to the current generation, the event stops looking stale, and it is accepted against the newer speech-stop mark. The **commit is still correct**, but the recorded latency is understated for that turn, in `endpointCommitLatencyMs` and `endpointDecisionMaxLatencyMs` alike, since they share the anchor.
169
+
170
+ Drops log at `info` and so are visible at the default log level, carrying `turnIndex`, `openTurnIndex`, `boundaryGeneration`, and `speechGeneration`:
171
+
172
+ ```
173
+ Dropping a stale provider end-of-turn: the caller resumed speaking past the boundary it closed
174
+ ```
175
+
176
+ A drop is not itself an error: the cycle stays open and still commits, on the end-of-turn for the resumed speech or on the fail-open deadline. Read the line's fields. A drop with both `turnIndex` and `openTurnIndex` present is the turn-index path working, and it also drops the silent variant of the same race. A drop with `turnIndex` absent means Deepgram is sending unnumbered events and the run is on the generation fallback, where individual samples can be skewed and legitimate fast end-of-turns are dropped conservatively. Either way, report medians over a run of turns and ignore individual extremes.
177
+
178
+ ## 5. Confirm the session is really running Flux
179
+
180
+ The dialed URL is logged at `info` on every session open, and it carries the clamped query parameters. The API key travels in an `Authorization` header, never in the URL, so the log line is safe to read and paste.
181
+
182
+ ```
183
+ grep "Opening Deepgram Flux session" ~/.vellum/workspace/data/logs/assistant-*.log
184
+ ```
185
+
186
+ Check the URL is `/v2/listen` and contains `model=flux-general-en`, `eot_threshold`, `eot_timeout_ms`, and **no** `eager_eot_threshold`. That is the cheapest confirmation that the tuning you wrote is the tuning in force.
187
+
188
+ Deepgram's authoritative thresholds would come back on a `ConfigureSuccess` frame, but the adapter never sends `Configure` and the parser has no case for the response, so nothing echoes the tuning back. The dialed URL is the confirmation you have.
189
+
190
+ ### Log levels
191
+
192
+ Every diagnostic this runbook tells you to read is at `info` or `warn`, and the pino level is `info` (`src/util/logger.ts`), so nothing here needs a source edit or a rebuild. That includes the stale-turn-end drop in section 4 and the chunk-cadence line in section 6.
193
+
194
+ One Flux line does sit at `debug`: `buildFluxQueryParams` reports clamping `eager_eot_threshold` down to the effective `eot_threshold`. It fires only if you set `eagerEotThreshold`, which this spike says to leave unset.
195
+
196
+ ## 6. Chunk cadence
197
+
198
+ Deepgram recommends 80ms audio chunks for Flux. Capture cadence is a client concern and this spike does not change it, but the adapter makes it measurable from the daemon side: once per session, on the first audio frame, it logs at `info`
199
+
200
+ ```
201
+ Deepgram Flux audio chunk cadence
202
+ ```
203
+
204
+ with `byteLength`, `sampleRate`, `encoding`, `observedChunkMs`, and `recommendedChunkMs: 80`. `observedChunkMs` is derived from byte length and the negotiated sample rate and is only present for `linear16`, which is the default encoding. Read it before anyone proposes tuning the client; if it is already near 80 there is nothing to win there.
205
+
206
+ ## 7. Known gaps
207
+
208
+ Do not rediscover these.
209
+
210
+ - **English only.** The spike pins `flux-general-en`. No language is forwarded to the adapter, because `language_hint` means nothing to a monolingual model. The catalog entry carries `languageSelection: "auto"`, so the Settings speech-to-text card renders no language picker for Flux. Read that as "no picker", not as detection: audio in another language transcribes as English rather than being detected.
211
+ - **No telephony, and no fallback either.** The catalog entry sets `telephonyMode: "none"`, so `resolveTelephonySttCapability` reports Flux as unsupported and the call session reports that as its error. Nothing reroutes the call to the `deepgram` provider: with Flux configured, calls on this assistant are not transcribed at all.
212
+ - **No batch, workspace-wide.** `supportedBoundaries` is `daemon-streaming` only, and `resolveBatchTranscriber` throws for it rather than returning the `null` that every batch caller reports as "no speech-to-text provider is configured". Callers surface the thrown message instead: `Deepgram Flux is streaming-only. Batch transcription requires the deepgram provider: set services.stt.provider to "deepgram".` See the warning in section 2 for the full list of surfaces this takes down.
213
+ - **No managed / velay path.** This is BYOK through the daemon only. The relay pins the model server-side, so managed rollout is a relay change.
214
+ - **Eager end-of-turn is off, and its being off is load-bearing.** `eagerEotThreshold` is optional with no default, and leaving it unset is precisely what stops Deepgram emitting `EagerEndOfTurn` / `TurnResumed` at all. The parser handles both frames and the session no-ops them, so the follow-up is small, but Deepgram warns that enabling speculation raises LLM calls by 50 to 70 percent. Note also that `buildFluxQueryParams` clamps `eager_eot_threshold` **down** to the effective `eot_threshold`, because Deepgram rejects the inverse combination.
215
+ - **The hold machinery is present and unchanged.** `HOLD_VERDICT_TOKEN`, the `includeHold` branch, the speculative dispatch and rollback state, `endpointExtensionMs`, and `endpointMaxExtensions` are all still there. The latch only skips them.
216
+ - **Local VAD still owns barge-in in both modes,** and must. A local energy gate on audio already in hand beats a provider roundtrip for an interrupt during playback, and the echo-adaptive part of the guard has to stay upstream of Flux, which hears only the microphone and cannot tell our own TTS bleeding through imperfect echo cancellation from a real caller turn.
217
+ - **Self-hosted bundles and `KeepAlive`.** The adapter sends a `KeepAlive` control frame every 5s, which cloud `/v2/listen` accepts and which is the only thing that resets Deepgram's server-side inactivity timer during silence. Self-hosted SageMaker Flux bundles reject it as a fatal `UNPARSABLE_CLIENT_MESSAGE`. The escape hatch is the adapter's `keepaliveIntervalMs: 0` option, but it is a constructor option only: `resolve.ts` passes just `sampleRate`, so pointing the adapter at a self-hosted bundle means editing that call site. There is no config key for it.
218
+
219
+ ### Two invariants the measurement rests on
220
+
221
+ Check these if you port the spike anywhere else, because a run that violates either produces numbers that look plausible and are not.
222
+
223
+ - **The latch is up before the session's first boundary.** `start()` sends `ready` without waiting on the transcriber resolve, so the caller can speak and close a whole silence boundary while the handshake is in flight. `beginUtterance` seeds the latch from the **configured** provider before the dial and reconciles it against the resolved provider afterwards, so the opening turn is a Flux turn like every other one rather than a front-door turn inside a session that looks like a Flux session.
224
+ - **Every recorded latency is anchored at a real speech-stop.** `localSpeechStopAtMs` is stamped on every above-gate chunk in every server-VAD session, and `markEndpointCommit` records nothing at all when there is no mark, which is push-to-talk. `msSinceLocalSpeechStop()` therefore never reports a turn measured from zero, and both arms share the anchor.
225
+
226
+ ## 8. Falling back
227
+
228
+ Set `liveVoice.flux.turnEnd.enabled` back to `false` (or delete the key) and restart. That is the whole rollback: the latch goes down, the turn-detection events become no-ops, and the silence-boundary path runs unchanged. You can leave `services.stt.provider` on `deepgram-flux` or move it back to `deepgram`; either is a working configuration.
229
+
230
+ Runtime fallback needs no action. A Flux stream that never emits `turn-end` is caught by the fail-open deadline and the utterance replays onto the silence path, so an outage degrades to today's behavior rather than to a hung turn.
231
+
232
+ ## 9. What would justify deleting the hold path
233
+
234
+ The follow-up cleanup is not small: `HOLD_VERDICT_TOKEN`, the `includeHold` branch of `frontDoorDecisionRule`, the `holdEnabled` parameter on `classifyFrontDoorLeading`, six pieces of speculative-dispatch state in the session, and the `endpointExtensionMs` / `endpointMaxExtensions` config surface. It is worth doing only if Flux clears a real bar. Deepgram claims Flux decides in under 400ms; that claim is the hypothesis under test here, not an input.
235
+
236
+ Suggested bar, all measured on `endpointCommitLatencyMs` over enough turns for a median to mean something. That field deliberately excludes the per-turn socket teardown, which keeps the arms comparable but also means a bar cleared here is not by itself a case for running Flux in front of users: the teardown is a cost Flux keeps paying and `deepgram` does not.
237
+
238
+ 1. **Median `endpointCommitLatencyMs` in arm B beats the median in arm A.** Most arm-A turns are unheld, so that median sits near `silenceThresholdMs`, about 1200ms with the defaults. Do not set the bar against arm A's held turns: beating those is easy and proves nothing, because most turns are not held.
239
+ 2. **No regression on thinking pauses.** The hold path exists so a mid-sentence pause does not trigger a premature reply. Count premature commits on the scripted pause utterances in both arms. Flux has to be at least as good, not merely faster. A fast path that interrupts people is worse than the slow one.
240
+ 3. **The fallback is rare.** If `warn`-level fallback lines appear on a meaningful fraction of turns, the hold path is not dead code, it is the live path, and deleting it removes the thing keeping the feature usable.
241
+ 4. **Escalate and the spoken ack / progress phrasing still behave.** Flux bypasses only the `[0]` hold branch. Confirm `[1]` escalate still fires and acks still land before concluding the front door has nothing left to do.
242
+
243
+ If 1 and 2 both hold and 3 is clean, the cleanup is justified. If 1 holds but 2 does not, the answer is to tune `eotThreshold` upward and re-measure, not to delete anything.
@@ -13,6 +13,7 @@ Add a new entry to the `CATALOG` map with:
13
13
  - `supportedBoundaries` — the set of `SttBoundaryId` values the provider supports. Valid values are `"daemon-batch"` (post-recording transcription) and `"daemon-streaming"` (real-time streaming transcription during conversation).
14
14
  - `conversationStreamingMode` — how the provider handles streaming transcription in conversation mode: `"realtime-ws"` (provider supports real-time streaming natively via WebSocket), `"incremental-batch"` (streaming emulated via throttled polling), or `"none"` (no streaming support). Required for all providers.
15
15
  - `telephonyMode` — how the provider participates in real-time telephony STT: `"realtime-ws"`, `"batch-only"`, or `"none"`. The telephony capability resolver (`resolveTelephonySttCapability()` in `src/providers/speech-to-text/resolve.ts`) reads this field plus credential availability to decide whether phone calls can run with the provider.
16
+ - `turnDetection`: whether the provider decides end-of-turn itself, `"provider"` or `"none"`. Use `"provider"` only when the adapter emits `turn-start` / `turn-end` on its transcript stream; a live-voice session reads this via `supportsProviderTurnDetection()` to decide whether to arm its provider turn-end path. Default to `"none"`: a provider that declares `"provider"` but never emits the events makes every turn wait out the fail-open deadline before falling back to the silence boundary. A provider declaring `"provider"` should also number its turns, because the staleness check prefers the turn index on the event and degrades to the session's VAD generation counter without one.
16
17
 
17
18
  ## 2. Type-system registration
18
19
 
@@ -71,10 +72,11 @@ Native clients fetch this metadata at launch via `GET /v1/stt/providers`. No sep
71
72
  | ---------------- | -------------------- | ------------- |
72
73
  | `openai-whisper` | `openai` | shared |
73
74
  | `deepgram` | `deepgram` | exclusive |
75
+ | `deepgram-flux` | `deepgram` | shared |
74
76
  | `google-gemini` | `gemini` | shared |
75
77
  | `xai` | `xai` | exclusive |
76
78
 
77
- When the provider ID differs from the credential provider name (e.g. `google-gemini` maps to `gemini`), the key is **shared** with other services that use the same credential.
79
+ When the provider ID differs from the credential provider name (e.g. `google-gemini` maps to `gemini`), the key is **shared** with other services that use the same credential. Two STT providers may also name the same credential: `deepgram-flux` is a model on the same Deepgram account as `deepgram`, so it reads that key rather than introducing one of its own. Reuse an existing `credentialProvider` whenever the new provider authenticates against an account the catalog already covers, and keep the `credentialsGuide` text identical across them (`DEEPGRAM_CREDENTIALS_GUIDE` is shared by both entries for that reason).
78
80
 
79
81
  ### Client settings key behavior
80
82
 
@@ -6,6 +6,16 @@
6
6
  * assistant through (Slack, Telegram, WhatsApp, phone, …) plus a couple of
7
7
  * internal ids (`vellum` for native app conversations, `platform` for the
8
8
  * internal control plane). This is the single source of truth for that set:
9
+ *
10
+ * One id, `plugin`, does not name a surface: it names *every* surface a plugin
11
+ * brings. A plugin channel's real identity is the plugin, which is workspace
12
+ * state and cannot be a compile-time union member, so the plugin name travels
13
+ * in `sourceMetadata.plugin` and is prefixed onto every external id the gateway
14
+ * forwards (`imessage:+15551234567`). Two plugins therefore share a channel
15
+ * row — one admission floor, one set of channel-wide defaults — while their
16
+ * conversations, contacts, and trust records stay disjoint. See
17
+ * `gateway/src/channels/plugin-inbound.ts` for what that concedes.
18
+ *
9
19
  * the assistant adopts it wholesale as its `ChannelId`, and the gateway
10
20
  * asserts its own (narrower) inbound list is a subset of it so the two sides
11
21
  * cannot silently drift.
@@ -30,6 +40,7 @@ export const CHANNEL_IDS = [
30
40
  "platform",
31
41
  "a2a",
32
42
  "discord",
43
+ "plugin",
33
44
  ] as const;
34
45
 
35
46
  export type ChannelId = (typeof CHANNEL_IDS)[number];
@@ -6,6 +6,16 @@
6
6
  * assistant through (Slack, Telegram, WhatsApp, phone, …) plus a couple of
7
7
  * internal ids (`vellum` for native app conversations, `platform` for the
8
8
  * internal control plane). This is the single source of truth for that set:
9
+ *
10
+ * One id, `plugin`, does not name a surface: it names *every* surface a plugin
11
+ * brings. A plugin channel's real identity is the plugin, which is workspace
12
+ * state and cannot be a compile-time union member, so the plugin name travels
13
+ * in `sourceMetadata.plugin` and is prefixed onto every external id the gateway
14
+ * forwards (`imessage:+15551234567`). Two plugins therefore share a channel
15
+ * row — one admission floor, one set of channel-wide defaults — while their
16
+ * conversations, contacts, and trust records stay disjoint. See
17
+ * `gateway/src/channels/plugin-inbound.ts` for what that concedes.
18
+ *
9
19
  * the assistant adopts it wholesale as its `ChannelId`, and the gateway
10
20
  * asserts its own (narrower) inbound list is a subset of it so the two sides
11
21
  * cannot silently drift.
@@ -30,6 +40,7 @@ export const CHANNEL_IDS = [
30
40
  "platform",
31
41
  "a2a",
32
42
  "discord",
43
+ "plugin",
33
44
  ] as const;
34
45
 
35
46
  export type ChannelId = (typeof CHANNEL_IDS)[number];
@@ -10,6 +10,8 @@
10
10
 
11
11
  import { z } from "zod";
12
12
 
13
+ import type { TrustClass } from "./trust-verdict-contract.js";
14
+
13
15
  /**
14
16
  * Per-channel inbound admission policy — ordered from most-restrictive
15
17
  * (`no_one`, hard kill switch) to most-permissive (`strangers`, admits any
@@ -101,3 +103,35 @@ export function isAdmissionPolicy(value: unknown): value is AdmissionPolicy {
101
103
  (ADMISSION_POLICY_VALUES as readonly string[]).includes(value)
102
104
  );
103
105
  }
106
+
107
+ /**
108
+ * Trust-class ordinal compared against {@link ADMISSION_FLOOR} to make the
109
+ * admission decision. Higher rank = more trusted. Blocked and revoked members
110
+ * never reach this comparison, short-circuiting to deny on member status, so
111
+ * they carry no rank.
112
+ */
113
+ export const TRUST_CLASS_RANK: Record<TrustClass, number> = {
114
+ guardian: 4,
115
+ trusted_contact: 3,
116
+ unverified_contact: 2,
117
+ unknown: 1,
118
+ };
119
+
120
+ /**
121
+ * Whether a sender of this trust class clears a channel's admission floor.
122
+ *
123
+ * The two halves of the check live together because they are meaningless
124
+ * apart: this compares a table keyed by {@link TrustClass} against one keyed
125
+ * by {@link AdmissionPolicy}, and a floor added to one without a rank in the
126
+ * other silently admits or denies everyone.
127
+ *
128
+ * Both enforcement points read this. The runtime's admission stage answers for
129
+ * every channel it receives; the gateway answers for a channel it delivers
130
+ * somewhere other than the runtime, where there is no later stage to ask.
131
+ */
132
+ export function meetsAdmissionFloor(
133
+ policy: AdmissionPolicy,
134
+ trustClass: TrustClass,
135
+ ): boolean {
136
+ return TRUST_CLASS_RANK[trustClass] >= ADMISSION_FLOOR[policy];
137
+ }
@@ -71,6 +71,8 @@ export {
71
71
  isAdmissionPolicy,
72
72
  isAdmissionPolicyExemptChannel,
73
73
  isAdmissionPolicyHiddenChannel,
74
+ meetsAdmissionFloor,
75
+ TRUST_CLASS_RANK,
74
76
  } from "./admission-policy-contract.js";
75
77
 
76
78
  export type { AdmissionPolicy } from "./admission-policy-contract.js";
@@ -6,6 +6,16 @@
6
6
  * assistant through (Slack, Telegram, WhatsApp, phone, …) plus a couple of
7
7
  * internal ids (`vellum` for native app conversations, `platform` for the
8
8
  * internal control plane). This is the single source of truth for that set:
9
+ *
10
+ * One id, `plugin`, does not name a surface: it names *every* surface a plugin
11
+ * brings. A plugin channel's real identity is the plugin, which is workspace
12
+ * state and cannot be a compile-time union member, so the plugin name travels
13
+ * in `sourceMetadata.plugin` and is prefixed onto every external id the gateway
14
+ * forwards (`imessage:+15551234567`). Two plugins therefore share a channel
15
+ * row — one admission floor, one set of channel-wide defaults — while their
16
+ * conversations, contacts, and trust records stay disjoint. See
17
+ * `gateway/src/channels/plugin-inbound.ts` for what that concedes.
18
+ *
9
19
  * the assistant adopts it wholesale as its `ChannelId`, and the gateway
10
20
  * asserts its own (narrower) inbound list is a subset of it so the two sides
11
21
  * cannot silently drift.
@@ -30,6 +40,7 @@ export const CHANNEL_IDS = [
30
40
  "platform",
31
41
  "a2a",
32
42
  "discord",
43
+ "plugin",
33
44
  ] as const;
34
45
 
35
46
  export type ChannelId = (typeof CHANNEL_IDS)[number];