@vellumai/assistant 0.11.4-staging.2 → 0.11.4-staging.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (189) hide show
  1. package/AGENTS.md +8 -2
  2. package/ARCHITECTURE.md +2 -0
  3. package/docs/architecture/memory.md +15 -0
  4. package/docs/browser-use-architecture-phase2.md +128 -56
  5. package/docs/flux-turn-detection-spike.md +11 -6
  6. package/knip.json +3 -0
  7. package/openapi.yaml +136 -76
  8. package/package.json +1 -1
  9. package/scripts/write-plugin-api-shim.ts +10 -0
  10. package/src/__tests__/app-control-flow.test.ts +1 -0
  11. package/src/__tests__/approval-routes-http.test.ts +2 -2
  12. package/src/__tests__/assistant-feature-flag-guard.test.ts +25 -3
  13. package/src/__tests__/channel-setup-panel-ack.test.ts +1 -1
  14. package/src/__tests__/compaction-events.test.ts +8 -10
  15. package/src/__tests__/conversation-agent-loop.test.ts +4 -1
  16. package/src/__tests__/conversation-confirmation-signals.test.ts +112 -0
  17. package/src/__tests__/conversation-notifiers-provenance.test.ts +1 -1
  18. package/src/__tests__/conversation-queue.test.ts +39 -62
  19. package/src/__tests__/conversation-routes-disk-view.test.ts +1 -1
  20. package/src/__tests__/conversation-routes-enabled-plugins.test.ts +1 -1
  21. package/src/__tests__/conversation-routes-guardian-reply.test.ts +9 -9
  22. package/src/__tests__/conversation-routes-hidden-queue.test.ts +1 -1
  23. package/src/__tests__/conversation-routes-slash-commands.test.ts +1 -1
  24. package/src/__tests__/conversation-slash-queue.test.ts +3 -0
  25. package/src/__tests__/conversation-surfaces-action-delivery.test.ts +1 -0
  26. package/src/__tests__/conversation-surfaces-activation-emit.test.ts +1 -0
  27. package/src/__tests__/conversation-surfaces-app-control.test.ts +1 -0
  28. package/src/__tests__/conversation-surfaces-app-open.test.ts +1 -1
  29. package/src/__tests__/conversation-surfaces-data-persist.test.ts +1 -1
  30. package/src/__tests__/conversation-surfaces-history-restored-completion.test.ts +21 -14
  31. package/src/__tests__/conversation-surfaces-queued-emit.test.ts +1 -0
  32. package/src/__tests__/conversation-surfaces-standalone-payloads.test.ts +1 -0
  33. package/src/__tests__/conversation-surfaces-standalone.test.ts +1 -0
  34. package/src/__tests__/conversation-surfaces-state-update.test.ts +1 -1
  35. package/src/__tests__/conversation-surfaces-table-action.test.ts +1 -1
  36. package/src/__tests__/conversation-surfaces-task-progress.test.ts +1 -1
  37. package/src/__tests__/conversation-tool-setup-app-refresh.test.ts +1 -1
  38. package/src/__tests__/conversation-tool-setup-attribution.test.ts +1 -1
  39. package/src/__tests__/cu-unified-flow.test.ts +1 -0
  40. package/src/__tests__/document-sync-tags.test.ts +0 -75
  41. package/src/__tests__/gateway-only-guard.test.ts +2 -5
  42. package/src/__tests__/http-user-message-parity.test.ts +1 -1
  43. package/src/__tests__/init-feature-flag-overrides.test.ts +49 -0
  44. package/src/__tests__/managed-skill-lifecycle.test.ts +7 -0
  45. package/src/__tests__/media-generate-image.test.ts +131 -21
  46. package/src/__tests__/memory-retrieval-hook.test.ts +94 -2
  47. package/src/__tests__/plugin-api-webhook-url.test.ts +10 -7
  48. package/src/__tests__/plugin-import-boundary-reverse-guard.test.ts +7 -3
  49. package/src/__tests__/proxy-approval-callback.test.ts +1 -0
  50. package/src/__tests__/qdrant-manager.test.ts +14 -1
  51. package/src/__tests__/run-due-schedules.test.ts +21 -0
  52. package/src/__tests__/scaffold-managed-skill-tool.test.ts +187 -18
  53. package/src/__tests__/schedule-routes.test.ts +23 -0
  54. package/src/__tests__/schedule-store.test.ts +17 -0
  55. package/src/__tests__/secret-ingress-http.test.ts +1 -1
  56. package/src/__tests__/send-endpoint-busy.test.ts +3 -3
  57. package/src/__tests__/starter-task-flow.test.ts +5 -4
  58. package/src/__tests__/subagent-fork-prompt-role.test.ts +1 -1
  59. package/src/__tests__/subagent-spawn-and-await.test.ts +4 -7
  60. package/src/__tests__/subagent-tool-gate-mode.test.ts +1 -1
  61. package/src/__tests__/subagent-tools.test.ts +81 -101
  62. package/src/__tests__/surface-completion-in-flight-snapshot.test.ts +1 -0
  63. package/src/__tests__/ui-choice-copy-surfaces.test.ts +1 -1
  64. package/src/__tests__/ui-visual-surface.test.ts +1 -1
  65. package/src/__tests__/ui-voice-picker-surface.test.ts +1 -1
  66. package/src/__tests__/ui-work-result-surface.test.ts +1 -1
  67. package/src/__tests__/voice-scoped-grant-consumer.test.ts +5 -3
  68. package/src/__tests__/voice-session-bridge.test.ts +85 -29
  69. package/src/acp/session-manager.ts +8 -1
  70. package/src/api/surfaces.ts +5 -0
  71. package/src/calls/__tests__/voice-session-bridge.test.ts +21 -10
  72. package/src/calls/__tests__/voice-triage-escalate.test.ts +8 -0
  73. package/src/calls/voice-session-bridge.ts +44 -25
  74. package/src/calls/voice-triage-escalate.ts +1 -0
  75. package/src/cli/bundled-modules.ts +29 -0
  76. package/src/cli/commands/db/repair.ts +4 -8
  77. package/src/cli/commands/domain.ts +6 -3
  78. package/src/cli/commands/email.ts +6 -3
  79. package/src/cli/commands/keys.ts +8 -3
  80. package/src/cli/commands/plugins.ts +85 -36
  81. package/src/cli/commands/schedules.ts +35 -1
  82. package/src/cli/lib/bundled-marketplace.json +1 -1
  83. package/src/config/__tests__/balanced-model-experiment.test.ts +278 -0
  84. package/src/config/assistant-feature-flags.ts +36 -15
  85. package/src/config/balanced-model-experiment.ts +35 -0
  86. package/src/config/bundled-skills/image-studio/SKILL.md +5 -4
  87. package/src/config/bundled-skills/image-studio/TOOLS.json +1 -1
  88. package/src/config/bundled-skills/image-studio/tools/media-generate-image.ts +101 -0
  89. package/src/config/bundled-skills/skill-management/TOOLS.json +9 -3
  90. package/src/config/bundled-skills/subagent/SKILL.md +17 -12
  91. package/src/config/bundled-skills/subagent/TOOLS.json +4 -4
  92. package/src/config/call-site-defaults.ts +7 -0
  93. package/src/config/default-profile-catalog.ts +96 -4
  94. package/src/config/feature-flag-registry.json +11 -11
  95. package/src/config/skills.ts +9 -2
  96. package/src/daemon/__tests__/conversation-surfaces-launch.test.ts +1 -1
  97. package/src/daemon/conversation-agent-loop.ts +14 -13
  98. package/src/daemon/conversation-notifiers.ts +11 -9
  99. package/src/daemon/conversation-process.ts +0 -27
  100. package/src/daemon/conversation-store.ts +4 -4
  101. package/src/daemon/conversation-surfaces.ts +27 -13
  102. package/src/daemon/conversation-tool-setup.ts +2 -4
  103. package/src/daemon/conversation.ts +81 -44
  104. package/src/daemon/doordash-steps.ts +2 -2
  105. package/src/daemon/lifecycle.ts +14 -1
  106. package/src/daemon/process-message.ts +0 -13
  107. package/src/daemon/windows-compiled-entry.ts +4 -0
  108. package/src/documents/document-store.ts +5 -235
  109. package/src/hooks/types.ts +5 -0
  110. package/src/ipc/gateway-flag-listener.ts +17 -3
  111. package/src/live-voice/__tests__/live-voice-flux-turn-end.test.ts +118 -0
  112. package/src/live-voice/live-voice-manager.ts +16 -3
  113. package/src/live-voice/live-voice-session.ts +63 -9
  114. package/src/live-voice/windows-compiled-live-voice.ts +4 -0
  115. package/src/monitoring/control.ts +1 -0
  116. package/src/monitoring/db-integrity-sample.ts +4 -5
  117. package/src/permissions/prompter.ts +1 -5
  118. package/src/persistence/conversation-queries.ts +66 -16
  119. package/src/persistence/embeddings/qdrant-manager.ts +84 -49
  120. package/src/persistence/migrations/360-add-document-workspace-path.ts +5 -14
  121. package/src/persistence/schema/documents.ts +4 -4
  122. package/src/plugin-api/constants.ts +8 -0
  123. package/src/plugin-api/index.ts +5 -1
  124. package/src/plugin-api/webhook-url.ts +13 -11
  125. package/src/plugins/defaults/main.ts +6 -7
  126. package/src/plugins/defaults/memory/graph/__tests__/conversation-graph-memory-v2-routing.test.ts +77 -0
  127. package/src/plugins/defaults/memory/graph/conversation-graph-memory.ts +24 -6
  128. package/src/plugins/defaults/memory/hooks/user-prompt-submit.ts +53 -5
  129. package/src/plugins/defaults/memory/memory-retrospective-job.ts +4 -4
  130. package/src/plugins/defaults/memory/v3/__tests__/injection.test.ts +18 -0
  131. package/src/plugins/defaults/memory/v3/__tests__/shadow-plugin.test.ts +66 -1
  132. package/src/plugins/defaults/memory/v3/injector.ts +8 -0
  133. package/src/plugins/defaults/memory/v3/shadow-plugin.ts +23 -9
  134. package/src/plugins/defaults/memory/worker-control.ts +1 -0
  135. package/src/plugins/defaults/worker-entrypoints.ts +3 -0
  136. package/src/plugins/mtime-cache.ts +17 -0
  137. package/src/prompts/templates/system-sections.ts +0 -7
  138. package/src/providers/__tests__/context-overflow-error.test.ts +24 -0
  139. package/src/providers/__tests__/retry-callsite.test.ts +20 -0
  140. package/src/providers/openai/chat-completions-provider.ts +11 -1
  141. package/src/providers/speech-to-text/__tests__/deepgram-flux-realtime.test.ts +11 -6
  142. package/src/providers/speech-to-text/deepgram-flux-realtime.ts +9 -49
  143. package/src/routes/control.ts +1 -0
  144. package/src/routes/route-host-client.ts +1 -0
  145. package/src/runtime/AGENTS.md +16 -17
  146. package/src/runtime/agent-wake.ts +15 -12
  147. package/src/runtime/routes/__tests__/conversation-list-routes.test.ts +170 -1
  148. package/src/runtime/routes/__tests__/schedule-routes-disarm-reason.test.ts +215 -0
  149. package/src/runtime/routes/conversation-list-routes.ts +54 -22
  150. package/src/runtime/routes/conversation-management-routes.ts +2 -3
  151. package/src/runtime/routes/conversation-routes.ts +11 -13
  152. package/src/runtime/routes/documents-routes.ts +3 -222
  153. package/src/runtime/routes/playground/__tests__/inject-failures.test.ts +2 -0
  154. package/src/runtime/routes/playground/__tests__/reset-circuit.test.ts +3 -0
  155. package/src/runtime/routes/playground/inject-failures.ts +2 -2
  156. package/src/runtime/routes/playground/reset-circuit.ts +1 -1
  157. package/src/runtime/routes/schedule-routes.ts +94 -7
  158. package/src/runtime/routes/workspace-routes.ts +0 -9
  159. package/src/runtime/routes/workspace-utils.ts +3 -13
  160. package/src/runtime/services/conversation-serializer.ts +7 -2
  161. package/src/schedule/__tests__/plugin-schedule-declarations.test.ts +68 -5
  162. package/src/schedule/__tests__/plugin-schedule-reconciler.test.ts +81 -0
  163. package/src/schedule/plugin-schedule-availability.ts +58 -0
  164. package/src/schedule/plugin-schedule-declarations.ts +23 -27
  165. package/src/schedule/plugin-schedule-reconciler.ts +12 -3
  166. package/src/schedule/schedule-store.ts +5 -1
  167. package/src/schedule/scheduler.ts +9 -4
  168. package/src/schedule/worker-control.ts +1 -0
  169. package/src/subagent/__tests__/consult-prompt.test.ts +26 -15
  170. package/src/subagent/consult-context.ts +11 -11
  171. package/src/subagent/consult-prompt.ts +26 -35
  172. package/src/subagent/manager.ts +20 -37
  173. package/src/subagent/notify.ts +7 -1
  174. package/src/subagent/types.ts +8 -6
  175. package/src/tools/acp/spawn.ts +6 -4
  176. package/src/tools/skills/scaffold-managed.ts +25 -7
  177. package/src/tools/subagent/spawn.ts +28 -88
  178. package/src/tools/ui-surface/surface-shape-docs.ts +1 -1
  179. package/src/util/__tests__/worker-process-command.test.ts +37 -0
  180. package/src/util/logger.ts +16 -0
  181. package/src/util/worker-process.ts +37 -4
  182. package/src/windows-compiled-cli.ts +32 -0
  183. package/src/windows-compiled-entry.ts +4 -0
  184. package/src/windows-compiled-logger.ts +6 -0
  185. package/src/windows-compiled-worker-entry.ts +29 -0
  186. package/src/__tests__/document-workspace-file.test.ts +0 -467
  187. package/src/daemon/interactive-turn-sender.ts +0 -59
  188. package/src/subagent/__tests__/consult-transcript.test.ts +0 -184
  189. package/src/subagent/consult-transcript.ts +0 -90
@@ -512,6 +512,10 @@ export class ConversationGraphMemory {
512
512
  * - Turn 1 (or after compaction): full context load
513
513
  * - Every other turn: per-turn injection
514
514
  *
515
+ * `routingMessages` may end before an internal continuation message. The
516
+ * retrievers derive their query from that view, while memory is still
517
+ * injected into `messages`, which is the history sent to the model.
518
+ *
515
519
  * Returns augmented messages with memory context prepended to the last
516
520
  * user message, following the same injection pattern as the old system.
517
521
  */
@@ -520,6 +524,7 @@ export class ConversationGraphMemory {
520
524
  config: AssistantConfig,
521
525
  abortSignal: AbortSignal,
522
526
  onEvent: (msg: AssistantEvent) => void,
527
+ routingMessages: Message[] = messages,
523
528
  ): Promise<{
524
529
  runMessages: Message[];
525
530
  injectedTokens: number;
@@ -579,7 +584,7 @@ export class ConversationGraphMemory {
579
584
 
580
585
  // Gate: skip for empty/tool-result-only messages — unless we need to
581
586
  // reload after compaction (needsReload) or haven't initialized yet.
582
- const lastMessage = messages[messages.length - 1];
587
+ const lastMessage = routingMessages[routingMessages.length - 1];
583
588
  if (!lastMessage || lastMessage.role !== "user") {
584
589
  return noopResult;
585
590
  }
@@ -598,6 +603,7 @@ export class ConversationGraphMemory {
598
603
 
599
604
  return await this.runContextLoad(
600
605
  messages,
606
+ routingMessages,
601
607
  config,
602
608
  recentSummaries,
603
609
  firstUserText ?? undefined,
@@ -606,7 +612,12 @@ export class ConversationGraphMemory {
606
612
  );
607
613
  }
608
614
 
609
- return await this.runPerTurn(messages, config, abortSignal);
615
+ return await this.runPerTurn(
616
+ messages,
617
+ routingMessages,
618
+ config,
619
+ abortSignal,
620
+ );
610
621
  } catch (err) {
611
622
  const errCode =
612
623
  err instanceof z.ZodError ? err.issues[0]?.code : undefined;
@@ -629,6 +640,7 @@ export class ConversationGraphMemory {
629
640
 
630
641
  private async runContextLoad(
631
642
  messages: Message[],
643
+ routingMessages: Message[],
632
644
  config: AssistantConfig,
633
645
  recentSummaries: string[],
634
646
  userQuery: string | undefined,
@@ -640,9 +652,12 @@ export class ConversationGraphMemory {
640
652
  // The activation pipeline is robust to weak ANN signal — it falls back
641
653
  // to spreading + nowText to surface candidates.
642
654
  const startedAt = Date.now();
643
- const rawUserText = readRawUserText(messages[messages.length - 1]);
655
+ const rawUserText = readRawUserText(
656
+ routingMessages[routingMessages.length - 1],
657
+ );
644
658
  const v2 = await this.maybeRouteV2Injection(
645
659
  messages,
660
+ routingMessages,
646
661
  config,
647
662
  "context-load",
648
663
  // Context-load runs before the messages array necessarily contains
@@ -780,6 +795,7 @@ export class ConversationGraphMemory {
780
795
 
781
796
  private async runPerTurn(
782
797
  messages: Message[],
798
+ routingMessages: Message[],
783
799
  config: AssistantConfig,
784
800
  signal: AbortSignal,
785
801
  ) {
@@ -788,8 +804,8 @@ export class ConversationGraphMemory {
788
804
  let userLast = "";
789
805
  let userLastBlocks: ContentBlock[] = [];
790
806
 
791
- for (let i = messages.length - 1; i >= 0; i--) {
792
- const msg = messages[i];
807
+ for (let i = routingMessages.length - 1; i >= 0; i--) {
808
+ const msg = routingMessages[i];
793
809
  const text = msg.content
794
810
  .filter(
795
811
  (b): b is Extract<typeof b, { type: "text" }> => b.type === "text",
@@ -815,6 +831,7 @@ export class ConversationGraphMemory {
815
831
  const startedAt = Date.now();
816
832
  const v2 = await this.maybeRouteV2Injection(
817
833
  messages,
834
+ routingMessages,
818
835
  config,
819
836
  "per-turn",
820
837
  null,
@@ -978,6 +995,7 @@ export class ConversationGraphMemory {
978
995
  */
979
996
  private async maybeRouteV2Injection(
980
997
  messages: Message[],
998
+ routingMessages: Message[],
981
999
  config: AssistantConfig,
982
1000
  mode: InjectMemoryV2Mode,
983
1001
  /**
@@ -1004,7 +1022,7 @@ export class ConversationGraphMemory {
1004
1022
  const recentTurnPairs =
1005
1023
  userMessageOverride !== null
1006
1024
  ? [{ assistantMessage: "", userMessage: userMessageOverride }]
1007
- : extractRecentTurnPairs(messages, historicalPairs);
1025
+ : extractRecentTurnPairs(routingMessages, historicalPairs);
1008
1026
 
1009
1027
  const result = await injectMemoryV2Block({
1010
1028
  conversationId: this.conversationId,
@@ -31,7 +31,10 @@ import type {
31
31
  HookFunction,
32
32
  UserPromptSubmitContext,
33
33
  } from "@vellumai/plugin-api";
34
- import { updateMessageMetadata } from "@vellumai/plugin-api";
34
+ import {
35
+ updateMessageMetadata,
36
+ VOICE_ESCALATION_CONTINUATION_MESSAGE_KIND,
37
+ } from "@vellumai/plugin-api";
35
38
 
36
39
  import type { MemoryRecalledEvent } from "../../../../api/events/memory-recalled.js";
37
40
  import { getConfig } from "../../../../config/loader.js";
@@ -48,6 +51,7 @@ import { timeLatencySubSpan } from "../../../../daemon/turn-latency-sub-spans.js
48
51
  import { broadcastMessage } from "../../../../runtime/assistant-event-hub.js";
49
52
  import type { GraphMemoryResult } from "../graph/conversation-graph-memory.js";
50
53
  import { recordMemoryRecallLog } from "../memory-recall-log-store.js";
54
+ import { stripTailInjectionsForReinjection } from "../tail-reinjection-strip.js";
51
55
  import { MEMORY_V3_INJECTED_BLOCK_METADATA_KEY } from "../v3/ever-injected-store.js";
52
56
 
53
57
  /**
@@ -69,6 +73,39 @@ export function shouldRunLegacyMemoryRetrieval(params: {
69
73
  return params.isTrustedActor && !params.memoryV3Live;
70
74
  }
71
75
 
76
+ /**
77
+ * The voice escalation continuation is a model-control message, not a
78
+ * retrieval query. Route memory on the latest preceding user message and strip
79
+ * the runtime context that was frozen onto it during its original turn.
80
+ */
81
+ function legacyRetrievalRoutingMessages(
82
+ messages: UserPromptSubmitContext["latestMessages"],
83
+ routeOnPreviousUser: boolean,
84
+ ): UserPromptSubmitContext["latestMessages"] {
85
+ if (!routeOnPreviousUser) {
86
+ return messages;
87
+ }
88
+
89
+ for (let i = messages.length - 2; i >= 0; i -= 1) {
90
+ if (messages[i]?.role !== "user") {
91
+ continue;
92
+ }
93
+ const candidate = stripTailInjectionsForReinjection(
94
+ messages.slice(0, i + 1),
95
+ );
96
+ const tail = candidate[candidate.length - 1];
97
+ if (
98
+ tail?.role === "user" &&
99
+ tail.content.some(
100
+ (block) => block.type === "text" && block.text.trim().length > 0,
101
+ )
102
+ ) {
103
+ return candidate;
104
+ }
105
+ }
106
+ return [];
107
+ }
108
+
72
109
  /**
73
110
  * Persist and broadcast the retrieval's side effects: the injected block on
74
111
  * the user message's metadata (so it survives reloads), a recall-log row, and
@@ -263,10 +300,11 @@ async function persistInjectionBlocks(
263
300
  * runs for every actor, writing the fully injected result back onto
264
301
  * `latestMessages` and persisting the assembled blocks.
265
302
  *
266
- * Memory retrieval blocks the turn — there is no soft timeout here. Memory is
267
- * critical context, and silently dropping it produces a worse outcome than a
268
- * slower turn. Cancellation still works via `ctx.signal`, which is threaded
269
- * into `prepareMemory`.
303
+ * Memory retrieval ordinarily blocks the turn. The voice front door is the
304
+ * latency-sensitive exception: it keeps carried memory in the front prompt
305
+ * and defers current-turn retrieval to the escalated leg when one is needed.
306
+ * Cancellation still works via the live conversation signal for ordinary
307
+ * retrieval.
270
308
  */
271
309
  const userPromptSubmitMemoryRetrieval: HookFunction<
272
310
  UserPromptSubmitContext
@@ -292,9 +330,14 @@ const userPromptSubmitMemoryRetrieval: HookFunction<
292
330
  // fallback — a v3 empty/failed selection yields no NEW injected memory that
293
331
  // turn (prior turns' frozen v3 cards still ride history).
294
332
  const memoryV3Live = isMemoryV3Live(config);
333
+ const isVoiceFrontDoor = conversation?.currentCallSite === "voiceFrontDoor";
334
+ if (isVoiceFrontDoor && conversation) {
335
+ conversation.graphMemory.recordPkbQueryVectors(undefined, undefined);
336
+ }
295
337
  let v2BlockPersisted = false;
296
338
  if (
297
339
  shouldRunLegacyMemoryRetrieval({ isTrustedActor, memoryV3Live }) &&
340
+ !isVoiceFrontDoor &&
298
341
  conversation &&
299
342
  abortSignal
300
343
  ) {
@@ -311,6 +354,11 @@ const userPromptSubmitMemoryRetrieval: HookFunction<
311
354
  config,
312
355
  abortSignal,
313
356
  broadcastMessage,
357
+ legacyRetrievalRoutingMessages(
358
+ ctx.latestMessages,
359
+ ctx.isHiddenPrompt === true &&
360
+ ctx.messageKind === VOICE_ESCALATION_CONTINUATION_MESSAGE_KIND,
361
+ ),
314
362
  ),
315
363
  );
316
364
 
@@ -702,10 +702,10 @@ interface SourceParityPins {
702
702
  * parity; no system-prompt section branches on the flag, so this pin does not
703
703
  * affect prompt output) and the tool-context pin (the live consumer, gating
704
704
  * tool availability), using the live-turn derivation: interactive
705
- * interfaces run `updateClient(_, false)` (`hasNoClient = false`), while
706
- * channel-routed and chrome-extension turns stay clientless (`true`) — the
707
- * exact `isInteractiveInterface` predicate `conversation-routes.ts` /
708
- * `process-message.ts` apply. Pinned explicitly even when it matches the
705
+ * interfaces run their turns with `isInteractive: true` (`hasNoClient` reads
706
+ * `false` for the turn), while channel-routed and chrome-extension turns stay
707
+ * clientless (`true`) — the exact `isInteractiveInterface` predicate
708
+ * `conversation-routes.ts` / `process-message.ts` apply. Pinned explicitly even when it matches the
709
709
  * fork's hydrated value (`true`) so the parity contract doesn't depend on
710
710
  * hydration defaults.
711
711
  *
@@ -364,6 +364,24 @@ describe("memoryV3Injector — frozen net-new cards", () => {
364
364
  expect(getActiveSlugs("conv-1")).toEqual(new Set());
365
365
  });
366
366
 
367
+ test("voice front door skips current-turn orchestration in both injectors", async () => {
368
+ liveEnabled = true;
369
+ turnResults.set(0, result(["page-a"]));
370
+ seedMemoryConfig();
371
+ const ctx = {
372
+ requestId: "req-voice",
373
+ conversationId: "conv-voice",
374
+ turnIndex: 0,
375
+ trust: GUARDIAN_TRUST,
376
+ callSite: "voiceFrontDoor" as const,
377
+ };
378
+
379
+ expect(await memoryV3Injector.produce(ctx)).toBeNull();
380
+ expect(await memoryV3SpotlightInjector.produce(ctx)).toBeNull();
381
+ expect(observeTurnSpy).not.toHaveBeenCalled();
382
+ expect(getActiveSlugs("conv-voice")).toEqual(new Set());
383
+ });
384
+
367
385
  test("live retrieval failure queues a degraded-memory notice", async () => {
368
386
  liveEnabled = true;
369
387
  turnResults.set(
@@ -24,9 +24,11 @@ import { join } from "node:path";
24
24
  import { Database } from "bun:sqlite";
25
25
  import { afterAll, beforeEach, describe, expect, mock, test } from "bun:test";
26
26
 
27
+ import { VOICE_ESCALATION_CONTINUATION_MESSAGE_KIND } from "@vellumai/plugin-api";
27
28
  import { drizzle } from "drizzle-orm/bun-sqlite";
28
29
 
29
30
  import { setConfig } from "../../../../../__tests__/helpers/set-config.js";
31
+ import { ESCALATION_CONTINUATION_CONTENT } from "../../../../../calls/voice-triage-escalate.js";
30
32
  import { MemoryV3GateSchema } from "../../../../../config/schemas/memory-v3.js";
31
33
  import { ensureMemoryV3SelectionsSchema } from "../../../../../persistence/migrations/338-move-memory-v3-selections-to-memory-db.js";
32
34
  import { ensureMemoryV3EverInjectedSchema } from "../../../../../persistence/migrations/345-move-memory-v3-ever-injected-to-memory-db.js";
@@ -105,7 +107,11 @@ let selectorEnabledCfg = false;
105
107
  // Mutable `memory.v3.gate.enabled` config kill-switch carried by the mocked
106
108
  // config (default on, mirroring the schema default).
107
109
  let gateEnabledCfg = true;
108
- let messages: Array<{ role: string; content: string }> = [];
110
+ let messages: Array<{
111
+ role: string;
112
+ content: string;
113
+ metadata?: string | null;
114
+ }> = [];
109
115
 
110
116
  // Schema defaults for `memory.v3.gate` (the tuning the mocked config carries
111
117
  // and the gate-config threading test asserts against). Includes the default-on
@@ -714,6 +720,65 @@ describe("memory-v3 engine", () => {
714
720
  expect(turn.currentMessage).toBe("hello world");
715
721
  });
716
722
 
723
+ test("voice escalation continuation routes on the preceding caller message", async () => {
724
+ messages = [
725
+ {
726
+ role: "user",
727
+ content: JSON.stringify([
728
+ { type: "text", text: "what did I say my favorite color was?" },
729
+ ]),
730
+ },
731
+ {
732
+ role: "assistant",
733
+ content: JSON.stringify([
734
+ { type: "text", text: "Let me think about that for a second." },
735
+ ]),
736
+ },
737
+ {
738
+ role: "user",
739
+ content: JSON.stringify([
740
+ { type: "text", text: ESCALATION_CONTINUATION_CONTENT },
741
+ ]),
742
+ metadata: JSON.stringify({
743
+ hidden: true,
744
+ messageKind: VOICE_ESCALATION_CONTINUATION_MESSAGE_KIND,
745
+ }),
746
+ },
747
+ ];
748
+
749
+ await observeTurn("conv-1", 1);
750
+
751
+ const turn = (
752
+ orchestrateSpy.mock.calls as unknown as unknown[][]
753
+ )[0]![0] as MemoryRoutingTurn;
754
+ expect(turn.currentMessage).toBe("what did I say my favorite color was?");
755
+ });
756
+
757
+ test("generic hidden prompts route on their own content", async () => {
758
+ messages = [
759
+ {
760
+ role: "user",
761
+ content: JSON.stringify([
762
+ { type: "text", text: "visible caller message" },
763
+ ]),
764
+ },
765
+ {
766
+ role: "user",
767
+ content: JSON.stringify([
768
+ { type: "text", text: "generic hidden prompt" },
769
+ ]),
770
+ metadata: JSON.stringify({ hidden: true }),
771
+ },
772
+ ];
773
+
774
+ await observeTurn("conv-1", 1);
775
+
776
+ const turn = (
777
+ orchestrateSpy.mock.calls as unknown as unknown[][]
778
+ )[0]![0] as MemoryRoutingTurn;
779
+ expect(turn.currentMessage).toBe("generic hidden prompt");
780
+ });
781
+
717
782
  test("orchestrate receives the lane deps", async () => {
718
783
  await observeTurn("conv-1", 0);
719
784
  const deps = (
@@ -286,6 +286,11 @@ export const memoryV3Injector: Injector = {
286
286
  if (!isPersonalMemoryAllowed(ctx.trust)) {
287
287
  return null;
288
288
  }
289
+ // The front door keeps carried cards from history but defers current-turn
290
+ // retrieval to the escalated leg so memory cannot delay its first token.
291
+ if (ctx.callSite === "voiceFrontDoor") {
292
+ return null;
293
+ }
289
294
 
290
295
  let observed: OrchestrateResult | null;
291
296
  try {
@@ -412,6 +417,9 @@ export const memoryV3SpotlightInjector: Injector = {
412
417
  if (!isPersonalMemoryAllowed(ctx.trust)) {
413
418
  return null;
414
419
  }
420
+ if (ctx.callSite === "voiceFrontDoor") {
421
+ return null;
422
+ }
415
423
 
416
424
  try {
417
425
  const result = await observeTurnOnce(ctx.conversationId, ctx.turnIndex);
@@ -28,7 +28,9 @@ import { existsSync, readFileSync } from "node:fs";
28
28
  import {
29
29
  getMessages,
30
30
  listInstalledSkills,
31
+ parseMessageMetadata,
31
32
  stringifyMessageContent,
33
+ VOICE_ESCALATION_CONTINUATION_MESSAGE_KIND,
32
34
  } from "@vellumai/plugin-api";
33
35
 
34
36
  import { getConfig } from "../../../../config/loader.js";
@@ -490,9 +492,11 @@ function buildSituationalContext(): string {
490
492
 
491
493
  /**
492
494
  * Build a v3 {@link MemoryRoutingTurn} from the conversation's persisted messages.
493
- * `currentMessage` is the latest user message; `previousAssistantMessage` is
494
- * the tail of the last assistant reply BEFORE that message (the reply-query
495
- * pass's input — absent on a conversation's first turn); `recentContext` is
495
+ * `currentMessage` is the latest user message, except that the synthetic voice
496
+ * escalation continuation routes on the preceding caller message;
497
+ * `previousAssistantMessage` is the tail of the last assistant reply BEFORE
498
+ * the routed message (the reply-query pass's input, absent on a conversation's
499
+ * first turn); `recentContext` is
496
500
  * the tail of the recent transcript; `situationalContext` carries the current
497
501
  * date and the live NOW.md scratchpad. Returns `null` when there is no user
498
502
  * message to route on (nothing to shadow this turn).
@@ -509,12 +513,22 @@ async function buildShadowTurn(
509
513
  let currentMessage = "";
510
514
  let currentIndex = -1;
511
515
  for (let i = rows.length - 1; i >= 0; i--) {
512
- if (rows[i]!.role === "user") {
513
- currentMessage = stringifyMessageContent(rows[i]!.content);
514
- if (currentMessage.length > 0) {
515
- currentIndex = i;
516
- break;
517
- }
516
+ const row = rows[i]!;
517
+ if (row.role !== "user") {
518
+ continue;
519
+ }
520
+ const candidate = stringifyMessageContent(row.content);
521
+ const metadata = await parseMessageMetadata(row.metadata);
522
+ const isVoiceEscalationContinuation =
523
+ metadata?.hidden === true &&
524
+ metadata.messageKind === VOICE_ESCALATION_CONTINUATION_MESSAGE_KIND;
525
+ if (isVoiceEscalationContinuation) {
526
+ continue;
527
+ }
528
+ if (candidate.length > 0) {
529
+ currentMessage = candidate;
530
+ currentIndex = i;
531
+ break;
518
532
  }
519
533
  }
520
534
  if (currentMessage.length === 0) {
@@ -52,6 +52,7 @@ export async function spawnMemoryWorkerProcess(
52
52
  return await spawnWorkerProcess({
53
53
  pidPath: getMemoryWorkerPidPath(),
54
54
  entry: new URL("./worker.ts", import.meta.url),
55
+ packagedEntry: "memory",
55
56
  workerLabel: "Memory worker",
56
57
  options: opts,
57
58
  });
@@ -0,0 +1,3 @@
1
+ export async function loadDefaultMemoryWorker(): Promise<void> {
2
+ await import("./memory/worker.js");
3
+ }
@@ -212,6 +212,23 @@ export function getDiscoveredUserPluginNames(): Iterable<string> {
212
212
  return discoveredPluginDirs.values();
213
213
  }
214
214
 
215
+ /**
216
+ * True when this daemon process brought the plugin directory `dir` up:
217
+ * {@link bringUpPlugin} ran for it and its `init` was attempted. An `init` that
218
+ * threw still counts, because activation never aborts on init failure (see
219
+ * {@link activatePlugin}), and that is deliberate: an init-failed plugin's
220
+ * hooks and tools stay live, so its schedules must too. What membership
221
+ * excludes is a directory dropped in out-of-band that no boot scan and no
222
+ * reconcile pass has activated yet.
223
+ *
224
+ * The backing map is per-process, so a process that runs no plugin loader (the
225
+ * schedule worker, sidecar turn workers) reads `false` for every directory.
226
+ * Only code that provably runs in the main daemon may treat this as an answer.
227
+ */
228
+ export function isPluginDirActivated(dir: string): boolean {
229
+ return discoveredPluginDirs.has(dir);
230
+ }
231
+
215
232
  // ─── Source-versions reconcile ───────────────────────────────────────────────
216
233
 
217
234
  /**
@@ -267,13 +267,6 @@ export const BUNDLED_SYSTEM_SECTIONS: readonly BundledSection[] = [
267
267
  body: "",
268
268
  enabled: "!excludeCustomPrefix",
269
269
  },
270
- {
271
- id: "01-delegate-subagents",
272
- body: `## Delegate independent work
273
-
274
- When part of a task can run on its own — a research sweep, a multi-file investigation, a build-and-test loop — hand it off instead of grinding through it inline: load the \`subagent\` skill, then \`subagent_spawn\` early and in parallel. Make delegating that kind of work your default, not a last resort; an unnecessary subagent is cheaper than serialized work, and a long inline dig floods your own context.
275
- `,
276
- },
277
270
  {
278
271
  id: "01-parallel-tool-calls",
279
272
  body: `<use_parallel_tool_calls>
@@ -239,6 +239,30 @@ describe("detectOpenAICompatibleContextOverflow", () => {
239
239
  expect(detectOpenAICompatibleContextOverflow(err)).toBeNull();
240
240
  });
241
241
 
242
+ test("matches OpenAI's per-part 10MiB string_above_max_length 400", () => {
243
+ const err = buildOpenAIApiError(400, {
244
+ message:
245
+ "Invalid 'input[191].content[1].text': string too long. Expected a string with maximum length 10485760, but got a string with length 11436754 instead.",
246
+ type: "invalid_request_error",
247
+ code: "string_above_max_length",
248
+ });
249
+ const out = detectOpenAICompatibleContextOverflow(err);
250
+ expect(out).not.toBeNull();
251
+ // Byte lengths in this message must not be misread as token counts.
252
+ expect(out?.actualTokens).toBeUndefined();
253
+ expect(out?.maxTokens).toBeUndefined();
254
+ });
255
+
256
+ test("returns null for string_above_max_length on a non-content field", () => {
257
+ const err = buildOpenAIApiError(400, {
258
+ message:
259
+ "Invalid 'tools[0].function.description': string too long. Expected a string with maximum length 1048576, but got a string with length 2000000 instead.",
260
+ type: "invalid_request_error",
261
+ code: "string_above_max_length",
262
+ });
263
+ expect(detectOpenAICompatibleContextOverflow(err)).toBeNull();
264
+ });
265
+
242
266
  test("matches 'too many input tokens' variant emitted by some OpenAI-compatible providers", () => {
243
267
  const err = buildOpenAIApiError(400, {
244
268
  message: "too many input tokens: 250000",
@@ -370,6 +370,26 @@ describe("RetryProvider — callSite resolution", () => {
370
370
  }
371
371
  });
372
372
 
373
+ test("memory-v3 selection does not inherit high effort", async () => {
374
+ setLlmConfig({ defaultProvider: { provider: "anthropic" } });
375
+
376
+ let seen: SendMessageOptions | undefined;
377
+ const wrapped = new RetryProvider(
378
+ makeProvider("anthropic", (options) => {
379
+ seen = options;
380
+ }),
381
+ );
382
+
383
+ await wrapped.sendMessage(DUMMY_MESSAGES, {
384
+ config: { callSite: "memoryV3SelectL2" },
385
+ });
386
+
387
+ const config = seen?.config as Record<string, unknown>;
388
+ expect(config.effort).toBe("low");
389
+ expect(config.thinking).toEqual({ type: "disabled" });
390
+ expect(config.temperature).toBe(0);
391
+ });
392
+
373
393
  test("propagates resolved effort/speed/temperature; omits server-side fields", async () => {
374
394
  setLlmConfig({
375
395
  callSites: {
@@ -85,7 +85,17 @@ export function detectOpenAICompatibleContextOverflow(
85
85
  /context.?length.?exceeded|context.?window.?exceeded|prompt.?is.?too.?long|prompt_too_long|input.?too.?long|too.?many.?(?:input.?)?tokens|maximum.?context/i.test(
86
86
  message,
87
87
  );
88
- if (!codeMatches && !messageMatches) {
88
+ // string_above_max_length is OpenAI's generic oversized-string validation
89
+ // code, so only treat it as overflow when the error points at a message
90
+ // content part (e.g. "Invalid 'input[191].content[1].text': string too
91
+ // long" — OpenAI's per-part 10 MiB cap). The overflow ladder can shrink
92
+ // message content (media stubbing collapses a file's extracted_text to a
93
+ // preview) but cannot fix other oversized fields like tool definitions.
94
+ const oversizedContentPart =
95
+ /string.?too.?long|string_above_max_length/i.test(
96
+ `${code ?? ""} ${message}`,
97
+ ) && /\b(?:input|messages)\[\d+\]\.content/i.test(message);
98
+ if (!codeMatches && !messageMatches && !oversizedContentPart) {
89
99
  return null;
90
100
  }
91
101
  // OpenAI-compatible providers rarely report usable token counts; best-effort extract.
@@ -185,7 +185,6 @@ describe("DeepgramFluxRealtimeTranscriber", () => {
185
185
  const transcriber = new DeepgramFluxRealtimeTranscriber(TEST_API_KEY, {
186
186
  // Long enough that no watchdog fires mid-test.
187
187
  inactivityTimeoutMs: 60_000,
188
- keepaliveIntervalMs: 0,
189
188
  ...options,
190
189
  });
191
190
  const events: SttStreamServerEvent[] = [];
@@ -571,16 +570,22 @@ describe("DeepgramFluxRealtimeTranscriber", () => {
571
570
  expect(events).toEqual([{ type: "closed" }]);
572
571
  });
573
572
 
574
- test("keepalive frames go out on the configured interval", async () => {
575
- const { transcriber } = await startSession({ keepaliveIntervalMs: 10 });
573
+ test("no KeepAlive is ever sent: Flux rejects it and closes", async () => {
574
+ const { transcriber } = await startSession();
576
575
 
576
+ // Flux accepts only CloseStream and Configure. A KeepAlive earns an
577
+ // UNPARSABLE_CLIENT_MESSAGE error frame and a server close, which on a
578
+ // stream held across turns kills it every keepalive interval.
577
579
  await Bun.sleep(35);
578
580
  transcriber.stop();
579
581
 
580
- const keepalives = mockWs.sentData.filter(
581
- (data) => data === JSON.stringify({ type: "KeepAlive" }),
582
+ const controlFrames = mockWs.sentData.filter(
583
+ (data) => typeof data === "string",
584
+ );
585
+ expect(controlFrames).not.toContain(
586
+ JSON.stringify({ type: "KeepAlive" }),
582
587
  );
583
- expect(keepalives.length).toBeGreaterThanOrEqual(2);
588
+ expect(controlFrames).toContain(JSON.stringify({ type: "CloseStream" }));
584
589
  });
585
590
 
586
591
  test("finalizeUtterance is absent, Flux has no mid-stream flush", async () => {