@vellumai/assistant 0.11.4-staging.2 → 0.11.4-staging.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +8 -2
- package/ARCHITECTURE.md +2 -0
- package/docs/architecture/memory.md +15 -0
- package/docs/browser-use-architecture-phase2.md +128 -56
- package/docs/flux-turn-detection-spike.md +11 -6
- package/knip.json +3 -0
- package/openapi.yaml +136 -76
- package/package.json +1 -1
- package/scripts/write-plugin-api-shim.ts +10 -0
- package/src/__tests__/app-control-flow.test.ts +1 -0
- package/src/__tests__/approval-routes-http.test.ts +2 -2
- package/src/__tests__/assistant-feature-flag-guard.test.ts +25 -3
- package/src/__tests__/channel-setup-panel-ack.test.ts +1 -1
- package/src/__tests__/compaction-events.test.ts +8 -10
- package/src/__tests__/conversation-agent-loop.test.ts +4 -1
- package/src/__tests__/conversation-confirmation-signals.test.ts +112 -0
- package/src/__tests__/conversation-notifiers-provenance.test.ts +1 -1
- package/src/__tests__/conversation-queue.test.ts +39 -62
- package/src/__tests__/conversation-routes-disk-view.test.ts +1 -1
- package/src/__tests__/conversation-routes-enabled-plugins.test.ts +1 -1
- package/src/__tests__/conversation-routes-guardian-reply.test.ts +9 -9
- package/src/__tests__/conversation-routes-hidden-queue.test.ts +1 -1
- package/src/__tests__/conversation-routes-slash-commands.test.ts +1 -1
- package/src/__tests__/conversation-slash-queue.test.ts +3 -0
- package/src/__tests__/conversation-surfaces-action-delivery.test.ts +1 -0
- package/src/__tests__/conversation-surfaces-activation-emit.test.ts +1 -0
- package/src/__tests__/conversation-surfaces-app-control.test.ts +1 -0
- package/src/__tests__/conversation-surfaces-app-open.test.ts +1 -1
- package/src/__tests__/conversation-surfaces-data-persist.test.ts +1 -1
- package/src/__tests__/conversation-surfaces-history-restored-completion.test.ts +21 -14
- package/src/__tests__/conversation-surfaces-queued-emit.test.ts +1 -0
- package/src/__tests__/conversation-surfaces-standalone-payloads.test.ts +1 -0
- package/src/__tests__/conversation-surfaces-standalone.test.ts +1 -0
- package/src/__tests__/conversation-surfaces-state-update.test.ts +1 -1
- package/src/__tests__/conversation-surfaces-table-action.test.ts +1 -1
- package/src/__tests__/conversation-surfaces-task-progress.test.ts +1 -1
- package/src/__tests__/conversation-tool-setup-app-refresh.test.ts +1 -1
- package/src/__tests__/conversation-tool-setup-attribution.test.ts +1 -1
- package/src/__tests__/cu-unified-flow.test.ts +1 -0
- package/src/__tests__/document-sync-tags.test.ts +0 -75
- package/src/__tests__/gateway-only-guard.test.ts +2 -5
- package/src/__tests__/http-user-message-parity.test.ts +1 -1
- package/src/__tests__/init-feature-flag-overrides.test.ts +49 -0
- package/src/__tests__/managed-skill-lifecycle.test.ts +7 -0
- package/src/__tests__/media-generate-image.test.ts +131 -21
- package/src/__tests__/memory-retrieval-hook.test.ts +94 -2
- package/src/__tests__/plugin-api-webhook-url.test.ts +10 -7
- package/src/__tests__/plugin-import-boundary-reverse-guard.test.ts +7 -3
- package/src/__tests__/proxy-approval-callback.test.ts +1 -0
- package/src/__tests__/qdrant-manager.test.ts +14 -1
- package/src/__tests__/run-due-schedules.test.ts +21 -0
- package/src/__tests__/scaffold-managed-skill-tool.test.ts +187 -18
- package/src/__tests__/schedule-routes.test.ts +23 -0
- package/src/__tests__/schedule-store.test.ts +17 -0
- package/src/__tests__/secret-ingress-http.test.ts +1 -1
- package/src/__tests__/send-endpoint-busy.test.ts +3 -3
- package/src/__tests__/starter-task-flow.test.ts +5 -4
- package/src/__tests__/subagent-fork-prompt-role.test.ts +1 -1
- package/src/__tests__/subagent-spawn-and-await.test.ts +4 -7
- package/src/__tests__/subagent-tool-gate-mode.test.ts +1 -1
- package/src/__tests__/subagent-tools.test.ts +81 -101
- package/src/__tests__/surface-completion-in-flight-snapshot.test.ts +1 -0
- package/src/__tests__/ui-choice-copy-surfaces.test.ts +1 -1
- package/src/__tests__/ui-visual-surface.test.ts +1 -1
- package/src/__tests__/ui-voice-picker-surface.test.ts +1 -1
- package/src/__tests__/ui-work-result-surface.test.ts +1 -1
- package/src/__tests__/voice-scoped-grant-consumer.test.ts +5 -3
- package/src/__tests__/voice-session-bridge.test.ts +85 -29
- package/src/acp/session-manager.ts +8 -1
- package/src/api/surfaces.ts +5 -0
- package/src/calls/__tests__/voice-session-bridge.test.ts +21 -10
- package/src/calls/__tests__/voice-triage-escalate.test.ts +8 -0
- package/src/calls/voice-session-bridge.ts +44 -25
- package/src/calls/voice-triage-escalate.ts +1 -0
- package/src/cli/bundled-modules.ts +29 -0
- package/src/cli/commands/db/repair.ts +4 -8
- package/src/cli/commands/domain.ts +6 -3
- package/src/cli/commands/email.ts +6 -3
- package/src/cli/commands/keys.ts +8 -3
- package/src/cli/commands/plugins.ts +85 -36
- package/src/cli/commands/schedules.ts +35 -1
- package/src/cli/lib/bundled-marketplace.json +1 -1
- package/src/config/__tests__/balanced-model-experiment.test.ts +278 -0
- package/src/config/assistant-feature-flags.ts +36 -15
- package/src/config/balanced-model-experiment.ts +35 -0
- package/src/config/bundled-skills/image-studio/SKILL.md +5 -4
- package/src/config/bundled-skills/image-studio/TOOLS.json +1 -1
- package/src/config/bundled-skills/image-studio/tools/media-generate-image.ts +101 -0
- package/src/config/bundled-skills/skill-management/TOOLS.json +9 -3
- package/src/config/bundled-skills/subagent/SKILL.md +17 -12
- package/src/config/bundled-skills/subagent/TOOLS.json +4 -4
- package/src/config/call-site-defaults.ts +7 -0
- package/src/config/default-profile-catalog.ts +96 -4
- package/src/config/feature-flag-registry.json +11 -11
- package/src/config/skills.ts +9 -2
- package/src/daemon/__tests__/conversation-surfaces-launch.test.ts +1 -1
- package/src/daemon/conversation-agent-loop.ts +14 -13
- package/src/daemon/conversation-notifiers.ts +11 -9
- package/src/daemon/conversation-process.ts +0 -27
- package/src/daemon/conversation-store.ts +4 -4
- package/src/daemon/conversation-surfaces.ts +27 -13
- package/src/daemon/conversation-tool-setup.ts +2 -4
- package/src/daemon/conversation.ts +81 -44
- package/src/daemon/doordash-steps.ts +2 -2
- package/src/daemon/lifecycle.ts +14 -1
- package/src/daemon/process-message.ts +0 -13
- package/src/daemon/windows-compiled-entry.ts +4 -0
- package/src/documents/document-store.ts +5 -235
- package/src/hooks/types.ts +5 -0
- package/src/ipc/gateway-flag-listener.ts +17 -3
- package/src/live-voice/__tests__/live-voice-flux-turn-end.test.ts +118 -0
- package/src/live-voice/live-voice-manager.ts +16 -3
- package/src/live-voice/live-voice-session.ts +63 -9
- package/src/live-voice/windows-compiled-live-voice.ts +4 -0
- package/src/monitoring/control.ts +1 -0
- package/src/monitoring/db-integrity-sample.ts +4 -5
- package/src/permissions/prompter.ts +1 -5
- package/src/persistence/conversation-queries.ts +66 -16
- package/src/persistence/embeddings/qdrant-manager.ts +84 -49
- package/src/persistence/migrations/360-add-document-workspace-path.ts +5 -14
- package/src/persistence/schema/documents.ts +4 -4
- package/src/plugin-api/constants.ts +8 -0
- package/src/plugin-api/index.ts +5 -1
- package/src/plugin-api/webhook-url.ts +13 -11
- package/src/plugins/defaults/main.ts +6 -7
- package/src/plugins/defaults/memory/graph/__tests__/conversation-graph-memory-v2-routing.test.ts +77 -0
- package/src/plugins/defaults/memory/graph/conversation-graph-memory.ts +24 -6
- package/src/plugins/defaults/memory/hooks/user-prompt-submit.ts +53 -5
- package/src/plugins/defaults/memory/memory-retrospective-job.ts +4 -4
- package/src/plugins/defaults/memory/v3/__tests__/injection.test.ts +18 -0
- package/src/plugins/defaults/memory/v3/__tests__/shadow-plugin.test.ts +66 -1
- package/src/plugins/defaults/memory/v3/injector.ts +8 -0
- package/src/plugins/defaults/memory/v3/shadow-plugin.ts +23 -9
- package/src/plugins/defaults/memory/worker-control.ts +1 -0
- package/src/plugins/defaults/worker-entrypoints.ts +3 -0
- package/src/plugins/mtime-cache.ts +17 -0
- package/src/prompts/templates/system-sections.ts +0 -7
- package/src/providers/__tests__/context-overflow-error.test.ts +24 -0
- package/src/providers/__tests__/retry-callsite.test.ts +20 -0
- package/src/providers/openai/chat-completions-provider.ts +11 -1
- package/src/providers/speech-to-text/__tests__/deepgram-flux-realtime.test.ts +11 -6
- package/src/providers/speech-to-text/deepgram-flux-realtime.ts +9 -49
- package/src/routes/control.ts +1 -0
- package/src/routes/route-host-client.ts +1 -0
- package/src/runtime/AGENTS.md +16 -17
- package/src/runtime/agent-wake.ts +15 -12
- package/src/runtime/routes/__tests__/conversation-list-routes.test.ts +170 -1
- package/src/runtime/routes/__tests__/schedule-routes-disarm-reason.test.ts +215 -0
- package/src/runtime/routes/conversation-list-routes.ts +54 -22
- package/src/runtime/routes/conversation-management-routes.ts +2 -3
- package/src/runtime/routes/conversation-routes.ts +11 -13
- package/src/runtime/routes/documents-routes.ts +3 -222
- package/src/runtime/routes/playground/__tests__/inject-failures.test.ts +2 -0
- package/src/runtime/routes/playground/__tests__/reset-circuit.test.ts +3 -0
- package/src/runtime/routes/playground/inject-failures.ts +2 -2
- package/src/runtime/routes/playground/reset-circuit.ts +1 -1
- package/src/runtime/routes/schedule-routes.ts +94 -7
- package/src/runtime/routes/workspace-routes.ts +0 -9
- package/src/runtime/routes/workspace-utils.ts +3 -13
- package/src/runtime/services/conversation-serializer.ts +7 -2
- package/src/schedule/__tests__/plugin-schedule-declarations.test.ts +68 -5
- package/src/schedule/__tests__/plugin-schedule-reconciler.test.ts +81 -0
- package/src/schedule/plugin-schedule-availability.ts +58 -0
- package/src/schedule/plugin-schedule-declarations.ts +23 -27
- package/src/schedule/plugin-schedule-reconciler.ts +12 -3
- package/src/schedule/schedule-store.ts +5 -1
- package/src/schedule/scheduler.ts +9 -4
- package/src/schedule/worker-control.ts +1 -0
- package/src/subagent/__tests__/consult-prompt.test.ts +26 -15
- package/src/subagent/consult-context.ts +11 -11
- package/src/subagent/consult-prompt.ts +26 -35
- package/src/subagent/manager.ts +20 -37
- package/src/subagent/notify.ts +7 -1
- package/src/subagent/types.ts +8 -6
- package/src/tools/acp/spawn.ts +6 -4
- package/src/tools/skills/scaffold-managed.ts +25 -7
- package/src/tools/subagent/spawn.ts +28 -88
- package/src/tools/ui-surface/surface-shape-docs.ts +1 -1
- package/src/util/__tests__/worker-process-command.test.ts +37 -0
- package/src/util/logger.ts +16 -0
- package/src/util/worker-process.ts +37 -4
- package/src/windows-compiled-cli.ts +32 -0
- package/src/windows-compiled-entry.ts +4 -0
- package/src/windows-compiled-logger.ts +6 -0
- package/src/windows-compiled-worker-entry.ts +29 -0
- package/src/__tests__/document-workspace-file.test.ts +0 -467
- package/src/daemon/interactive-turn-sender.ts +0 -59
- package/src/subagent/__tests__/consult-transcript.test.ts +0 -184
- package/src/subagent/consult-transcript.ts +0 -90
|
@@ -512,6 +512,10 @@ export class ConversationGraphMemory {
|
|
|
512
512
|
* - Turn 1 (or after compaction): full context load
|
|
513
513
|
* - Every other turn: per-turn injection
|
|
514
514
|
*
|
|
515
|
+
* `routingMessages` may end before an internal continuation message. The
|
|
516
|
+
* retrievers derive their query from that view, while memory is still
|
|
517
|
+
* injected into `messages`, which is the history sent to the model.
|
|
518
|
+
*
|
|
515
519
|
* Returns augmented messages with memory context prepended to the last
|
|
516
520
|
* user message, following the same injection pattern as the old system.
|
|
517
521
|
*/
|
|
@@ -520,6 +524,7 @@ export class ConversationGraphMemory {
|
|
|
520
524
|
config: AssistantConfig,
|
|
521
525
|
abortSignal: AbortSignal,
|
|
522
526
|
onEvent: (msg: AssistantEvent) => void,
|
|
527
|
+
routingMessages: Message[] = messages,
|
|
523
528
|
): Promise<{
|
|
524
529
|
runMessages: Message[];
|
|
525
530
|
injectedTokens: number;
|
|
@@ -579,7 +584,7 @@ export class ConversationGraphMemory {
|
|
|
579
584
|
|
|
580
585
|
// Gate: skip for empty/tool-result-only messages — unless we need to
|
|
581
586
|
// reload after compaction (needsReload) or haven't initialized yet.
|
|
582
|
-
const lastMessage =
|
|
587
|
+
const lastMessage = routingMessages[routingMessages.length - 1];
|
|
583
588
|
if (!lastMessage || lastMessage.role !== "user") {
|
|
584
589
|
return noopResult;
|
|
585
590
|
}
|
|
@@ -598,6 +603,7 @@ export class ConversationGraphMemory {
|
|
|
598
603
|
|
|
599
604
|
return await this.runContextLoad(
|
|
600
605
|
messages,
|
|
606
|
+
routingMessages,
|
|
601
607
|
config,
|
|
602
608
|
recentSummaries,
|
|
603
609
|
firstUserText ?? undefined,
|
|
@@ -606,7 +612,12 @@ export class ConversationGraphMemory {
|
|
|
606
612
|
);
|
|
607
613
|
}
|
|
608
614
|
|
|
609
|
-
return await this.runPerTurn(
|
|
615
|
+
return await this.runPerTurn(
|
|
616
|
+
messages,
|
|
617
|
+
routingMessages,
|
|
618
|
+
config,
|
|
619
|
+
abortSignal,
|
|
620
|
+
);
|
|
610
621
|
} catch (err) {
|
|
611
622
|
const errCode =
|
|
612
623
|
err instanceof z.ZodError ? err.issues[0]?.code : undefined;
|
|
@@ -629,6 +640,7 @@ export class ConversationGraphMemory {
|
|
|
629
640
|
|
|
630
641
|
private async runContextLoad(
|
|
631
642
|
messages: Message[],
|
|
643
|
+
routingMessages: Message[],
|
|
632
644
|
config: AssistantConfig,
|
|
633
645
|
recentSummaries: string[],
|
|
634
646
|
userQuery: string | undefined,
|
|
@@ -640,9 +652,12 @@ export class ConversationGraphMemory {
|
|
|
640
652
|
// The activation pipeline is robust to weak ANN signal — it falls back
|
|
641
653
|
// to spreading + nowText to surface candidates.
|
|
642
654
|
const startedAt = Date.now();
|
|
643
|
-
const rawUserText = readRawUserText(
|
|
655
|
+
const rawUserText = readRawUserText(
|
|
656
|
+
routingMessages[routingMessages.length - 1],
|
|
657
|
+
);
|
|
644
658
|
const v2 = await this.maybeRouteV2Injection(
|
|
645
659
|
messages,
|
|
660
|
+
routingMessages,
|
|
646
661
|
config,
|
|
647
662
|
"context-load",
|
|
648
663
|
// Context-load runs before the messages array necessarily contains
|
|
@@ -780,6 +795,7 @@ export class ConversationGraphMemory {
|
|
|
780
795
|
|
|
781
796
|
private async runPerTurn(
|
|
782
797
|
messages: Message[],
|
|
798
|
+
routingMessages: Message[],
|
|
783
799
|
config: AssistantConfig,
|
|
784
800
|
signal: AbortSignal,
|
|
785
801
|
) {
|
|
@@ -788,8 +804,8 @@ export class ConversationGraphMemory {
|
|
|
788
804
|
let userLast = "";
|
|
789
805
|
let userLastBlocks: ContentBlock[] = [];
|
|
790
806
|
|
|
791
|
-
for (let i =
|
|
792
|
-
const msg =
|
|
807
|
+
for (let i = routingMessages.length - 1; i >= 0; i--) {
|
|
808
|
+
const msg = routingMessages[i];
|
|
793
809
|
const text = msg.content
|
|
794
810
|
.filter(
|
|
795
811
|
(b): b is Extract<typeof b, { type: "text" }> => b.type === "text",
|
|
@@ -815,6 +831,7 @@ export class ConversationGraphMemory {
|
|
|
815
831
|
const startedAt = Date.now();
|
|
816
832
|
const v2 = await this.maybeRouteV2Injection(
|
|
817
833
|
messages,
|
|
834
|
+
routingMessages,
|
|
818
835
|
config,
|
|
819
836
|
"per-turn",
|
|
820
837
|
null,
|
|
@@ -978,6 +995,7 @@ export class ConversationGraphMemory {
|
|
|
978
995
|
*/
|
|
979
996
|
private async maybeRouteV2Injection(
|
|
980
997
|
messages: Message[],
|
|
998
|
+
routingMessages: Message[],
|
|
981
999
|
config: AssistantConfig,
|
|
982
1000
|
mode: InjectMemoryV2Mode,
|
|
983
1001
|
/**
|
|
@@ -1004,7 +1022,7 @@ export class ConversationGraphMemory {
|
|
|
1004
1022
|
const recentTurnPairs =
|
|
1005
1023
|
userMessageOverride !== null
|
|
1006
1024
|
? [{ assistantMessage: "", userMessage: userMessageOverride }]
|
|
1007
|
-
: extractRecentTurnPairs(
|
|
1025
|
+
: extractRecentTurnPairs(routingMessages, historicalPairs);
|
|
1008
1026
|
|
|
1009
1027
|
const result = await injectMemoryV2Block({
|
|
1010
1028
|
conversationId: this.conversationId,
|
|
@@ -31,7 +31,10 @@ import type {
|
|
|
31
31
|
HookFunction,
|
|
32
32
|
UserPromptSubmitContext,
|
|
33
33
|
} from "@vellumai/plugin-api";
|
|
34
|
-
import {
|
|
34
|
+
import {
|
|
35
|
+
updateMessageMetadata,
|
|
36
|
+
VOICE_ESCALATION_CONTINUATION_MESSAGE_KIND,
|
|
37
|
+
} from "@vellumai/plugin-api";
|
|
35
38
|
|
|
36
39
|
import type { MemoryRecalledEvent } from "../../../../api/events/memory-recalled.js";
|
|
37
40
|
import { getConfig } from "../../../../config/loader.js";
|
|
@@ -48,6 +51,7 @@ import { timeLatencySubSpan } from "../../../../daemon/turn-latency-sub-spans.js
|
|
|
48
51
|
import { broadcastMessage } from "../../../../runtime/assistant-event-hub.js";
|
|
49
52
|
import type { GraphMemoryResult } from "../graph/conversation-graph-memory.js";
|
|
50
53
|
import { recordMemoryRecallLog } from "../memory-recall-log-store.js";
|
|
54
|
+
import { stripTailInjectionsForReinjection } from "../tail-reinjection-strip.js";
|
|
51
55
|
import { MEMORY_V3_INJECTED_BLOCK_METADATA_KEY } from "../v3/ever-injected-store.js";
|
|
52
56
|
|
|
53
57
|
/**
|
|
@@ -69,6 +73,39 @@ export function shouldRunLegacyMemoryRetrieval(params: {
|
|
|
69
73
|
return params.isTrustedActor && !params.memoryV3Live;
|
|
70
74
|
}
|
|
71
75
|
|
|
76
|
+
/**
|
|
77
|
+
* The voice escalation continuation is a model-control message, not a
|
|
78
|
+
* retrieval query. Route memory on the latest preceding user message and strip
|
|
79
|
+
* the runtime context that was frozen onto it during its original turn.
|
|
80
|
+
*/
|
|
81
|
+
function legacyRetrievalRoutingMessages(
|
|
82
|
+
messages: UserPromptSubmitContext["latestMessages"],
|
|
83
|
+
routeOnPreviousUser: boolean,
|
|
84
|
+
): UserPromptSubmitContext["latestMessages"] {
|
|
85
|
+
if (!routeOnPreviousUser) {
|
|
86
|
+
return messages;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
for (let i = messages.length - 2; i >= 0; i -= 1) {
|
|
90
|
+
if (messages[i]?.role !== "user") {
|
|
91
|
+
continue;
|
|
92
|
+
}
|
|
93
|
+
const candidate = stripTailInjectionsForReinjection(
|
|
94
|
+
messages.slice(0, i + 1),
|
|
95
|
+
);
|
|
96
|
+
const tail = candidate[candidate.length - 1];
|
|
97
|
+
if (
|
|
98
|
+
tail?.role === "user" &&
|
|
99
|
+
tail.content.some(
|
|
100
|
+
(block) => block.type === "text" && block.text.trim().length > 0,
|
|
101
|
+
)
|
|
102
|
+
) {
|
|
103
|
+
return candidate;
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
return [];
|
|
107
|
+
}
|
|
108
|
+
|
|
72
109
|
/**
|
|
73
110
|
* Persist and broadcast the retrieval's side effects: the injected block on
|
|
74
111
|
* the user message's metadata (so it survives reloads), a recall-log row, and
|
|
@@ -263,10 +300,11 @@ async function persistInjectionBlocks(
|
|
|
263
300
|
* runs for every actor, writing the fully injected result back onto
|
|
264
301
|
* `latestMessages` and persisting the assembled blocks.
|
|
265
302
|
*
|
|
266
|
-
* Memory retrieval blocks the turn
|
|
267
|
-
*
|
|
268
|
-
*
|
|
269
|
-
*
|
|
303
|
+
* Memory retrieval ordinarily blocks the turn. The voice front door is the
|
|
304
|
+
* latency-sensitive exception: it keeps carried memory in the front prompt
|
|
305
|
+
* and defers current-turn retrieval to the escalated leg when one is needed.
|
|
306
|
+
* Cancellation still works via the live conversation signal for ordinary
|
|
307
|
+
* retrieval.
|
|
270
308
|
*/
|
|
271
309
|
const userPromptSubmitMemoryRetrieval: HookFunction<
|
|
272
310
|
UserPromptSubmitContext
|
|
@@ -292,9 +330,14 @@ const userPromptSubmitMemoryRetrieval: HookFunction<
|
|
|
292
330
|
// fallback — a v3 empty/failed selection yields no NEW injected memory that
|
|
293
331
|
// turn (prior turns' frozen v3 cards still ride history).
|
|
294
332
|
const memoryV3Live = isMemoryV3Live(config);
|
|
333
|
+
const isVoiceFrontDoor = conversation?.currentCallSite === "voiceFrontDoor";
|
|
334
|
+
if (isVoiceFrontDoor && conversation) {
|
|
335
|
+
conversation.graphMemory.recordPkbQueryVectors(undefined, undefined);
|
|
336
|
+
}
|
|
295
337
|
let v2BlockPersisted = false;
|
|
296
338
|
if (
|
|
297
339
|
shouldRunLegacyMemoryRetrieval({ isTrustedActor, memoryV3Live }) &&
|
|
340
|
+
!isVoiceFrontDoor &&
|
|
298
341
|
conversation &&
|
|
299
342
|
abortSignal
|
|
300
343
|
) {
|
|
@@ -311,6 +354,11 @@ const userPromptSubmitMemoryRetrieval: HookFunction<
|
|
|
311
354
|
config,
|
|
312
355
|
abortSignal,
|
|
313
356
|
broadcastMessage,
|
|
357
|
+
legacyRetrievalRoutingMessages(
|
|
358
|
+
ctx.latestMessages,
|
|
359
|
+
ctx.isHiddenPrompt === true &&
|
|
360
|
+
ctx.messageKind === VOICE_ESCALATION_CONTINUATION_MESSAGE_KIND,
|
|
361
|
+
),
|
|
314
362
|
),
|
|
315
363
|
);
|
|
316
364
|
|
|
@@ -702,10 +702,10 @@ interface SourceParityPins {
|
|
|
702
702
|
* parity; no system-prompt section branches on the flag, so this pin does not
|
|
703
703
|
* affect prompt output) and the tool-context pin (the live consumer, gating
|
|
704
704
|
* tool availability), using the live-turn derivation: interactive
|
|
705
|
-
* interfaces run `
|
|
706
|
-
* channel-routed and chrome-extension turns stay
|
|
707
|
-
*
|
|
708
|
-
* `process-message.ts` apply. Pinned explicitly even when it matches the
|
|
705
|
+
* interfaces run their turns with `isInteractive: true` (`hasNoClient` reads
|
|
706
|
+
* `false` for the turn), while channel-routed and chrome-extension turns stay
|
|
707
|
+
* clientless (`true`) — the exact `isInteractiveInterface` predicate
|
|
708
|
+
* `conversation-routes.ts` / `process-message.ts` apply. Pinned explicitly even when it matches the
|
|
709
709
|
* fork's hydrated value (`true`) so the parity contract doesn't depend on
|
|
710
710
|
* hydration defaults.
|
|
711
711
|
*
|
|
@@ -364,6 +364,24 @@ describe("memoryV3Injector — frozen net-new cards", () => {
|
|
|
364
364
|
expect(getActiveSlugs("conv-1")).toEqual(new Set());
|
|
365
365
|
});
|
|
366
366
|
|
|
367
|
+
test("voice front door skips current-turn orchestration in both injectors", async () => {
|
|
368
|
+
liveEnabled = true;
|
|
369
|
+
turnResults.set(0, result(["page-a"]));
|
|
370
|
+
seedMemoryConfig();
|
|
371
|
+
const ctx = {
|
|
372
|
+
requestId: "req-voice",
|
|
373
|
+
conversationId: "conv-voice",
|
|
374
|
+
turnIndex: 0,
|
|
375
|
+
trust: GUARDIAN_TRUST,
|
|
376
|
+
callSite: "voiceFrontDoor" as const,
|
|
377
|
+
};
|
|
378
|
+
|
|
379
|
+
expect(await memoryV3Injector.produce(ctx)).toBeNull();
|
|
380
|
+
expect(await memoryV3SpotlightInjector.produce(ctx)).toBeNull();
|
|
381
|
+
expect(observeTurnSpy).not.toHaveBeenCalled();
|
|
382
|
+
expect(getActiveSlugs("conv-voice")).toEqual(new Set());
|
|
383
|
+
});
|
|
384
|
+
|
|
367
385
|
test("live retrieval failure queues a degraded-memory notice", async () => {
|
|
368
386
|
liveEnabled = true;
|
|
369
387
|
turnResults.set(
|
|
@@ -24,9 +24,11 @@ import { join } from "node:path";
|
|
|
24
24
|
import { Database } from "bun:sqlite";
|
|
25
25
|
import { afterAll, beforeEach, describe, expect, mock, test } from "bun:test";
|
|
26
26
|
|
|
27
|
+
import { VOICE_ESCALATION_CONTINUATION_MESSAGE_KIND } from "@vellumai/plugin-api";
|
|
27
28
|
import { drizzle } from "drizzle-orm/bun-sqlite";
|
|
28
29
|
|
|
29
30
|
import { setConfig } from "../../../../../__tests__/helpers/set-config.js";
|
|
31
|
+
import { ESCALATION_CONTINUATION_CONTENT } from "../../../../../calls/voice-triage-escalate.js";
|
|
30
32
|
import { MemoryV3GateSchema } from "../../../../../config/schemas/memory-v3.js";
|
|
31
33
|
import { ensureMemoryV3SelectionsSchema } from "../../../../../persistence/migrations/338-move-memory-v3-selections-to-memory-db.js";
|
|
32
34
|
import { ensureMemoryV3EverInjectedSchema } from "../../../../../persistence/migrations/345-move-memory-v3-ever-injected-to-memory-db.js";
|
|
@@ -105,7 +107,11 @@ let selectorEnabledCfg = false;
|
|
|
105
107
|
// Mutable `memory.v3.gate.enabled` config kill-switch carried by the mocked
|
|
106
108
|
// config (default on, mirroring the schema default).
|
|
107
109
|
let gateEnabledCfg = true;
|
|
108
|
-
let messages: Array<{
|
|
110
|
+
let messages: Array<{
|
|
111
|
+
role: string;
|
|
112
|
+
content: string;
|
|
113
|
+
metadata?: string | null;
|
|
114
|
+
}> = [];
|
|
109
115
|
|
|
110
116
|
// Schema defaults for `memory.v3.gate` (the tuning the mocked config carries
|
|
111
117
|
// and the gate-config threading test asserts against). Includes the default-on
|
|
@@ -714,6 +720,65 @@ describe("memory-v3 engine", () => {
|
|
|
714
720
|
expect(turn.currentMessage).toBe("hello world");
|
|
715
721
|
});
|
|
716
722
|
|
|
723
|
+
test("voice escalation continuation routes on the preceding caller message", async () => {
|
|
724
|
+
messages = [
|
|
725
|
+
{
|
|
726
|
+
role: "user",
|
|
727
|
+
content: JSON.stringify([
|
|
728
|
+
{ type: "text", text: "what did I say my favorite color was?" },
|
|
729
|
+
]),
|
|
730
|
+
},
|
|
731
|
+
{
|
|
732
|
+
role: "assistant",
|
|
733
|
+
content: JSON.stringify([
|
|
734
|
+
{ type: "text", text: "Let me think about that for a second." },
|
|
735
|
+
]),
|
|
736
|
+
},
|
|
737
|
+
{
|
|
738
|
+
role: "user",
|
|
739
|
+
content: JSON.stringify([
|
|
740
|
+
{ type: "text", text: ESCALATION_CONTINUATION_CONTENT },
|
|
741
|
+
]),
|
|
742
|
+
metadata: JSON.stringify({
|
|
743
|
+
hidden: true,
|
|
744
|
+
messageKind: VOICE_ESCALATION_CONTINUATION_MESSAGE_KIND,
|
|
745
|
+
}),
|
|
746
|
+
},
|
|
747
|
+
];
|
|
748
|
+
|
|
749
|
+
await observeTurn("conv-1", 1);
|
|
750
|
+
|
|
751
|
+
const turn = (
|
|
752
|
+
orchestrateSpy.mock.calls as unknown as unknown[][]
|
|
753
|
+
)[0]![0] as MemoryRoutingTurn;
|
|
754
|
+
expect(turn.currentMessage).toBe("what did I say my favorite color was?");
|
|
755
|
+
});
|
|
756
|
+
|
|
757
|
+
test("generic hidden prompts route on their own content", async () => {
|
|
758
|
+
messages = [
|
|
759
|
+
{
|
|
760
|
+
role: "user",
|
|
761
|
+
content: JSON.stringify([
|
|
762
|
+
{ type: "text", text: "visible caller message" },
|
|
763
|
+
]),
|
|
764
|
+
},
|
|
765
|
+
{
|
|
766
|
+
role: "user",
|
|
767
|
+
content: JSON.stringify([
|
|
768
|
+
{ type: "text", text: "generic hidden prompt" },
|
|
769
|
+
]),
|
|
770
|
+
metadata: JSON.stringify({ hidden: true }),
|
|
771
|
+
},
|
|
772
|
+
];
|
|
773
|
+
|
|
774
|
+
await observeTurn("conv-1", 1);
|
|
775
|
+
|
|
776
|
+
const turn = (
|
|
777
|
+
orchestrateSpy.mock.calls as unknown as unknown[][]
|
|
778
|
+
)[0]![0] as MemoryRoutingTurn;
|
|
779
|
+
expect(turn.currentMessage).toBe("generic hidden prompt");
|
|
780
|
+
});
|
|
781
|
+
|
|
717
782
|
test("orchestrate receives the lane deps", async () => {
|
|
718
783
|
await observeTurn("conv-1", 0);
|
|
719
784
|
const deps = (
|
|
@@ -286,6 +286,11 @@ export const memoryV3Injector: Injector = {
|
|
|
286
286
|
if (!isPersonalMemoryAllowed(ctx.trust)) {
|
|
287
287
|
return null;
|
|
288
288
|
}
|
|
289
|
+
// The front door keeps carried cards from history but defers current-turn
|
|
290
|
+
// retrieval to the escalated leg so memory cannot delay its first token.
|
|
291
|
+
if (ctx.callSite === "voiceFrontDoor") {
|
|
292
|
+
return null;
|
|
293
|
+
}
|
|
289
294
|
|
|
290
295
|
let observed: OrchestrateResult | null;
|
|
291
296
|
try {
|
|
@@ -412,6 +417,9 @@ export const memoryV3SpotlightInjector: Injector = {
|
|
|
412
417
|
if (!isPersonalMemoryAllowed(ctx.trust)) {
|
|
413
418
|
return null;
|
|
414
419
|
}
|
|
420
|
+
if (ctx.callSite === "voiceFrontDoor") {
|
|
421
|
+
return null;
|
|
422
|
+
}
|
|
415
423
|
|
|
416
424
|
try {
|
|
417
425
|
const result = await observeTurnOnce(ctx.conversationId, ctx.turnIndex);
|
|
@@ -28,7 +28,9 @@ import { existsSync, readFileSync } from "node:fs";
|
|
|
28
28
|
import {
|
|
29
29
|
getMessages,
|
|
30
30
|
listInstalledSkills,
|
|
31
|
+
parseMessageMetadata,
|
|
31
32
|
stringifyMessageContent,
|
|
33
|
+
VOICE_ESCALATION_CONTINUATION_MESSAGE_KIND,
|
|
32
34
|
} from "@vellumai/plugin-api";
|
|
33
35
|
|
|
34
36
|
import { getConfig } from "../../../../config/loader.js";
|
|
@@ -490,9 +492,11 @@ function buildSituationalContext(): string {
|
|
|
490
492
|
|
|
491
493
|
/**
|
|
492
494
|
* Build a v3 {@link MemoryRoutingTurn} from the conversation's persisted messages.
|
|
493
|
-
* `currentMessage` is the latest user message
|
|
494
|
-
*
|
|
495
|
-
*
|
|
495
|
+
* `currentMessage` is the latest user message, except that the synthetic voice
|
|
496
|
+
* escalation continuation routes on the preceding caller message;
|
|
497
|
+
* `previousAssistantMessage` is the tail of the last assistant reply BEFORE
|
|
498
|
+
* the routed message (the reply-query pass's input, absent on a conversation's
|
|
499
|
+
* first turn); `recentContext` is
|
|
496
500
|
* the tail of the recent transcript; `situationalContext` carries the current
|
|
497
501
|
* date and the live NOW.md scratchpad. Returns `null` when there is no user
|
|
498
502
|
* message to route on (nothing to shadow this turn).
|
|
@@ -509,12 +513,22 @@ async function buildShadowTurn(
|
|
|
509
513
|
let currentMessage = "";
|
|
510
514
|
let currentIndex = -1;
|
|
511
515
|
for (let i = rows.length - 1; i >= 0; i--) {
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
516
|
+
const row = rows[i]!;
|
|
517
|
+
if (row.role !== "user") {
|
|
518
|
+
continue;
|
|
519
|
+
}
|
|
520
|
+
const candidate = stringifyMessageContent(row.content);
|
|
521
|
+
const metadata = await parseMessageMetadata(row.metadata);
|
|
522
|
+
const isVoiceEscalationContinuation =
|
|
523
|
+
metadata?.hidden === true &&
|
|
524
|
+
metadata.messageKind === VOICE_ESCALATION_CONTINUATION_MESSAGE_KIND;
|
|
525
|
+
if (isVoiceEscalationContinuation) {
|
|
526
|
+
continue;
|
|
527
|
+
}
|
|
528
|
+
if (candidate.length > 0) {
|
|
529
|
+
currentMessage = candidate;
|
|
530
|
+
currentIndex = i;
|
|
531
|
+
break;
|
|
518
532
|
}
|
|
519
533
|
}
|
|
520
534
|
if (currentMessage.length === 0) {
|
|
@@ -212,6 +212,23 @@ export function getDiscoveredUserPluginNames(): Iterable<string> {
|
|
|
212
212
|
return discoveredPluginDirs.values();
|
|
213
213
|
}
|
|
214
214
|
|
|
215
|
+
/**
|
|
216
|
+
* True when this daemon process brought the plugin directory `dir` up:
|
|
217
|
+
* {@link bringUpPlugin} ran for it and its `init` was attempted. An `init` that
|
|
218
|
+
* threw still counts, because activation never aborts on init failure (see
|
|
219
|
+
* {@link activatePlugin}), and that is deliberate: an init-failed plugin's
|
|
220
|
+
* hooks and tools stay live, so its schedules must too. What membership
|
|
221
|
+
* excludes is a directory dropped in out-of-band that no boot scan and no
|
|
222
|
+
* reconcile pass has activated yet.
|
|
223
|
+
*
|
|
224
|
+
* The backing map is per-process, so a process that runs no plugin loader (the
|
|
225
|
+
* schedule worker, sidecar turn workers) reads `false` for every directory.
|
|
226
|
+
* Only code that provably runs in the main daemon may treat this as an answer.
|
|
227
|
+
*/
|
|
228
|
+
export function isPluginDirActivated(dir: string): boolean {
|
|
229
|
+
return discoveredPluginDirs.has(dir);
|
|
230
|
+
}
|
|
231
|
+
|
|
215
232
|
// ─── Source-versions reconcile ───────────────────────────────────────────────
|
|
216
233
|
|
|
217
234
|
/**
|
|
@@ -267,13 +267,6 @@ export const BUNDLED_SYSTEM_SECTIONS: readonly BundledSection[] = [
|
|
|
267
267
|
body: "",
|
|
268
268
|
enabled: "!excludeCustomPrefix",
|
|
269
269
|
},
|
|
270
|
-
{
|
|
271
|
-
id: "01-delegate-subagents",
|
|
272
|
-
body: `## Delegate independent work
|
|
273
|
-
|
|
274
|
-
When part of a task can run on its own — a research sweep, a multi-file investigation, a build-and-test loop — hand it off instead of grinding through it inline: load the \`subagent\` skill, then \`subagent_spawn\` early and in parallel. Make delegating that kind of work your default, not a last resort; an unnecessary subagent is cheaper than serialized work, and a long inline dig floods your own context.
|
|
275
|
-
`,
|
|
276
|
-
},
|
|
277
270
|
{
|
|
278
271
|
id: "01-parallel-tool-calls",
|
|
279
272
|
body: `<use_parallel_tool_calls>
|
|
@@ -239,6 +239,30 @@ describe("detectOpenAICompatibleContextOverflow", () => {
|
|
|
239
239
|
expect(detectOpenAICompatibleContextOverflow(err)).toBeNull();
|
|
240
240
|
});
|
|
241
241
|
|
|
242
|
+
test("matches OpenAI's per-part 10MiB string_above_max_length 400", () => {
|
|
243
|
+
const err = buildOpenAIApiError(400, {
|
|
244
|
+
message:
|
|
245
|
+
"Invalid 'input[191].content[1].text': string too long. Expected a string with maximum length 10485760, but got a string with length 11436754 instead.",
|
|
246
|
+
type: "invalid_request_error",
|
|
247
|
+
code: "string_above_max_length",
|
|
248
|
+
});
|
|
249
|
+
const out = detectOpenAICompatibleContextOverflow(err);
|
|
250
|
+
expect(out).not.toBeNull();
|
|
251
|
+
// Byte lengths in this message must not be misread as token counts.
|
|
252
|
+
expect(out?.actualTokens).toBeUndefined();
|
|
253
|
+
expect(out?.maxTokens).toBeUndefined();
|
|
254
|
+
});
|
|
255
|
+
|
|
256
|
+
test("returns null for string_above_max_length on a non-content field", () => {
|
|
257
|
+
const err = buildOpenAIApiError(400, {
|
|
258
|
+
message:
|
|
259
|
+
"Invalid 'tools[0].function.description': string too long. Expected a string with maximum length 1048576, but got a string with length 2000000 instead.",
|
|
260
|
+
type: "invalid_request_error",
|
|
261
|
+
code: "string_above_max_length",
|
|
262
|
+
});
|
|
263
|
+
expect(detectOpenAICompatibleContextOverflow(err)).toBeNull();
|
|
264
|
+
});
|
|
265
|
+
|
|
242
266
|
test("matches 'too many input tokens' variant emitted by some OpenAI-compatible providers", () => {
|
|
243
267
|
const err = buildOpenAIApiError(400, {
|
|
244
268
|
message: "too many input tokens: 250000",
|
|
@@ -370,6 +370,26 @@ describe("RetryProvider — callSite resolution", () => {
|
|
|
370
370
|
}
|
|
371
371
|
});
|
|
372
372
|
|
|
373
|
+
test("memory-v3 selection does not inherit high effort", async () => {
|
|
374
|
+
setLlmConfig({ defaultProvider: { provider: "anthropic" } });
|
|
375
|
+
|
|
376
|
+
let seen: SendMessageOptions | undefined;
|
|
377
|
+
const wrapped = new RetryProvider(
|
|
378
|
+
makeProvider("anthropic", (options) => {
|
|
379
|
+
seen = options;
|
|
380
|
+
}),
|
|
381
|
+
);
|
|
382
|
+
|
|
383
|
+
await wrapped.sendMessage(DUMMY_MESSAGES, {
|
|
384
|
+
config: { callSite: "memoryV3SelectL2" },
|
|
385
|
+
});
|
|
386
|
+
|
|
387
|
+
const config = seen?.config as Record<string, unknown>;
|
|
388
|
+
expect(config.effort).toBe("low");
|
|
389
|
+
expect(config.thinking).toEqual({ type: "disabled" });
|
|
390
|
+
expect(config.temperature).toBe(0);
|
|
391
|
+
});
|
|
392
|
+
|
|
373
393
|
test("propagates resolved effort/speed/temperature; omits server-side fields", async () => {
|
|
374
394
|
setLlmConfig({
|
|
375
395
|
callSites: {
|
|
@@ -85,7 +85,17 @@ export function detectOpenAICompatibleContextOverflow(
|
|
|
85
85
|
/context.?length.?exceeded|context.?window.?exceeded|prompt.?is.?too.?long|prompt_too_long|input.?too.?long|too.?many.?(?:input.?)?tokens|maximum.?context/i.test(
|
|
86
86
|
message,
|
|
87
87
|
);
|
|
88
|
-
|
|
88
|
+
// string_above_max_length is OpenAI's generic oversized-string validation
|
|
89
|
+
// code, so only treat it as overflow when the error points at a message
|
|
90
|
+
// content part (e.g. "Invalid 'input[191].content[1].text': string too
|
|
91
|
+
// long" — OpenAI's per-part 10 MiB cap). The overflow ladder can shrink
|
|
92
|
+
// message content (media stubbing collapses a file's extracted_text to a
|
|
93
|
+
// preview) but cannot fix other oversized fields like tool definitions.
|
|
94
|
+
const oversizedContentPart =
|
|
95
|
+
/string.?too.?long|string_above_max_length/i.test(
|
|
96
|
+
`${code ?? ""} ${message}`,
|
|
97
|
+
) && /\b(?:input|messages)\[\d+\]\.content/i.test(message);
|
|
98
|
+
if (!codeMatches && !messageMatches && !oversizedContentPart) {
|
|
89
99
|
return null;
|
|
90
100
|
}
|
|
91
101
|
// OpenAI-compatible providers rarely report usable token counts; best-effort extract.
|
|
@@ -185,7 +185,6 @@ describe("DeepgramFluxRealtimeTranscriber", () => {
|
|
|
185
185
|
const transcriber = new DeepgramFluxRealtimeTranscriber(TEST_API_KEY, {
|
|
186
186
|
// Long enough that no watchdog fires mid-test.
|
|
187
187
|
inactivityTimeoutMs: 60_000,
|
|
188
|
-
keepaliveIntervalMs: 0,
|
|
189
188
|
...options,
|
|
190
189
|
});
|
|
191
190
|
const events: SttStreamServerEvent[] = [];
|
|
@@ -571,16 +570,22 @@ describe("DeepgramFluxRealtimeTranscriber", () => {
|
|
|
571
570
|
expect(events).toEqual([{ type: "closed" }]);
|
|
572
571
|
});
|
|
573
572
|
|
|
574
|
-
test("
|
|
575
|
-
const { transcriber } = await startSession(
|
|
573
|
+
test("no KeepAlive is ever sent: Flux rejects it and closes", async () => {
|
|
574
|
+
const { transcriber } = await startSession();
|
|
576
575
|
|
|
576
|
+
// Flux accepts only CloseStream and Configure. A KeepAlive earns an
|
|
577
|
+
// UNPARSABLE_CLIENT_MESSAGE error frame and a server close, which on a
|
|
578
|
+
// stream held across turns kills it every keepalive interval.
|
|
577
579
|
await Bun.sleep(35);
|
|
578
580
|
transcriber.stop();
|
|
579
581
|
|
|
580
|
-
const
|
|
581
|
-
(data) => data ===
|
|
582
|
+
const controlFrames = mockWs.sentData.filter(
|
|
583
|
+
(data) => typeof data === "string",
|
|
584
|
+
);
|
|
585
|
+
expect(controlFrames).not.toContain(
|
|
586
|
+
JSON.stringify({ type: "KeepAlive" }),
|
|
582
587
|
);
|
|
583
|
-
expect(
|
|
588
|
+
expect(controlFrames).toContain(JSON.stringify({ type: "CloseStream" }));
|
|
584
589
|
});
|
|
585
590
|
|
|
586
591
|
test("finalizeUtterance is absent, Flux has no mid-stream flush", async () => {
|