@vellumai/assistant 0.11.4-staging.1 → 0.11.4-staging.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. package/openapi.yaml +1 -1
  2. package/package.json +1 -1
  3. package/src/__tests__/byok-default-profile-ensure.test.ts +15 -0
  4. package/src/__tests__/conversation-queue.test.ts +122 -0
  5. package/src/__tests__/ui-shape-teaching.test.ts +33 -0
  6. package/src/__tests__/ui-voice-picker-surface.test.ts +128 -0
  7. package/src/__tests__/voice-config-update.test.ts +40 -0
  8. package/src/__tests__/workspace-migration-146-repair-retired-fireworks-deepseek-flash-model-id.test.ts +235 -0
  9. package/src/api/surfaces.ts +7 -3
  10. package/src/calls/call-controller.ts +16 -7
  11. package/src/config/bundled-skills/settings/tools/navigate-settings-tab.test.ts +65 -0
  12. package/src/config/bundled-skills/settings/tools/navigate-settings-tab.ts +7 -1
  13. package/src/config/bundled-skills/settings/tools/shared.ts +16 -0
  14. package/src/config/bundled-skills/settings/tools/voice-config-update.ts +19 -1
  15. package/src/config/default-profile-catalog.ts +1 -1
  16. package/src/daemon/conversation-process.ts +27 -0
  17. package/src/daemon/conversation-surfaces.ts +27 -5
  18. package/src/daemon/interactive-turn-sender.ts +59 -0
  19. package/src/daemon/process-message.ts +11 -21
  20. package/src/plugin-api/vision-support.test.ts +1 -1
  21. package/src/providers/model-catalog.ts +3 -3
  22. package/src/providers/model-intents.ts +2 -2
  23. package/src/providers/openai/chat-completions-provider.ts +5 -6
  24. package/src/tools/ui-surface/surface-shape-docs.ts +11 -0
  25. package/src/tts/__tests__/reasoning-tag-filter.test.ts +15 -0
  26. package/src/tts/reasoning-tag-filter.ts +11 -32
  27. package/src/util/think-tag-stream.ts +95 -0
  28. package/src/workspace/byok-default-profile-ensure.ts +15 -6
  29. package/src/workspace/migrations/146-repair-retired-fireworks-deepseek-flash-model-id.ts +195 -0
  30. package/src/workspace/migrations/registry.ts +2 -0
@@ -730,8 +730,10 @@ export class CallController {
730
730
  let ttsBuffer = "";
731
731
  let fullResponseText = "";
732
732
  // Reasoning models can inline <think> spans in the content stream when a
733
- // profile has not opted into parseThinkTags; the spoken path must never
734
- // read them aloud. fullResponseText keeps the raw stream.
733
+ // profile has not opted into parseThinkTags. Neither the spoken path nor
734
+ // the post-turn consumers of fullResponseText (transcripts,
735
+ // assistant_spoke, END_CALL/ASK_GUARDIAN detection) may see them: both
736
+ // are fed only filtered text.
735
737
  const reasoningFilter = createReasoningTagFilter();
736
738
 
737
739
  // Synthesized path: text is split at speakable boundaries as it streams
@@ -935,8 +937,13 @@ export class CallController {
935
937
  if (!this.isCurrentRun(runVersion)) {
936
938
  return;
937
939
  }
938
- fullResponseText += text;
939
- ttsBuffer += reasoningFilter.push(text);
940
+ // One filter feeds both consumers: the spoken stream and the text
941
+ // used for transcripts, assistant_spoke, and END_CALL/ASK_GUARDIAN
942
+ // marker detection. A control marker inside a reasoning span must
943
+ // never trigger a real action the caller did not hear.
944
+ const speakable = reasoningFilter.push(text);
945
+ fullResponseText += speakable;
946
+ ttsBuffer += speakable;
940
947
  ttsBuffer = stripInternalSpeechMarkers(ttsBuffer);
941
948
  flushSafeText();
942
949
  };
@@ -1015,9 +1022,11 @@ export class CallController {
1015
1022
  return fullResponseText;
1016
1023
  }
1017
1024
 
1018
- // Final sweep: release any held-back partial tag, then strip any
1019
- // remaining control markers from the buffer.
1020
- ttsBuffer += reasoningFilter.flush();
1025
+ // Final sweep: release any held-back partial tag to both consumers,
1026
+ // then strip any remaining control markers from the buffer.
1027
+ const filterTail = reasoningFilter.flush();
1028
+ fullResponseText += filterTail;
1029
+ ttsBuffer += filterTail;
1021
1030
  ttsBuffer = stripInternalSpeechMarkers(ttsBuffer);
1022
1031
  if (ttsBuffer.length > 0) {
1023
1032
  emitSafeChunk(ttsBuffer);
@@ -0,0 +1,65 @@
1
+ import { beforeEach, describe, expect, test } from "bun:test";
2
+
3
+ import type { ToolContext } from "../../../../tools/types.js";
4
+ import { run } from "./navigate-settings-tab.js";
5
+
6
+ let sentMessages: Array<{ type: string; [key: string]: unknown }> = [];
7
+
8
+ function makeContext(overrides?: Partial<ToolContext>): ToolContext {
9
+ return {
10
+ workingDir: "/tmp",
11
+ conversationId: "conv-xyz",
12
+ trustClass: "guardian",
13
+ sendToClient: (msg) => {
14
+ sentMessages.push(msg);
15
+ },
16
+ ...overrides,
17
+ };
18
+ }
19
+
20
+ describe("navigate_settings_tab tool", () => {
21
+ beforeEach(() => {
22
+ sentMessages = [];
23
+ });
24
+
25
+ test("navigates to the Voice tab and hints the inline picker", async () => {
26
+ const result = await run({ tab: "Voice" }, makeContext());
27
+
28
+ expect(result.isError).toBe(false);
29
+ expect(sentMessages).toEqual([{ type: "navigate_settings", tab: "Voice" }]);
30
+ expect(result.content).toStartWith("Opened settings to the Voice tab.");
31
+ expect(result.content).toContain(
32
+ 'ui_show { surface_type: "voice_picker", data: {} }',
33
+ );
34
+ });
35
+
36
+ test("leaves a non-Voice tab's result unchanged", async () => {
37
+ const result = await run({ tab: "Billing" }, makeContext());
38
+
39
+ expect(result.isError).toBe(false);
40
+ expect(sentMessages).toEqual([
41
+ { type: "navigate_settings", tab: "Billing" },
42
+ ]);
43
+ expect(result.content).toBe("Opened settings to the Billing tab.");
44
+ expect(result.content).not.toContain("voice_picker");
45
+ });
46
+
47
+ test("resolves legacy tab aliases without hinting", async () => {
48
+ const result = await run({ tab: "Archived Conversations" }, makeContext());
49
+
50
+ expect(result.isError).toBe(false);
51
+ expect(sentMessages).toEqual([
52
+ { type: "navigate_settings", tab: "Archive" },
53
+ ]);
54
+ expect(result.content).toBe("Opened settings to the Archive tab.");
55
+ expect(result.content).not.toContain("voice_picker");
56
+ });
57
+
58
+ test("rejects an unknown tab without navigating", async () => {
59
+ const result = await run({ tab: "Telepathy" }, makeContext());
60
+
61
+ expect(result.isError).toBe(true);
62
+ expect(result.content).toContain('unknown tab "Telepathy"');
63
+ expect(sentMessages).toEqual([]);
64
+ });
65
+ });
@@ -2,6 +2,7 @@ import type {
2
2
  ToolContext,
3
3
  ToolExecutionResult,
4
4
  } from "../../../../tools/types.js";
5
+ import { voicePickerHint } from "./shared.js";
5
6
 
6
7
  const SETTINGS_TABS = [
7
8
  "General",
@@ -23,6 +24,10 @@ const LEGACY_TAB_ALIASES: Record<string, SettingsTab> = {
23
24
  "Archived Conversations": "Archive",
24
25
  };
25
26
 
27
+ const VOICE_TAB_PICKER_HINT = voicePickerHint(
28
+ "which puts the picker in the conversation without navigating away. Navigating here is right only when the user explicitly asked to open Settings.",
29
+ );
30
+
26
31
  export async function run(
27
32
  input: Record<string, unknown>,
28
33
  context: ToolContext,
@@ -45,8 +50,9 @@ export async function run(
45
50
  });
46
51
  }
47
52
 
53
+ const opened = `Opened settings to the ${tab} tab.`;
48
54
  return {
49
- content: `Opened settings to the ${tab} tab.`,
55
+ content: tab === "Voice" ? `${opened} ${VOICE_TAB_PICKER_HINT}` : opened,
50
56
  isError: false,
51
57
  };
52
58
  }
@@ -0,0 +1,16 @@
1
+ /**
2
+ * Shared helpers for the settings skill's tool executors.
3
+ */
4
+
5
+ /**
6
+ * Steer the model back to the inline voice picker.
7
+ *
8
+ * Both settings tools a "change my voice" request can land on (the Voice tab
9
+ * navigation and a managed `tts_voice_id` write) end by naming the picker, so
10
+ * the invocation itself lives here: one edit keeps both tools teaching the same
11
+ * call. Each caller supplies its own tail, because the reason the picker is
12
+ * better differs by the tool the model just reached for.
13
+ */
14
+ export function voicePickerHint(tail: string): string {
15
+ return `Next time the user wants to change or hear a voice, prefer \`ui_show { surface_type: "voice_picker", data: {} }\`, ${tail}`;
16
+ }
@@ -19,6 +19,7 @@ import {
19
19
  } from "../../../loader.js";
20
20
  import { VALID_CONVERSATION_TIMEOUTS } from "../../../schemas/elevenlabs.js";
21
21
  import { VALID_STT_PROVIDERS } from "../../../schemas/stt.js";
22
+ import { voicePickerHint } from "./shared.js";
22
23
 
23
24
  /**
24
25
  * Valid voice config settings and their UserDefaults key mappings.
@@ -187,6 +188,19 @@ const STT_LANGUAGE_ALIASES: Record<
187
188
  "code-switching": "multi",
188
189
  };
189
190
 
191
+ /**
192
+ * Same steer `navigate_settings_tab` puts on the Voice tab, applied to the
193
+ * strongest competing affordance: this tool's description coaches the model to
194
+ * read the managed catalog and set `tts_voice_id` itself, which is exactly the
195
+ * "assistant names voices in prose" outcome the inline picker exists to
196
+ * replace. Managed only, because that is the catalog the picker renders. The
197
+ * write still happens and the result stays non-error; this only redirects the
198
+ * next request.
199
+ */
200
+ const MANAGED_VOICE_PICKER_HINT = voicePickerHint(
201
+ "which lets them hear and pick from the managed catalog in the conversation. Set the id here only when the user named a specific voice.",
202
+ );
203
+
190
204
  function validateSetting(
191
205
  setting: string,
192
206
  value: unknown,
@@ -530,10 +544,14 @@ export async function run(
530
544
  AUTO_DETECT_STT_PROVIDERS.has(activeSttProviderId)
531
545
  ? ` Note: the configured STT provider (${activeSttProviderId}) auto-detects the spoken language natively and ignores this setting.`
532
546
  : "";
547
+ const pickerNote =
548
+ setting === "tts_voice_id" && activeTtsProviderId === "vellum"
549
+ ? ` ${MANAGED_VOICE_PICKER_HINT}`
550
+ : "";
533
551
  return {
534
552
  content: `${friendlyName} updated to ${JSON.stringify(
535
553
  validation.coerced,
536
- )}.${broadcastNote}${autoDetectNote}`,
554
+ )}.${broadcastNote}${autoDetectNote}${pickerNote}`,
537
555
  isError: false,
538
556
  };
539
557
  }
@@ -98,7 +98,7 @@ const VELLUM_PROFILE_IMPLS: ProfileImpls = {
98
98
  },
99
99
  },
100
100
  "cost-optimized": {
101
- model: "accounts/fireworks/models/deepseek-v4-flash",
101
+ model: "accounts/fireworks/models/deepseek-v4-flash-0731",
102
102
  provider: "vellum",
103
103
  source: "managed",
104
104
  label: "Cost",
@@ -60,6 +60,7 @@ import {
60
60
  } from "./conversation-slash.js";
61
61
  import { getModelInfo } from "./handlers/config-model.js";
62
62
  import { preactivateHostProxySkills } from "./host-proxy-preactivation.js";
63
+ import { bindInteractiveTurnSender } from "./interactive-turn-sender.js";
63
64
  import type { UserMessageAttachment } from "./message-protocol.js";
64
65
  import { buildTransportHints } from "./transport-hints.js";
65
66
  import { sameTrustIdentity, type TrustContext } from "./trust-context-types.js";
@@ -1292,11 +1293,26 @@ async function drainSingleMessage(
1292
1293
  drainLoopOptions.isHiddenPrompt = true;
1293
1294
  }
1294
1295
 
1296
+ // By the time the drain runs, the conversation-level sender is usually the
1297
+ // no-op the just-finished interactive turn's restore (or conversation
1298
+ // birth) left behind, and the drain bypasses `processMessage`, so without
1299
+ // a fresh bind the PermissionPrompter's confirmation_request (which reads
1300
+ // the conversation-level sender, not the turn's `onEvent`) would be
1301
+ // emitted into nothing and hang until the permission timeout. Same
1302
+ // bind/restore contract as `processMessage`.
1303
+ const restoreSender =
1304
+ next.isInteractive === true
1305
+ ? bindInteractiveTurnSender(conversation)
1306
+ : undefined;
1307
+
1295
1308
  conversation
1296
1309
  .runAgentLoop(agentLoopContent, userMessageId, {
1297
1310
  ...drainLoopOptions,
1298
1311
  onEvent: next.onEvent,
1299
1312
  })
1313
+ .finally(() => {
1314
+ restoreSender?.();
1315
+ })
1300
1316
  .catch((err) => {
1301
1317
  const message = err instanceof Error ? err.message : String(err);
1302
1318
  log.error(
@@ -1779,6 +1795,14 @@ async function drainBatch(
1779
1795
  drainLoopOptions.isHiddenPrompt = true;
1780
1796
  }
1781
1797
 
1798
+ // Same sender contract as drainSingleMessage: an interactive drained batch
1799
+ // must rebind the conversation-level sender or prompter confirmations
1800
+ // raised during this turn reach no client.
1801
+ const restoreSender =
1802
+ drainLoopOptions.isInteractive === true
1803
+ ? bindInteractiveTurnSender(conversation)
1804
+ : undefined;
1805
+
1782
1806
  // Fire-and-forget: runAgentLoop's finally block recursively calls drainQueue
1783
1807
  // when this run completes. Mirrors drainSingleMessage.
1784
1808
  conversation
@@ -1786,6 +1810,9 @@ async function drainBatch(
1786
1810
  ...drainLoopOptions,
1787
1811
  onEvent: fanOutOnEvent,
1788
1812
  })
1813
+ .finally(() => {
1814
+ restoreSender?.();
1815
+ })
1789
1816
  .catch((err) => {
1790
1817
  const message = err instanceof Error ? err.message : String(err);
1791
1818
  log.error(
@@ -140,14 +140,28 @@ const MAX_UNDO_DEPTH = 10;
140
140
 
141
141
  /**
142
142
  * Pending surface types that do not hold the one-interactive-surface-at-a-time
143
- * lock. Both render content the user reads rather than a question they must
144
- * answer, so a live one must not block the next surface.
143
+ * lock. Each renders content the user reads (or settles on its own) rather
144
+ * than a question they must answer, so a live one must not block the next
145
+ * surface.
145
146
  */
146
147
  const NON_BLOCKING_PENDING_SURFACE_TYPES = new Set<SurfaceType>([
147
148
  "dynamic_page",
148
149
  "visual",
150
+ "voice_picker",
149
151
  ]);
150
152
 
153
+ /**
154
+ * Surface types that carry no terminal action: the card settles when the user
155
+ * interacts with it, so no click could ever satisfy an attached `actions`
156
+ * entry or an explicit `await_action`. Both are generic ui_show params though,
157
+ * so nothing stops the model attaching them here, and doing so wedges the
158
+ * turn: the client latches `awaiting_user_input` on the presence of actions
159
+ * alone and no action ever arrives to clear it, leaving the composer disabled
160
+ * and Stop hidden while the daemon is still streaming. Stripping them makes
161
+ * the "never blocks a turn" contract structural rather than advisory.
162
+ */
163
+ const ACTIONLESS_SURFACE_TYPES = new Set<SurfaceType>(["voice_picker"]);
164
+
151
165
  /**
152
166
  * Debounce window for persisting `ui_surface_update` data back to the
153
167
  * message row. Surfaces typically receive bursts of updates (e.g. a
@@ -3279,8 +3293,12 @@ export async function surfaceProxyResolver(
3279
3293
  }
3280
3294
  inputActions = valid.length > 0 ? valid : undefined;
3281
3295
  }
3282
- const actions =
3283
- choiceData !== undefined ? buildChoiceActions(choiceData) : inputActions;
3296
+ const isActionless = ACTIONLESS_SURFACE_TYPES.has(surfaceType);
3297
+ const actions = isActionless
3298
+ ? undefined
3299
+ : choiceData !== undefined
3300
+ ? buildChoiceActions(choiceData)
3301
+ : inputActions;
3284
3302
  const hasActions = Array.isArray(actions) && actions.length > 0;
3285
3303
  if (surfaceType === "choice" && !hasActions) {
3286
3304
  return {
@@ -3332,7 +3350,11 @@ export async function surfaceProxyResolver(
3332
3350
  : surfaceType === "table"
3333
3351
  ? hasActions
3334
3352
  : INTERACTIVE_SURFACE_TYPES.includes(surfaceType);
3335
- const awaitAction = (input.await_action as boolean) ?? isInteractive;
3353
+ // An explicit `await_action: true` is honored for every other type; an
3354
+ // actionless surface has nothing to await, so it is forced false.
3355
+ const awaitAction = isActionless
3356
+ ? false
3357
+ : ((input.await_action as boolean) ?? isInteractive);
3336
3358
 
3337
3359
  // Only one non-persistent interactive surface at a time. If another
3338
3360
  // surface is already awaiting user input, reject this one so the LLM
@@ -0,0 +1,59 @@
1
+ /**
2
+ * The interactive-turn sender contract, in one place.
3
+ *
4
+ * `Conversation.sendToClient` is conversation-level mutable state: born a
5
+ * no-op, bound to the SSE hub for the duration of an interactive turn, and
6
+ * restored when that turn ends. Some subsystems read this conversation-level
7
+ * sender instead of the turn's own `onEvent` sink; the `PermissionPrompter`
8
+ * is the load-bearing one, so a `confirmation_request` raised by a turn that
9
+ * never bound the sender is emitted into the no-op, reaches no client, and
10
+ * hangs until the permission timeout auto-denies. Every path that runs an
11
+ * interactive agent turn MUST therefore bind before the loop and restore
12
+ * after it. The queue drain bypasses `processMessage`, so it uses this
13
+ * helper directly.
14
+ *
15
+ * The restore is a snapshot of the sender state taken at bind time, not an
16
+ * unconditional reset to a no-op: a live binding that predated the turn
17
+ * (e.g. installed by the send route while a non-interactive turn was
18
+ * running) survives the turn, so out-of-turn producers that read the live
19
+ * sender (call transcript and completion notifiers) keep their client. In
20
+ * the ordinary interactive flow the snapshot is the no-op the previous
21
+ * turn's restore left behind, so restoring it is identical to the reset
22
+ * this contract has always performed. The restore is also identity-guarded:
23
+ * it only replaces the hub binding this contract installed, so a sender
24
+ * some other subsystem rebound mid-turn to anything else is left alone.
25
+ *
26
+ * TRANSITIONAL: this helper is the paved road for a pattern we want less
27
+ * of. Bind/restore of shared mutable sender state is why the drain, the
28
+ * compaction card, and the host-browser proxy have each needed their own
29
+ * restore step. The durable design is a per-turn sink threaded to the
30
+ * prompter and the other conversation-level-sender readers, at which point
31
+ * this module and every call site of it should be deleted. Prefer that
32
+ * refactor over adding new bind/restore call sites.
33
+ */
34
+
35
+ import { broadcastMessage } from "../runtime/assistant-event-hub.js";
36
+ import { getSubagentManager } from "../subagent/index.js";
37
+ import type { Conversation } from "./conversation.js";
38
+
39
+ /**
40
+ * Point the conversation-level sender (and the subagent parent sender) at
41
+ * the SSE hub for the duration of an interactive turn. Returns the restore
42
+ * closure the turn's cleanup must invoke.
43
+ */
44
+ export function bindInteractiveTurnSender(
45
+ conversation: Conversation,
46
+ ): () => void {
47
+ const previousSender = conversation.getCurrentSender();
48
+ const previousHasNoClient = conversation.hasNoClient;
49
+ conversation.updateClient(broadcastMessage, false);
50
+ getSubagentManager().updateParentSender(
51
+ conversation.conversationId,
52
+ broadcastMessage,
53
+ );
54
+ return () => {
55
+ if (conversation.getCurrentSender() === broadcastMessage) {
56
+ conversation.updateClient(previousSender, previousHasNoClient);
57
+ }
58
+ };
59
+ }
@@ -34,7 +34,6 @@ import { updateMetaFile } from "../persistence/conversation-disk-view.js";
34
34
  import { broadcastMessage } from "../runtime/assistant-event-hub.js";
35
35
  import { DAEMON_INTERNAL_ASSISTANT_ID } from "../runtime/assistant-scope.js";
36
36
  import { publishConversationMessagesChanged } from "../runtime/sync/resource-sync-events.js";
37
- import { getSubagentManager } from "../subagent/index.js";
38
37
  import {
39
38
  readTurnFailure,
40
39
  type TurnFailure,
@@ -70,6 +69,7 @@ import {
70
69
  preactivateHostProxySkills,
71
70
  shouldAttachHostProxyForCapability,
72
71
  } from "./host-proxy-preactivation.js";
72
+ import { bindInteractiveTurnSender } from "./interactive-turn-sender.js";
73
73
  import type { SubagentToolGateMode } from "./tool-setup-types.js";
74
74
  import { restingTrust } from "./trust-context-types.js";
75
75
 
@@ -709,10 +709,10 @@ export async function processMessage(
709
709
  };
710
710
  }
711
711
 
712
- if (options?.isInteractive === true) {
713
- conversation.updateClient(broadcastMessage, false);
714
- getSubagentManager().updateParentSender(conversationId, broadcastMessage);
715
- }
712
+ const restoreSender =
713
+ options?.isInteractive === true
714
+ ? bindInteractiveTurnSender(conversation)
715
+ : undefined;
716
716
 
717
717
  const restoreToolScope = applyTurnToolAllowlist(conversation, options);
718
718
  try {
@@ -731,12 +731,7 @@ export async function processMessage(
731
731
  });
732
732
  } finally {
733
733
  restoreToolScope();
734
- if (
735
- options?.isInteractive === true &&
736
- conversation.getCurrentSender() === broadcastMessage
737
- ) {
738
- conversation.updateClient(() => {}, true);
739
- }
734
+ restoreSender?.();
740
735
  }
741
736
 
742
737
  // Read the just-finished turn's outcome from the stamp `runAgentLoop`'s
@@ -795,10 +790,10 @@ export async function processMessageInBackground(
795
790
  return { messageId };
796
791
  }
797
792
 
798
- if (options?.isInteractive === true) {
799
- conversation.updateClient(broadcastMessage, false);
800
- getSubagentManager().updateParentSender(conversationId, broadcastMessage);
801
- }
793
+ const restoreSender =
794
+ options?.isInteractive === true
795
+ ? bindInteractiveTurnSender(conversation)
796
+ : undefined;
802
797
 
803
798
  const restoreToolScope = applyTurnToolAllowlist(conversation, options);
804
799
  conversation
@@ -814,12 +809,7 @@ export async function processMessageInBackground(
814
809
  })
815
810
  .finally(() => {
816
811
  restoreToolScope();
817
- if (
818
- options?.isInteractive === true &&
819
- conversation.getCurrentSender() === broadcastMessage
820
- ) {
821
- conversation.updateClient(() => {}, true);
822
- }
812
+ restoreSender?.();
823
813
  })
824
814
  .catch((err) => {
825
815
  log.error({ err, conversationId }, "Background agent loop failed");
@@ -107,7 +107,7 @@ describe("doesSupportVision", () => {
107
107
  managed: { provider: "vellum", model: "claude-fable-5" },
108
108
  "managed-text": {
109
109
  provider: "vellum",
110
- model: "accounts/fireworks/models/deepseek-v4-flash",
110
+ model: "accounts/fireworks/models/deepseek-v4-flash-0731",
111
111
  },
112
112
  });
113
113
  expect(doesSupportVision(profile("managed"))).toBe(true);
@@ -939,7 +939,7 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
939
939
  pricing: { inputPer1mTokens: 1.74, outputPer1mTokens: 3.48 },
940
940
  },
941
941
  {
942
- id: "accounts/fireworks/models/deepseek-v4-flash",
942
+ id: "accounts/fireworks/models/deepseek-v4-flash-0731",
943
943
  displayName: "DeepSeek V4 Flash",
944
944
  contextWindowTokens: 1040000,
945
945
  maxOutputTokens: 131072,
@@ -951,11 +951,11 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
951
951
  pricing: {
952
952
  inputPer1mTokens: 0.14,
953
953
  outputPer1mTokens: 0.28,
954
- cacheReadPer1mTokens: 0.03,
954
+ cacheReadPer1mTokens: 0.028,
955
955
  },
956
956
  },
957
957
  ],
958
- defaultModel: "accounts/fireworks/models/deepseek-v4-flash",
958
+ defaultModel: "accounts/fireworks/models/deepseek-v4-flash-0731",
959
959
  apiKeyUrl: "https://fireworks.ai/account/api-keys",
960
960
  apiKeyPlaceholder: "fw_...",
961
961
  },
@@ -46,8 +46,8 @@ const PROVIDER_MODEL_INTENTS: Record<string, Record<ModelIntent, string>> = {
46
46
  },
47
47
  fireworks: {
48
48
  balanced: "accounts/fireworks/models/minimax-m3",
49
- "cost-optimized": "accounts/fireworks/models/deepseek-v4-flash",
50
- "latency-optimized": "accounts/fireworks/models/deepseek-v4-flash",
49
+ "cost-optimized": "accounts/fireworks/models/deepseek-v4-flash-0731",
50
+ "latency-optimized": "accounts/fireworks/models/deepseek-v4-flash-0731",
51
51
  "quality-optimized": "accounts/fireworks/models/kimi-k2p6",
52
52
  "vision-optimized": "accounts/fireworks/models/kimi-k2p6",
53
53
  },
@@ -5,6 +5,7 @@ import { isAbortReason } from "../../util/abort-reasons.js";
5
5
  import { ProviderError, type ProviderErrorReason } from "../../util/errors.js";
6
6
  import { getLogger } from "../../util/logger.js";
7
7
  import { extractRetryAfterMs } from "../../util/retry.js";
8
+ import { partialTagSuffix as sharedPartialTagSuffix } from "../../util/think-tag-stream.js";
8
9
  import { escapeXmlAttr } from "../../util/xml.js";
9
10
  import {
10
11
  base64Source,
@@ -293,13 +294,11 @@ const OPENAI_SUPPORTED_IMAGE_TYPES = new Set([
293
294
  "image/webp",
294
295
  ]);
295
296
 
297
+ // Think-tag scanning primitives are shared with the TTS reasoning filter
298
+ // (util/think-tag-stream.ts) so the two stream parsers cannot drift. This
299
+ // provider keeps its exact historical behavior: case-sensitive, <think> only.
296
300
  function partialTagSuffix(text: string, tag: string): number {
297
- for (let len = Math.min(text.length, tag.length - 1); len > 0; len--) {
298
- if (text.endsWith(tag.substring(0, len))) {
299
- return len;
300
- }
301
- }
302
- return 0;
301
+ return sharedPartialTagSuffix(text, [tag], false);
303
302
  }
304
303
 
305
304
  /**
@@ -211,6 +211,17 @@ export const SURFACE_SHAPE_DOCS: Record<string, SurfaceShapeDoc> = {
211
211
  ? null
212
212
  : '`data.channel` must be one of "slack", "telegram", "phone"',
213
213
  },
214
+ voice_picker: {
215
+ // The steering lives in `purpose` because that is the only part of a cold
216
+ // type the model ever sees: `shape` ships solely inside a teaching error,
217
+ // and an empty payload has no missing-content state to raise one. The
218
+ // surface exists to replace prose voice tours, so losing the directive
219
+ // would leave nothing but a renderer nobody invokes.
220
+ purpose:
221
+ "inline picker for the voice the assistant speaks in; show it when the user asks to change, hear, or pick a voice rather than describing voices in prose or sending them to Settings; attach no actions and never set await_action; during a live voice call answer in speech instead, the room already carries its own picker",
222
+ shape:
223
+ "{}: no payload, the card reads the current voice and the catalog itself. Selecting a voice applies on the assistant's next spoken turn",
224
+ },
214
225
  };
215
226
 
216
227
  /** Model-facing surface_type enum, derived so docs and schema cannot drift. */
@@ -60,4 +60,19 @@ describe("ReasoningTagFilter", () => {
60
60
  test("holds back nothing across an empty stream", () => {
61
61
  expect(run([""])).toBe("");
62
62
  });
63
+
64
+ test("does not shift indexes for characters whose lowercase form is longer", () => {
65
+ // U+0130 (dotted capital I) lowers to two code units; positional
66
+ // scanning must never lowercase the haystack wholesale.
67
+ expect(run(["\u0130stanbul rocks. <think>hidden</think> Done."])).toBe(
68
+ "\u0130stanbul rocks. Done.",
69
+ );
70
+ expect(run(["\u0130\u0130\u0130<THINK>x</THINK>ok"])).toBe(
71
+ "\u0130\u0130\u0130ok",
72
+ );
73
+ });
74
+
75
+ test("prefers the longer tag when both match at one index", () => {
76
+ expect(run(["a<thinking>x</thinking>b"])).toBe("ab");
77
+ });
63
78
  });
@@ -18,39 +18,18 @@
18
18
  * Display paths are deliberately untouched: `assistant_text_delta` frames
19
19
  * keep the raw text, exactly like the markdown handling in
20
20
  * `calls/tts-text-sanitizer.ts`.
21
+ *
22
+ * Tag scanning lives in `util/think-tag-stream.ts`, shared with the
23
+ * chat-completions provider's think-tag router, and is positionally exact
24
+ * on the original string (no full-string lowercasing, which shifts indexes
25
+ * for characters like the dotted capital I).
21
26
  */
22
27
 
28
+ import { indexOfTag, partialTagSuffix } from "../util/think-tag-stream.js";
29
+
23
30
  const OPEN_TAGS = ["<think>", "<thinking>"] as const;
24
31
  const CLOSE_TAGS = ["</think>", "</thinking>"] as const;
25
32
 
26
- /** Longest suffix of `text` that is a proper prefix of any listed tag. */
27
- function partialTagSuffix(text: string, tags: readonly string[]): number {
28
- const max = Math.max(...tags.map((tag) => tag.length)) - 1;
29
- const window = Math.min(max, text.length);
30
- for (let len = window; len > 0; len -= 1) {
31
- const suffix = text.slice(text.length - len).toLowerCase();
32
- if (tags.some((tag) => tag.startsWith(suffix))) {
33
- return len;
34
- }
35
- }
36
- return 0;
37
- }
38
-
39
- function indexOfAny(
40
- haystack: string,
41
- tags: readonly string[],
42
- ): { index: number; tag: string } | null {
43
- const lower = haystack.toLowerCase();
44
- let best: { index: number; tag: string } | null = null;
45
- for (const tag of tags) {
46
- const index = lower.indexOf(tag);
47
- if (index >= 0 && (best === null || index < best.index)) {
48
- best = { index, tag };
49
- }
50
- }
51
- return best;
52
- }
53
-
54
33
  export class ReasoningTagFilter {
55
34
  private insideReasoning = false;
56
35
  private pending = "";
@@ -65,7 +44,7 @@ export class ReasoningTagFilter {
65
44
  let out = "";
66
45
  for (;;) {
67
46
  if (this.insideReasoning) {
68
- const close = indexOfAny(this.pending, CLOSE_TAGS);
47
+ const close = indexOfTag(this.pending, CLOSE_TAGS, true);
69
48
  if (close) {
70
49
  this.pending = this.pending.slice(close.index + close.tag.length);
71
50
  this.insideReasoning = false;
@@ -73,18 +52,18 @@ export class ReasoningTagFilter {
73
52
  }
74
53
  // Everything buffered is reasoning except a possible partial close
75
54
  // tag at the end; drop the reasoning, keep the partial.
76
- const partial = partialTagSuffix(this.pending, CLOSE_TAGS);
55
+ const partial = partialTagSuffix(this.pending, CLOSE_TAGS, true);
77
56
  this.pending = partial > 0 ? this.pending.slice(-partial) : "";
78
57
  return out;
79
58
  }
80
- const open = indexOfAny(this.pending, OPEN_TAGS);
59
+ const open = indexOfTag(this.pending, OPEN_TAGS, true);
81
60
  if (open) {
82
61
  out += this.pending.slice(0, open.index);
83
62
  this.pending = this.pending.slice(open.index + open.tag.length);
84
63
  this.insideReasoning = true;
85
64
  continue;
86
65
  }
87
- const partial = partialTagSuffix(this.pending, OPEN_TAGS);
66
+ const partial = partialTagSuffix(this.pending, OPEN_TAGS, true);
88
67
  const safeLength = this.pending.length - partial;
89
68
  out += this.pending.slice(0, safeLength);
90
69
  this.pending = partial > 0 ? this.pending.slice(safeLength) : "";