@vellumai/assistant 0.11.5-staging.1 → 0.11.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vellumai/assistant",
3
- "version": "0.11.5-staging.1",
3
+ "version": "0.11.5",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "exports": {
@@ -49,6 +49,7 @@ afterAll(() => {
49
49
  });
50
50
 
51
51
  import {
52
+ CODE_DEFAULT_PROFILE_ENTRIES,
52
53
  getEffectiveProfile,
53
54
  getEffectiveProfiles,
54
55
  getEffectiveProfilesForProvider,
@@ -164,6 +165,14 @@ const LEGACY_HATCH_PROFILE_NAMES = [
164
165
  "custom-cost-optimized",
165
166
  ] as const;
166
167
 
168
+ /**
169
+ * The managed Balanced model. These tests assert that resolution serves the
170
+ * code catalog rather than a workspace body, so they follow the catalog
171
+ * instead of restating its pin.
172
+ */
173
+ const MANAGED_BALANCED_MODEL = CODE_DEFAULT_PROFILE_ENTRIES.balanced
174
+ .model as string;
175
+
167
176
  function createProviderConnectionsDb(): DrizzleDb {
168
177
  const sqlite = new Database(":memory:");
169
178
  sqlite.exec("PRAGMA journal_mode=WAL");
@@ -669,7 +678,7 @@ describe("loadConfig startup behavior", () => {
669
678
  expect(raw.llm.profiles).toEqual({});
670
679
  // Default content resolves from the code catalog via the effective view.
671
680
  const effectiveBalanced = getEffectiveProfile(raw.llm.profiles, "balanced");
672
- expect(effectiveBalanced?.model).toBe("gpt-5.6-luna");
681
+ expect(effectiveBalanced?.model).toBe(MANAGED_BALANCED_MODEL);
673
682
  expect(effectiveBalanced?.provider).toBe("vellum");
674
683
  expect(effectiveBalanced?.provider_connection).toBeUndefined();
675
684
  });
@@ -997,7 +1006,7 @@ describe("loadConfig startup behavior", () => {
997
1006
  // Resolution ignores the drifted body: a managed-source entry contributes
998
1007
  // only label/status/topP, everything else comes from the catalog.
999
1008
  const effectiveBalanced = getEffectiveProfile(raw.llm.profiles, "balanced");
1000
- expect(effectiveBalanced?.model).toBe("gpt-5.6-luna");
1009
+ expect(effectiveBalanced?.model).toBe(MANAGED_BALANCED_MODEL);
1001
1010
  expect(effectiveBalanced?.provider_connection).toBeUndefined();
1002
1011
  });
1003
1012
 
@@ -1077,7 +1086,7 @@ describe("loadConfig startup behavior", () => {
1077
1086
  expect(raw.llm.profiles.balanced).toEqual(drifted);
1078
1087
  expect(raw.llm.activeProfile).toBe("balanced");
1079
1088
  const effectiveBalanced = getEffectiveProfile(raw.llm.profiles, "balanced");
1080
- expect(effectiveBalanced?.model).toBe("gpt-5.6-luna");
1089
+ expect(effectiveBalanced?.model).toBe(MANAGED_BALANCED_MODEL);
1081
1090
  expect(effectiveBalanced?.maxTokens).toBe(32000);
1082
1091
  expect(effectiveBalanced?.provider_connection).toBeUndefined();
1083
1092
  // The catalog body carries no topP and the entry has none, so the
@@ -1109,7 +1118,7 @@ describe("loadConfig startup behavior", () => {
1109
1118
  expect(raw.llm.profiles.balanced).toEqual(edited);
1110
1119
  const effectiveBalanced = getEffectiveProfile(raw.llm.profiles, "balanced");
1111
1120
  // Content is served from the catalog...
1112
- expect(effectiveBalanced?.model).toBe("gpt-5.6-luna");
1121
+ expect(effectiveBalanced?.model).toBe(MANAGED_BALANCED_MODEL);
1113
1122
  // ...with the user's label and status overlaid.
1114
1123
  expect(effectiveBalanced?.label).toBe("My Default");
1115
1124
  expect(effectiveBalanced?.status).toBe("disabled");
@@ -1134,7 +1143,7 @@ describe("loadConfig startup behavior", () => {
1134
1143
  expect(raw.llm.profiles.balanced).toEqual(stub);
1135
1144
  const effectiveBalanced = getEffectiveProfile(raw.llm.profiles, "balanced");
1136
1145
  expect(effectiveBalanced?.label).toBe("My Default");
1137
- expect(effectiveBalanced?.model).toBe("gpt-5.6-luna");
1146
+ expect(effectiveBalanced?.model).toBe(MANAGED_BALANCED_MODEL);
1138
1147
  });
1139
1148
 
1140
1149
  test("off-platform boot preserves user-toggled status on a managed stub", () => {
@@ -1156,7 +1165,7 @@ describe("loadConfig startup behavior", () => {
1156
1165
  expect(effectiveBalanced?.status).toBe("disabled");
1157
1166
  // Content still comes from the catalog — only label/status/topP are
1158
1167
  // workspace-owned.
1159
- expect(effectiveBalanced?.model).toBe("gpt-5.6-luna");
1168
+ expect(effectiveBalanced?.model).toBe(MANAGED_BALANCED_MODEL);
1160
1169
  });
1161
1170
 
1162
1171
  test("boot preserves a user-edited topP override on a managed stub", () => {
@@ -1180,7 +1189,7 @@ describe("loadConfig startup behavior", () => {
1180
1189
  expect(effectiveBalanced?.topP).toBe(0.5);
1181
1190
  // Content still comes from the catalog — topP is workspace-owned, the
1182
1191
  // rest is code-owned.
1183
- expect(effectiveBalanced?.model).toBe("gpt-5.6-luna");
1192
+ expect(effectiveBalanced?.model).toBe(MANAGED_BALANCED_MODEL);
1184
1193
  });
1185
1194
 
1186
1195
  test("effective balanced profile carries no topP override by default", () => {
@@ -1235,7 +1244,7 @@ describe("loadConfig startup behavior", () => {
1235
1244
  expect(raw.llm.profiles.balanced).toBeUndefined();
1236
1245
  const effectiveBalanced = getEffectiveProfile(raw.llm.profiles, "balanced");
1237
1246
  expect(effectiveBalanced?.label).toBe("Balanced");
1238
- expect(effectiveBalanced?.model).toBe("gpt-5.6-luna");
1247
+ expect(effectiveBalanced?.model).toBe(MANAGED_BALANCED_MODEL);
1239
1248
  // Status is unset — the default resolves active.
1240
1249
  expect("status" in (effectiveBalanced ?? {})).toBe(false);
1241
1250
  });
@@ -1314,7 +1323,7 @@ describe("loadConfig startup behavior", () => {
1314
1323
  // overlay boot. The overlay-set label is what shows through the
1315
1324
  // effective view.
1316
1325
  expect(mainAgentConfig.provider).toBe("vellum");
1317
- expect(mainAgentConfig.model).toBe("gpt-5.6-luna");
1326
+ expect(mainAgentConfig.model).toBe(MANAGED_BALANCED_MODEL);
1318
1327
 
1319
1328
  const raw = JSON.parse(readFileSync(CONFIG_PATH, "utf-8"));
1320
1329
  expect(raw.llm.profiles.balanced).toEqual({
@@ -1345,7 +1354,7 @@ describe("loadConfig startup behavior", () => {
1345
1354
  );
1346
1355
  expect(effectiveBalanced?.provider).toBe("vellum");
1347
1356
  expect(effectiveBalanced?.provider_connection).toBeUndefined();
1348
- expect(effectiveBalanced?.model).toBe("gpt-5.6-luna");
1357
+ expect(effectiveBalanced?.model).toBe(MANAGED_BALANCED_MODEL);
1349
1358
  expect(effectiveBalanced?.maxTokens).toBe(32000);
1350
1359
  expect(effectiveBalanced?.thinking).toEqual({
1351
1360
  enabled: true,
@@ -1090,7 +1090,7 @@ describe("code-owned default profiles — wire view and write normalization", ()
1090
1090
  });
1091
1091
  const response = (await getRoute.handler({})) as Record<string, any>;
1092
1092
  const wireBalanced = response.llm.profiles.balanced;
1093
- expect(wireBalanced.model).toBe("gpt-5.6-luna");
1093
+ expect(wireBalanced.model).toBe("accounts/fireworks/models/glm-5p2");
1094
1094
  expect(wireBalanced.provider).toBe("vellum");
1095
1095
  expect(wireBalanced.provider_connection).toBeUndefined();
1096
1096
  expect(wireBalanced.status).toBe("disabled");
@@ -1337,7 +1337,9 @@ describe("code-owned default profiles — echoes over stale on-disk bodies", ()
1337
1337
  });
1338
1338
  const response = (await getRoute.handler({})) as Record<string, any>;
1339
1339
  // The wire view serves catalog content, not the stale body.
1340
- expect(response.llm.profiles.balanced.model).toBe("gpt-5.6-luna");
1340
+ expect(response.llm.profiles.balanced.model).toBe(
1341
+ "accounts/fireworks/models/glm-5p2",
1342
+ );
1341
1343
  const result = await patchRoute.handler({
1342
1344
  body: { llm: { profiles: response.llm.profiles } },
1343
1345
  });
@@ -135,6 +135,8 @@ function createMockSession(opts?: {
135
135
  };
136
136
  },
137
137
  handleSecretResponse: () => {},
138
+ // The image-bearing profile pin reads the leg's history.
139
+ getMessages: () => [],
138
140
  runAgentLoop: async (
139
141
  _content: string,
140
142
  _messageId: string,
@@ -74,6 +74,8 @@ function makeStreamingSession(events: AssistantEvent[]): Conversation {
74
74
  }
75
75
  },
76
76
  handleConfirmationResponse: () => {},
77
+ // The image-bearing profile pin reads the leg's history.
78
+ getMessages: () => [],
77
79
  abort: () => {},
78
80
  } as unknown as Conversation;
79
81
  }
@@ -133,6 +135,8 @@ function makePersistingStreamingSession(
133
135
  },
134
136
  handleConfirmationResponse: () => {},
135
137
  abort: () => {},
138
+ // The image-bearing profile pin reads the leg's history.
139
+ getMessages: () => [],
136
140
  } as unknown as Conversation &
137
141
  PersistUserMessageContext & {
138
142
  callSessionId?: string;
@@ -263,6 +267,8 @@ describe("voice-session-bridge", () => {
263
267
  abort: () => {
264
268
  abortCalled = true;
265
269
  },
270
+ // The image-bearing profile pin reads the leg's history.
271
+ getMessages: () => [],
266
272
  } as unknown as Conversation;
267
273
 
268
274
  injectDeps(() => session);
@@ -302,6 +308,8 @@ describe("voice-session-bridge", () => {
302
308
  onEvent(event);
303
309
  }
304
310
  },
311
+ // The image-bearing profile pin reads the leg's history.
312
+ getMessages: () => [],
305
313
  } as unknown as Conversation;
306
314
 
307
315
  injectDeps(() => session);
@@ -348,6 +356,8 @@ describe("voice-session-bridge", () => {
348
356
  onEvent(event);
349
357
  }
350
358
  },
359
+ // The image-bearing profile pin reads the leg's history.
360
+ getMessages: () => [],
351
361
  } as unknown as Conversation;
352
362
 
353
363
  injectDeps(() => session);
@@ -393,6 +403,8 @@ describe("voice-session-bridge", () => {
393
403
  abort: () => {
394
404
  abortCalled = true;
395
405
  },
406
+ // The image-bearing profile pin reads the leg's history.
407
+ getMessages: () => [],
396
408
  } as unknown as Conversation;
397
409
 
398
410
  injectDeps(() => session);
@@ -430,6 +442,8 @@ describe("voice-session-bridge", () => {
430
442
  setTurnChannelContext: (ctx: unknown) => {
431
443
  capturedTurnChannelContext = ctx;
432
444
  },
445
+ // The image-bearing profile pin reads the leg's history.
446
+ getMessages: () => [],
433
447
  } as unknown as Conversation;
434
448
 
435
449
  injectDeps(() => session);
@@ -586,6 +600,8 @@ describe("voice-session-bridge", () => {
586
600
  capturedTrustContext = ctx;
587
601
  }
588
602
  },
603
+ // The image-bearing profile pin reads the leg's history.
604
+ getMessages: () => [],
589
605
  } as unknown as Conversation;
590
606
 
591
607
  injectDeps(() => session);
@@ -629,6 +645,8 @@ describe("voice-session-bridge", () => {
629
645
  capturedPrompt = prompt;
630
646
  }
631
647
  },
648
+ // The image-bearing profile pin reads the leg's history.
649
+ getMessages: () => [],
632
650
  } as unknown as Conversation;
633
651
 
634
652
  injectDeps(() => session);
@@ -690,6 +708,8 @@ describe("voice-session-bridge", () => {
690
708
  capturedPrompt = prompt;
691
709
  }
692
710
  },
711
+ // The image-bearing profile pin reads the leg's history.
712
+ getMessages: () => [],
693
713
  } as unknown as Conversation;
694
714
 
695
715
  injectDeps(() => session);
@@ -782,6 +802,8 @@ describe("voice-session-bridge", () => {
782
802
  });
783
803
  },
784
804
  abort: () => {},
805
+ // The image-bearing profile pin reads the leg's history.
806
+ getMessages: () => [],
785
807
  } as unknown as Conversation;
786
808
 
787
809
  injectDeps(() => session);
@@ -870,6 +892,8 @@ describe("voice-session-bridge", () => {
870
892
  });
871
893
  },
872
894
  abort: () => {},
895
+ // The image-bearing profile pin reads the leg's history.
896
+ getMessages: () => [],
873
897
  } as unknown as Conversation;
874
898
 
875
899
  injectDeps(() => session);
@@ -946,6 +970,8 @@ describe("voice-session-bridge", () => {
946
970
  handleConfirmationCalls.push({ requestId, decision });
947
971
  },
948
972
  abort: () => {},
973
+ // The image-bearing profile pin reads the leg's history.
974
+ getMessages: () => [],
949
975
  } as unknown as Conversation;
950
976
 
951
977
  injectDeps(() => session);
@@ -1014,6 +1040,8 @@ describe("voice-session-bridge", () => {
1014
1040
  handleConfirmationCalls.push({ requestId, decision });
1015
1041
  },
1016
1042
  abort: () => {},
1043
+ // The image-bearing profile pin reads the leg's history.
1044
+ getMessages: () => [],
1017
1045
  } as unknown as Conversation;
1018
1046
 
1019
1047
  injectDeps(() => session);
@@ -1080,6 +1108,8 @@ describe("voice-session-bridge", () => {
1080
1108
  handleConfirmationCalls.push({ requestId, decision });
1081
1109
  },
1082
1110
  abort: () => {},
1111
+ // The image-bearing profile pin reads the leg's history.
1112
+ getMessages: () => [],
1083
1113
  } as unknown as Conversation;
1084
1114
 
1085
1115
  injectDeps(() => session);
@@ -1164,6 +1194,8 @@ describe("voice-session-bridge", () => {
1164
1194
  } as AssistantEvent);
1165
1195
  },
1166
1196
  abort: () => {},
1197
+ // The image-bearing profile pin reads the leg's history.
1198
+ getMessages: () => [],
1167
1199
  } as unknown as Conversation;
1168
1200
  }
1169
1201
 
@@ -1303,6 +1335,8 @@ describe("voice-session-bridge", () => {
1303
1335
  handleSecretCalls.push({ requestId, value, delivery });
1304
1336
  },
1305
1337
  abort: () => {},
1338
+ // The image-bearing profile pin reads the leg's history.
1339
+ getMessages: () => [],
1306
1340
  } as unknown as Conversation;
1307
1341
 
1308
1342
  injectDeps(() => session);
@@ -1354,6 +1388,8 @@ describe("voice-session-bridge", () => {
1354
1388
  runAgentLoop: async () => {},
1355
1389
  handleConfirmationResponse: () => {},
1356
1390
  abort: () => {},
1391
+ // The image-bearing profile pin reads the leg's history.
1392
+ getMessages: () => [],
1357
1393
  } as unknown as Conversation & { forcePromptSideEffects: boolean };
1358
1394
 
1359
1395
  injectDeps(() => session);
@@ -1413,6 +1449,8 @@ describe("voice-session-bridge", () => {
1413
1449
  runAgentLoop: async () => {},
1414
1450
  handleConfirmationResponse: () => {},
1415
1451
  abort: () => {},
1452
+ // The image-bearing profile pin reads the leg's history.
1453
+ getMessages: () => [],
1416
1454
  } as unknown as Conversation & {
1417
1455
  forcePromptSideEffects: boolean;
1418
1456
  callSessionId?: string;
@@ -1479,6 +1517,8 @@ describe("voice-session-bridge", () => {
1479
1517
  runAgentLoop: async () => {},
1480
1518
  handleConfirmationResponse: () => {},
1481
1519
  abort: () => {},
1520
+ // The image-bearing profile pin reads the leg's history.
1521
+ getMessages: () => [],
1482
1522
  } as unknown as Conversation & {
1483
1523
  forcePromptSideEffects: boolean;
1484
1524
  callSessionId?: string;
@@ -1540,6 +1580,8 @@ describe("voice-session-bridge", () => {
1540
1580
  abort: () => {
1541
1581
  abortCalled = true;
1542
1582
  },
1583
+ // The image-bearing profile pin reads the leg's history.
1584
+ getMessages: () => [],
1543
1585
  } as unknown as Conversation;
1544
1586
 
1545
1587
  injectDeps(() => session);
@@ -31,6 +31,14 @@ mock.module("../../daemon/conversation-store.js", () => ({
31
31
  getOrCreateConversation: async () => fakeConversation,
32
32
  }));
33
33
 
34
+ // Vision capability of the image pin's target profile. Install-dependent in
35
+ // production (a BYO provider resolves the profile key through its own column
36
+ // of the intent matrix), so it is scripted rather than read from a catalog.
37
+ let pinProfileSupportsVision = true;
38
+ mock.module("../../plugin-api/vision-support.js", () => ({
39
+ doesSupportVision: () => pinProfileSupportsVision,
40
+ }));
41
+
34
42
  // Conversation-CRUD doubles for the teardown transcript-hygiene pass. The
35
43
  // real module is spread so every other export keeps its production behavior;
36
44
  // only the functions the hygiene pass (and discard) touch are recorded.
@@ -133,6 +141,7 @@ interface FakeConversation {
133
141
  opts?: { decisionContext?: string },
134
142
  ) => void;
135
143
  runAgentLoop: (...args: unknown[]) => Promise<void>;
144
+ getMessages: () => Array<{ role: string; content: unknown[] }>;
136
145
  abort: (reason?: unknown) => void;
137
146
  loadFromDb: () => Promise<void>;
138
147
  toolsDisabledDepth: number;
@@ -149,6 +158,8 @@ function makeFakeConversation(opts: {
149
158
  onPersist?: (attempt: number) => void;
150
159
  /** Workspace root; pass empty to model a missing boundary. */
151
160
  workingDir?: string;
161
+ /** In-memory history the profile pin reads; undefined models a text-only call. */
162
+ messages?: Array<{ role: string; content: unknown[] }>;
152
163
  }) {
153
164
  const waitForIdleCalls: WaitForIdleCall[] = [];
154
165
  const confirmationDecisions: Array<{ requestId: string; decision: string }> =
@@ -206,6 +217,7 @@ function makeFakeConversation(opts: {
206
217
  confirmationDecisions.push({ requestId, decision });
207
218
  },
208
219
  runAgentLoop: () => (opts.runAgentLoop ?? (async () => {}))(),
220
+ getMessages: () => opts.messages ?? [],
209
221
  abort: () => {},
210
222
  loadFromDb: async () => {
211
223
  opts.events?.push("loadFromDb");
@@ -2165,3 +2177,105 @@ describe("transcript hygiene (teardown pass)", () => {
2165
2177
  expect(crudLog.deletes).toContain("assistant-row-1");
2166
2178
  });
2167
2179
  });
2180
+
2181
+ describe("startVoiceTurn image-bearing profile pin", () => {
2182
+ /** A persisted user message carrying a photo taken mid-call. */
2183
+ const PHOTO_HISTORY = [
2184
+ { role: "user", content: [{ type: "text", text: "here's a photo:" }] },
2185
+ {
2186
+ role: "user",
2187
+ content: [
2188
+ { type: "text", text: "here's a photo:" },
2189
+ { type: "image", source: { type: "base64", data: "abc" } },
2190
+ ],
2191
+ },
2192
+ ];
2193
+
2194
+ beforeEach(() => {
2195
+ pinProfileSupportsVision = true;
2196
+ });
2197
+
2198
+ async function runOptionsFor(opts: {
2199
+ messages?: Array<{ role: string; content: unknown[] }>;
2200
+ turn?: Record<string, unknown>;
2201
+ }): Promise<Record<string, unknown>> {
2202
+ const fake = makeFakeConversation({
2203
+ processing: false,
2204
+ ...(opts.messages ? { messages: opts.messages } : {}),
2205
+ });
2206
+ fakeConversation = fake.conversation;
2207
+ let runOptions: Record<string, unknown> = {};
2208
+ fake.conversation.runAgentLoop = async (...args: unknown[]) => {
2209
+ runOptions = args[2] as Record<string, unknown>;
2210
+ };
2211
+ await startVoiceTurn({ ...makeTurnOptions(), ...(opts.turn ?? {}) });
2212
+ return runOptions;
2213
+ }
2214
+
2215
+ test("a text-only call keeps the call-site profile", async () => {
2216
+ const runOptions = await runOptionsFor({});
2217
+
2218
+ expect(runOptions.overrideProfile).toBeUndefined();
2219
+ expect(runOptions.forceOverrideProfile).toBeUndefined();
2220
+ });
2221
+
2222
+ test("an image in history pins the image-capable profile", async () => {
2223
+ // `callAgent`'s balanced profile carries no guarantee that its model takes
2224
+ // an image, and a model that rejects one fails the whole leg.
2225
+ const runOptions = await runOptionsFor({ messages: PHOTO_HISTORY });
2226
+
2227
+ expect(runOptions.overrideProfile).toBe("latency-optimized");
2228
+ // callAgent is not `mainAgent`, so an unforced override would sit below
2229
+ // the call-site profile and never apply.
2230
+ expect(runOptions.forceOverrideProfile).toBe(true);
2231
+ });
2232
+
2233
+ test("an image nested in a tool result counts too", async () => {
2234
+ const runOptions = await runOptionsFor({
2235
+ messages: [
2236
+ {
2237
+ role: "user",
2238
+ content: [
2239
+ {
2240
+ type: "tool_result",
2241
+ contentBlocks: [{ type: "image", source: { data: "abc" } }],
2242
+ },
2243
+ ],
2244
+ },
2245
+ ],
2246
+ });
2247
+
2248
+ expect(runOptions.overrideProfile).toBe("latency-optimized");
2249
+ });
2250
+
2251
+ test("a front-door leg is left alone — its own call site pins it", async () => {
2252
+ const runOptions = await runOptionsFor({
2253
+ messages: PHOTO_HISTORY,
2254
+ turn: { routingLeg: "front-door" },
2255
+ });
2256
+
2257
+ expect(runOptions.overrideProfile).toBeUndefined();
2258
+ expect(runOptions.callSite).toBe("voiceFrontDoor");
2259
+ });
2260
+
2261
+ test("no pin when the pin target can't take an image either", async () => {
2262
+ // Fireworks: `latency-optimized` resolves to a text-only model while
2263
+ // `balanced` is vision-capable, so pinning would break the very turn the
2264
+ // pin exists to save.
2265
+ pinProfileSupportsVision = false;
2266
+
2267
+ const runOptions = await runOptionsFor({ messages: PHOTO_HISTORY });
2268
+
2269
+ expect(runOptions.overrideProfile).toBeUndefined();
2270
+ expect(runOptions.forceOverrideProfile).toBeUndefined();
2271
+ });
2272
+
2273
+ test("an explicit routing pin wins over the image pin", async () => {
2274
+ const runOptions = await runOptionsFor({
2275
+ messages: PHOTO_HISTORY,
2276
+ turn: { overrideProfile: "quality-optimized" },
2277
+ });
2278
+
2279
+ expect(runOptions.overrideProfile).toBe("quality-optimized");
2280
+ });
2281
+ });
@@ -35,8 +35,9 @@ import {
35
35
  updateMessageContent,
36
36
  } from "../persistence/conversation-crud.js";
37
37
  import { VOICE_ESCALATION_CONTINUATION_MESSAGE_KIND } from "../plugin-api/constants.js";
38
+ import { doesSupportVision } from "../plugin-api/vision-support.js";
38
39
  import { pinnedListeningLanguage } from "../providers/speech-to-text/provider-catalog.js";
39
- import type { ContentBlock } from "../providers/types.js";
40
+ import type { ContentBlock, Message } from "../providers/types.js";
40
41
  import { broadcastMessage } from "../runtime/assistant-event-hub.js";
41
42
  import { DAEMON_INTERNAL_ASSISTANT_ID } from "../runtime/assistant-scope.js";
42
43
  import { getCurrentSeq } from "../runtime/assistant-stream-state.js";
@@ -67,6 +68,40 @@ import {
67
68
 
68
69
  const log = getLogger("voice-session-bridge");
69
70
 
71
+ /**
72
+ * Profile an image-bearing voice leg is pinned to.
73
+ *
74
+ * The latency-class profile is the one voice already leans on (it fronts every
75
+ * turn through `voiceFrontDoor`); `callAgent`'s `balanced` profile carries no
76
+ * guarantee that its model takes images, and a model that rejects an image
77
+ * fails the whole leg rather than degrading it.
78
+ *
79
+ * Whether THIS profile takes images is an install-level question, not a
80
+ * constant: a BYO provider resolves the key through its own column of the
81
+ * intent matrix, and on Fireworks that lands on a text-only model while its
82
+ * `balanced` column is vision-capable. Pinning there would break the exact
83
+ * turns this pin exists to save, hence the capability check at the call site.
84
+ */
85
+ const VOICE_IMAGE_PROFILE = "latency-optimized";
86
+
87
+ /**
88
+ * Does this conversation's history carry an image?
89
+ *
90
+ * Images persist inline and are re-sent on every later turn, so one photo
91
+ * taken mid-call makes every remaining turn of that call an image turn -- the
92
+ * check is over the whole history, not just this turn's own content.
93
+ */
94
+ function conversationCarriesImage(messages: readonly Message[]): boolean {
95
+ return messages.some((message) =>
96
+ message.content.some(
97
+ (block) =>
98
+ block.type === "image" ||
99
+ (block.type === "tool_result" &&
100
+ block.contentBlocks?.some((nested) => nested.type === "image")),
101
+ ),
102
+ );
103
+ }
104
+
70
105
  /**
71
106
  * Front-door decision rule with the registry-derived capability digest. The
72
107
  * front-door leg runs toolless (see the `toolsDisabledDepth` bracket in
@@ -1657,6 +1692,23 @@ export async function startVoiceTurn(
1657
1692
  conversation.toolsDisabledDepth++;
1658
1693
  frontDoorToolsSuppressed = true;
1659
1694
  }
1695
+ // Resolved once here rather than inside the options literal below, so
1696
+ // the history scan happens once per leg. A front-door leg is skipped:
1697
+ // its own call site already resolves to the same profile. The
1698
+ // capability check comes before the scan because it is the cheaper of
1699
+ // the two and it decides whether the pin is worth anything at all.
1700
+ const carriesImage =
1701
+ opts.routingLeg !== "front-door" &&
1702
+ doesSupportVision(VOICE_IMAGE_PROFILE) &&
1703
+ conversationCarriesImage(conversation.getMessages());
1704
+ if (carriesImage) {
1705
+ log.info(
1706
+ { turnId, routingLeg: opts.routingLeg ?? null },
1707
+ "Voice leg carries an image; pinning the image-capable profile",
1708
+ );
1709
+ }
1710
+ const profilePin =
1711
+ opts.overrideProfile ?? (carriesImage ? VOICE_IMAGE_PROFILE : null);
1660
1712
  await conversation.runAgentLoop(persistedContent, messageId, {
1661
1713
  onEvent: (msg: AssistantEvent) => {
1662
1714
  if (msg.type === "assistant_turn_start") {
@@ -1737,11 +1789,13 @@ export async function startVoiceTurn(
1737
1789
  // strong escalation profile. `forceOverrideProfile` floats it above the
1738
1790
  // callAgent call-site layers (callAgent is not `mainAgent`, so the
1739
1791
  // override would otherwise sit below the call-site profile).
1740
- ...(opts.overrideProfile != null
1741
- ? {
1742
- overrideProfile: opts.overrideProfile,
1743
- forceOverrideProfile: true,
1744
- }
1792
+ //
1793
+ // An explicit routing pin wins; failing that, a leg whose history
1794
+ // carries an image is pinned to a profile whose model takes one. A
1795
+ // front-door leg needs neither: its own call site already resolves
1796
+ // there.
1797
+ ...(profilePin != null
1798
+ ? { overrideProfile: profilePin, forceOverrideProfile: true }
1745
1799
  : {}),
1746
1800
  });
1747
1801
  if (lastError) {
@@ -31,8 +31,8 @@ import {
31
31
  // profile that runs another.
32
32
 
33
33
  const FLAG = BALANCED_MODEL_EXPERIMENT_FLAG_KEY;
34
- const SHIPPED_MODEL = "gpt-5.6-luna";
35
- const GLM_MODEL = "accounts/fireworks/models/glm-5p2";
34
+ const SHIPPED_MODEL = "accounts/fireworks/models/glm-5p2";
35
+ const GLM_MODEL = SHIPPED_MODEL;
36
36
 
37
37
  function setArm(value: boolean | string): void {
38
38
  setOverridesForTesting({ [FLAG]: value });
@@ -79,7 +79,7 @@ describe("balanced-model experiment arms", () => {
79
79
  expect(entry?.source).toBe("managed");
80
80
  });
81
81
 
82
- test("glm-5p2 repoints mainAgent at GLM 5.2 within the model's output cap", () => {
82
+ test("glm-5p2 resolves mainAgent to GLM 5.2 within the model's output cap", () => {
83
83
  setArm("glm-5p2");
84
84
  const resolved = resolveCallSiteConfig(
85
85
  "mainAgent",
@@ -93,7 +93,7 @@ describe("balanced-model experiment arms", () => {
93
93
  const cap = catalogMaxOutputTokens(upstream as string, GLM_MODEL);
94
94
  expect(cap).toBeDefined();
95
95
  expect(resolved.maxTokens).toBeLessThanOrEqual(cap as number);
96
- // Only the model moves: the shipped token budget stands.
96
+ // The shipped token budget stands.
97
97
  expect(resolved.maxTokens).toBe(
98
98
  CODE_DEFAULT_PROFILE_ENTRIES.balanced.maxTokens as number,
99
99
  );
@@ -111,7 +111,7 @@ describe("balanced-model experiment arms", () => {
111
111
  });
112
112
 
113
113
  test("the other default profiles are untouched by an arm", () => {
114
- setArm("glm-5p2");
114
+ setArm("terra");
115
115
  for (const key of [
116
116
  "quality-optimized",
117
117
  "cost-optimized",
@@ -239,10 +239,10 @@ describe("balanced-model experiment boundaries", () => {
239
239
  });
240
240
 
241
241
  test("an install with no defaultProvider still picks up the arm", () => {
242
- setArm("glm-5p2");
242
+ setArm("terra");
243
243
  expect(
244
244
  resolveDefaultProfileForProvider(undefined, "balanced", null)?.model,
245
- ).toBe(GLM_MODEL);
245
+ ).toBe("gpt-5.6-terra");
246
246
  });
247
247
  });
248
248
 
@@ -57,12 +57,12 @@ describe("getEffectiveProfiles", () => {
57
57
  }
58
58
  });
59
59
 
60
- test("the managed Balanced profile routes GPT-5.6 Luna through OpenAI", () => {
60
+ test("the managed Balanced profile routes GLM 5.2 through Fireworks", () => {
61
61
  const balanced = CODE_DEFAULT_PROFILE_ENTRIES.balanced;
62
- expect(balanced.model).toBe("gpt-5.6-luna");
62
+ expect(balanced.model).toBe("accounts/fireworks/models/glm-5p2");
63
63
  expect(resolveRoutingIdentity(balanced.provider, balanced.model)).toEqual({
64
64
  connectionName: "vellum",
65
- expectedProvider: "openai",
65
+ expectedProvider: "fireworks",
66
66
  });
67
67
  });
68
68
 
@@ -73,7 +73,7 @@ type ProfileImpls = Record<DefaultProfileKey, DefaultProfileTemplate>;
73
73
  */
74
74
  const VELLUM_PROFILE_IMPLS: ProfileImpls = {
75
75
  balanced: {
76
- model: "gpt-5.6-luna",
76
+ model: "accounts/fireworks/models/glm-5p2",
77
77
  provider: "vellum",
78
78
  source: "managed",
79
79
  label: "Balanced",
@@ -130,8 +130,7 @@ const VELLUM_PROFILE_IMPLS: ProfileImpls = {
130
130
  // provisioned in every environment, and it alone selects the upstream:
131
131
  // `provider` below is the provider-agnostic managed sentinel, so
132
132
  // `getManagedUpstream` resolves the real upstream from the model's catalog
133
- // owner. Sharing `balanced`'s model is deliberate: the split is effort
134
- // and thinking, and this model is the one live-voice TTFT drives validated.
133
+ // owner. This model is the one live-voice TTFT drives validated.
135
134
  model: "gpt-5.6-luna",
136
135
  provider: "vellum",
137
136
  source: "managed",
@@ -165,8 +164,8 @@ const VELLUM_PROFILE_IMPLS: ProfileImpls = {
165
164
  * arm is remote input, and an object lookup would resolve `constructor` or
166
165
  * `toString` to an inherited `Object.prototype` member instead of missing.
167
166
  *
168
- * `glm-5p2` is text-only: Balanced does not pass `doesSupportVision` on that
169
- * arm, so image input routes through the image-fallback captioning plugin.
167
+ * `glm-5p2` names the same model as the shipped pin and stays in the table so
168
+ * the arm keeps its meaning if the shipped pin moves again.
170
169
  */
171
170
  const BALANCED_EXPERIMENT_MODELS = new Map<string, string>([
172
171
  ["terra", "gpt-5.6-terra"],
@@ -206,7 +205,7 @@ const CHATGPT_PROFILE_IMPLS: ProfileImpls = {
206
205
  source: "managed",
207
206
  label: "Balanced",
208
207
  description: "Good balance of quality, cost, and speed",
209
- // Matches the vellum column's Balanced (same model): the Codex path
208
+ // Matches the vellum column's Balanced token budget: the Codex path
210
209
  // sends no max_output_tokens, so this only sizes internal budgeting.
211
210
  maxTokens: 32000,
212
211
  effort: "high",
@@ -228,14 +228,6 @@
228
228
  "description": "Enable self-hosted assistant configuration.",
229
229
  "defaultEnabled": false
230
230
  },
231
- {
232
- "id": "web-remote-ingress",
233
- "scope": "client",
234
- "key": "web-remote-ingress",
235
- "label": "Web Remote Ingress",
236
- "description": "Show the pair-a-device card in web settings. Visibility only; pairing itself is always available.",
237
- "defaultEnabled": false
238
- },
239
231
  {
240
232
  "id": "paired-devices-ui",
241
233
  "scope": "client",
@@ -324,14 +316,6 @@
324
316
  "description": "Persist chat secret redactions as sentinel markers instead of the legacy HTML marker, enriched with the vault identity when the value byte-matches a credential revealed this turn. Also gates the `credentials reveal --for-chat` channel and the live-stream swap that replaces an echoed reveal plaintext with its sentinel before emission. Redaction itself is NOT gated: a route-proven reveal plaintext stays out of persisted rows in both modes. Client chip rendering is gated on daemon version, not on this flag.",
325
317
  "defaultEnabled": false
326
318
  },
327
- {
328
- "id": "assistant-switcher",
329
- "scope": "client",
330
- "key": "assistant-switcher",
331
- "label": "Assistant Switcher",
332
- "description": "Show the Switch Assistant section in General settings, linking to the assistant chooser. Covers local clients, signed-in cloud-hub users, remote-gateway origins, and native mobile shells; remote-gateway desktop browsers hand off to the hub chooser with a self-registration param. Also makes the chooser's selection authoritative for the app's active assistant on every surface including the platform hub, a job that previously belonged to multi-platform-assistant alone.",
333
- "defaultEnabled": false
334
- },
335
319
  {
336
320
  "id": "marketing-pricing-takeover",
337
321
  "scope": "client",
@@ -377,7 +361,7 @@
377
361
  "scope": "assistant",
378
362
  "key": "experiment-balanced-model-2026-08-06",
379
363
  "label": "Experiment: Balanced Profile Model",
380
- "description": "Multivariate experiment on the model the managed Balanced inference profile resolves to. control = gpt-5.6-luna (the shipped pin); terra = gpt-5.6-terra; glm-5p2 = accounts/fireworks/models/glm-5p2. Only the managed (vellum) implementation moves: ChatGPT-subscription and BYOK installs run the provider the user chose and are outside the experiment. Rollout and targeting are managed in the LaunchDarkly dashboard; any value that is not an arm name resolves to control.",
364
+ "description": "Multivariate experiment on the model the managed Balanced inference profile resolves to. control = the shipped pin (accounts/fireworks/models/glm-5p2); terra = gpt-5.6-terra; glm-5p2 = accounts/fireworks/models/glm-5p2. Only the managed (vellum) implementation moves: ChatGPT-subscription and BYOK installs run the provider the user chose and are outside the experiment. Rollout and targeting are managed in the LaunchDarkly dashboard; any value that is not an arm name resolves to control.",
381
365
  "defaultEnabled": "control",
382
366
  "values": ["control", "terra", "glm-5p2"]
383
367
  },