@vellumai/assistant 0.12.0-dev.202609110124.4b134bf → 0.12.0-dev.202609110918.dd3fa5e

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vellumai/assistant",
3
- "version": "0.12.0-dev.202609110124.4b134bf",
3
+ "version": "0.12.0-dev.202609110918.dd3fa5e",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "exports": {
@@ -6,6 +6,7 @@ import {
6
6
  getCatalogProviderForModel,
7
7
  isModelInCatalog,
8
8
  PROVIDER_CATALOG,
9
+ supportsForcedToolChoiceWithThinking,
9
10
  } from "../providers/model-catalog.js";
10
11
  import { PLATFORM_PROVIDER_META } from "../providers/platform-proxy/constants.js";
11
12
  import { resolvePricing, resolvePricingForUsage } from "../util/pricing.js";
@@ -551,6 +552,30 @@ describe("LLM catalog parity: daemon vs client", () => {
551
552
  expect(getCatalogProviderForModel("unknown/model")).toBeUndefined();
552
553
  });
553
554
 
555
+ test("forced tool choice with thinking is scoped to OpenRouter Kimi K2.6", () => {
556
+ expect(
557
+ supportsForcedToolChoiceWithThinking("openrouter", "moonshotai/kimi-k2.6"),
558
+ ).toBe(false);
559
+ expect(
560
+ supportsForcedToolChoiceWithThinking(
561
+ "openrouter",
562
+ "moonshotai/kimi-k2.6-20260420",
563
+ ),
564
+ ).toBe(false);
565
+ expect(
566
+ supportsForcedToolChoiceWithThinking(
567
+ "vercel-ai-gateway",
568
+ "moonshotai/kimi-k2.6",
569
+ ),
570
+ ).toBe(true);
571
+ expect(
572
+ supportsForcedToolChoiceWithThinking("openrouter", "unknown/model"),
573
+ ).toBe(true);
574
+ expect(
575
+ supportsForcedToolChoiceWithThinking("unknown-provider", "unknown/model"),
576
+ ).toBe(true);
577
+ });
578
+
554
579
  test("Gemini 2.5 Pro catalog context matches provider limits", () => {
555
580
  const gemini = PROVIDER_CATALOG.find((entry) => entry.id === "gemini");
556
581
  expect(
@@ -39,6 +39,7 @@ import type {
39
39
  SendMessageOptions,
40
40
  } from "@vellumai/plugin-api";
41
41
 
42
+ import { OpenRouterProvider } from "../../../../../providers/openrouter/client.js";
42
43
  import { ProviderError } from "../../../../../util/errors.js";
43
44
  import { sectionHeadLine } from "../sections.js";
44
45
  import type { MemoryRoutingTurn, Section } from "../types.js";
@@ -843,3 +844,91 @@ describe("selectPool: sections and keyword-in-context snippets", () => {
843
844
  );
844
845
  });
845
846
  });
847
+
848
+ // ---------------------------------------------------------------------------
849
+ // selectPool: cataloged thinking and forced-tool compatibility.
850
+ // ---------------------------------------------------------------------------
851
+
852
+ describe("selectPool: cataloged thinking and forced-tool compatibility", () => {
853
+ test("OpenRouter Kimi K2.6 omits the forced choice and yields a structured selection", async () => {
854
+ // A real OpenAI chat-completions provider stands in for the cataloged
855
+ // OpenRouter Kimi profile. The request succeeds without a reactive retry
856
+ // because the forced select_pages choice is omitted before dispatch.
857
+ const wireRequests: unknown[] = [];
858
+ const kimi = new OpenRouterProvider(
859
+ "test-key",
860
+ "moonshotai/kimi-k2.6-20260420",
861
+ );
862
+ (kimi as unknown as { client: unknown }).client = {
863
+ chat: {
864
+ completions: {
865
+ create: async (params: unknown) => {
866
+ wireRequests.push(JSON.parse(JSON.stringify(params)));
867
+ return {
868
+ async *[Symbol.asyncIterator]() {
869
+ yield {
870
+ choices: [
871
+ {
872
+ delta: {
873
+ tool_calls: [
874
+ {
875
+ index: 0,
876
+ id: "call-1",
877
+ function: {
878
+ name: "select_pages",
879
+ arguments: '{"ids":[3]}',
880
+ },
881
+ },
882
+ ],
883
+ },
884
+ finish_reason: "tool_calls",
885
+ },
886
+ ],
887
+ usage: { prompt_tokens: 10, completion_tokens: 2 },
888
+ };
889
+ },
890
+ };
891
+ },
892
+ },
893
+ },
894
+ };
895
+
896
+ // The profile layer injects the thinking effort onto the call config in
897
+ // production; emulate that here so the forced tool_choice rides alongside
898
+ // reasoning on the wire.
899
+ providerStub = {
900
+ name: "kimi-openai-compat",
901
+ sendMessage: (messages: Message[], options?: SendMessageOptions) =>
902
+ kimi.sendMessage(messages, {
903
+ ...options,
904
+ config: {
905
+ ...options?.config,
906
+ effort: "high",
907
+ thinking: { enabled: true },
908
+ },
909
+ }),
910
+ };
911
+
912
+ const selection = await selectPool(makePool(), makeTurn("rollout?"));
913
+
914
+ // The preflight path avoids an incompatible first request, and the
915
+ // pool-level re-prompt loop never engages.
916
+ expect(wireRequests).toHaveLength(1);
917
+ const first = wireRequests[0] as {
918
+ tool_choice?: unknown;
919
+ reasoning?: { effort?: string; summary?: string };
920
+ };
921
+ expect(first.tool_choice).toBeUndefined();
922
+ expect(first.reasoning).toMatchObject({
923
+ effort: "high",
924
+ summary: "detailed",
925
+ });
926
+ expect(
927
+ warnPayloads().filter((p) => p.reason === "provider_error"),
928
+ ).toEqual([]);
929
+
930
+ // Structured selection survived: ids [3] is the topic-x finder line.
931
+ expect(selection.keptAll).toBe(false);
932
+ expect(selection.pages).toEqual([{ slug: "topic-x", sections: [] }]);
933
+ });
934
+ });
@@ -76,6 +76,13 @@ export interface CatalogModel {
76
76
  supportsAudioInput?: boolean;
77
77
  supportsToolUse?: boolean;
78
78
  supportsEffort?: boolean;
79
+ /**
80
+ * Whether this provider/model serving surface accepts a forced OpenAI
81
+ * chat-completions tool choice while thinking is enabled. Omit unless the
82
+ * combination is known incompatible. Daemon-only: not projected into the
83
+ * client catalog (see scripts/sync-llm-catalog.ts).
84
+ */
85
+ supportsForcedToolChoiceWithThinking?: boolean;
79
86
  pricing?: CatalogModelPricing;
80
87
  /**
81
88
  * Upper bound for `reasoning_effort` accepted by this model's upstream API.
@@ -1816,6 +1823,7 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1816
1823
  supportsCaching: true,
1817
1824
  supportsVision: true,
1818
1825
  supportsToolUse: true,
1826
+ supportsForcedToolChoiceWithThinking: false,
1819
1827
  pricing: {
1820
1828
  inputPer1mTokens: 0.95,
1821
1829
  outputPer1mTokens: 4.0,
@@ -2628,6 +2636,30 @@ export function modelSupportedEfforts(
2628
2636
  );
2629
2637
  }
2630
2638
 
2639
+ /**
2640
+ * Whether a provider/model serving surface accepts a forced OpenAI
2641
+ * chat-completions tool choice while thinking is enabled. Unknown providers
2642
+ * and models fail open so custom routes retain their existing request shape
2643
+ * and can rely on the bounded provider-error retry if needed.
2644
+ */
2645
+ export function supportsForcedToolChoiceWithThinking(
2646
+ providerId: string,
2647
+ modelId: string,
2648
+ ): boolean {
2649
+ const provider = PROVIDER_CATALOG.find((entry) => entry.id === providerId);
2650
+ if (!provider) {
2651
+ return true;
2652
+ }
2653
+ const stripDateSuffix = (id: string): string => id.replace(/-\d{8}$/, "");
2654
+ const normalizedModelId = stripDateSuffix(modelId);
2655
+ return !provider.models.some(
2656
+ (model) =>
2657
+ model.supportsForcedToolChoiceWithThinking === false &&
2658
+ (model.id === modelId ||
2659
+ stripDateSuffix(model.id) === normalizedModelId),
2660
+ );
2661
+ }
2662
+
2631
2663
  /**
2632
2664
  * Return the catalog provider that owns a model ID, if known. When multiple
2633
2665
  * providers list the same ID (e.g. OpenRouter and the Vercel AI Gateway share
@@ -934,6 +934,51 @@ describe("thinking-mode tool_choice rejection fallback", () => {
934
934
  expect(text?.text).toBe("ok");
935
935
  });
936
936
 
937
+ test("retries once when Kimi rejects a specified tool_choice in thinking mode", async () => {
938
+ const { provider, requests } = stubProviderWithErrors(
939
+ [rejection("tool_choice 'specified' is incompatible with thinking enabled")],
940
+ OK_CHUNKS,
941
+ );
942
+
943
+ const response = await provider.sendMessage(
944
+ [{ role: "user", content: [{ type: "text", text: "hi" }] }],
945
+ {
946
+ tools: [
947
+ {
948
+ name: "select_pages",
949
+ description: "Pick relevant memory pages",
950
+ input_schema: { type: "object", properties: {} },
951
+ },
952
+ ],
953
+ config: {
954
+ tool_choice: { type: "tool", name: "select_pages" },
955
+ effort: "high",
956
+ },
957
+ },
958
+ );
959
+
960
+ expect(requests).toHaveLength(2);
961
+ const first = requests[0] as {
962
+ tool_choice?: { type: string; function: { name: string } };
963
+ reasoning_effort?: string;
964
+ };
965
+ const second = requests[1] as {
966
+ tool_choice?: unknown;
967
+ reasoning_effort?: string;
968
+ };
969
+ expect(first.tool_choice).toEqual({
970
+ type: "function",
971
+ function: { name: "select_pages" },
972
+ });
973
+ expect(first.reasoning_effort).toBe("high");
974
+ expect(second.tool_choice).toBeUndefined();
975
+ expect(second.reasoning_effort).toBe("high");
976
+ const text = response.content.find((b) => b.type === "text") as
977
+ | { type: "text"; text: string }
978
+ | undefined;
979
+ expect(text?.text).toBe("ok");
980
+ });
981
+
937
982
  test("retries once for an OpenRouter-wrapped thinking-mode tool_choice rejection", async () => {
938
983
  const wrapped = new OpenAI.APIError(
939
984
  400,
@@ -26,13 +26,14 @@ const USER_MESSAGE: Message[] = [
26
26
  */
27
27
  function stubChatProvider(
28
28
  options?: ConstructorParameters<typeof OpenAIChatCompletionsProvider>[2],
29
+ model = "test-model",
29
30
  ): {
30
31
  provider: OpenAIChatCompletionsProvider;
31
32
  requests: Array<Record<string, unknown>>;
32
33
  } {
33
34
  const provider = new OpenAIChatCompletionsProvider(
34
35
  "test-key",
35
- "test-model",
36
+ model,
36
37
  options,
37
38
  );
38
39
  const requests: Array<Record<string, unknown>> = [];
@@ -235,6 +236,87 @@ describe("OpenAIChatCompletionsProvider tool_choice wiring", () => {
235
236
  });
236
237
  });
237
238
 
239
+ test("omits a forced tool choice for OpenRouter Kimi K2.6 with thinking", async () => {
240
+ const { provider, requests } = stubChatProvider(
241
+ { providerName: "openrouter" },
242
+ "moonshotai/kimi-k2.6-20260420",
243
+ );
244
+
245
+ await provider.sendMessage(USER_MESSAGE, {
246
+ tools: TOOLS,
247
+ config: { tool_choice: { type: "tool", name: "bash" }, effort: "high" },
248
+ });
249
+
250
+ expect(requests).toHaveLength(1);
251
+ expect(requests[0].reasoning_effort).toBe("high");
252
+ expect(requests[0].tool_choice).toBeUndefined();
253
+ });
254
+
255
+ test("omits required for OpenRouter Kimi K2.6 with thinking", async () => {
256
+ const { provider, requests } = stubChatProvider(
257
+ { providerName: "openrouter" },
258
+ "moonshotai/kimi-k2.6-20260420",
259
+ );
260
+
261
+ await provider.sendMessage(USER_MESSAGE, {
262
+ tools: TOOLS,
263
+ config: { tool_choice: { type: "any" }, effort: "high" },
264
+ });
265
+
266
+ expect(requests).toHaveLength(1);
267
+ expect(requests[0].reasoning_effort).toBe("high");
268
+ expect(requests[0].tool_choice).toBeUndefined();
269
+ });
270
+
271
+ test("keeps tool_choice none for OpenRouter Kimi K2.6 with thinking", async () => {
272
+ const { provider, requests } = stubChatProvider(
273
+ { providerName: "openrouter" },
274
+ "moonshotai/kimi-k2.6-20260420",
275
+ );
276
+
277
+ await provider.sendMessage(USER_MESSAGE, {
278
+ tools: TOOLS,
279
+ config: { tool_choice: { type: "none" }, effort: "high" },
280
+ });
281
+
282
+ expect(requests[0].reasoning_effort).toBe("high");
283
+ expect(requests[0].tool_choice).toBe("none");
284
+ });
285
+
286
+ test("keeps a forced tool choice for OpenRouter Kimi K2.6 when thinking is off", async () => {
287
+ const { provider, requests } = stubChatProvider(
288
+ { providerName: "openrouter" },
289
+ "moonshotai/kimi-k2.6-20260420",
290
+ );
291
+
292
+ await provider.sendMessage(USER_MESSAGE, {
293
+ tools: TOOLS,
294
+ config: { tool_choice: { type: "tool", name: "bash" } },
295
+ });
296
+
297
+ expect(requests[0].tool_choice).toEqual({
298
+ type: "function",
299
+ function: { name: "bash" },
300
+ });
301
+ });
302
+
303
+ test("keeps a forced tool choice for Vercel Kimi K2.6 with thinking", async () => {
304
+ const { provider, requests } = stubChatProvider(
305
+ { providerName: "vercel-ai-gateway" },
306
+ "moonshotai/kimi-k2.6",
307
+ );
308
+
309
+ await provider.sendMessage(USER_MESSAGE, {
310
+ tools: TOOLS,
311
+ config: { tool_choice: { type: "tool", name: "bash" }, effort: "high" },
312
+ });
313
+
314
+ expect(requests[0].tool_choice).toEqual({
315
+ type: "function",
316
+ function: { name: "bash" },
317
+ });
318
+ });
319
+
238
320
  test("omits every tool_choice in thinking mode when omitToolChoiceWhenReasoning is on", async () => {
239
321
  const { provider, requests } = stubChatProvider({
240
322
  omitToolChoiceWhenReasoning: true,
@@ -17,6 +17,7 @@ import {
17
17
  mediaSourceByteLength,
18
18
  resolveMediaReferences,
19
19
  } from "../media-resolve.js";
20
+ import { supportsForcedToolChoiceWithThinking } from "../model-catalog.js";
20
21
  import { PLACEHOLDER_EMPTY_TURN } from "../placeholder-sentinels.js";
21
22
  import { recordProviderRequestDiagnostics } from "../request-diagnostics.js";
22
23
  import { createStreamTimeout } from "../stream-timeout.js";
@@ -404,9 +405,15 @@ export function isThinkingEnabledOnWire(params: unknown): boolean {
404
405
  * rejected it because thinking/reasoning mode forbids that parameter.
405
406
  * DeepSeek thinking mode 400s with `Thinking mode does not support this
406
407
  * tool_choice` for any explicit value, including `"auto"` and `"none"`.
407
- * One retry without `tool_choice` lets the same provider succeed instead of
408
- * failing over to a different backend.
408
+ * Kimi 400s with `tool_choice 'specified' is incompatible with thinking
409
+ * enabled`. One retry without `tool_choice` lets the same provider succeed
410
+ * instead of failing over to a different backend.
409
411
  */
412
+ const THINKING_MODE_TOOL_CHOICE_REJECTION_PATTERNS: RegExp[] = [
413
+ /does not support this tool_choice/i,
414
+ /tool_choice\s+'specified'\s+is incompatible with thinking/i,
415
+ ];
416
+
410
417
  function isThinkingModeToolChoiceRejection(
411
418
  error: unknown,
412
419
  params: unknown,
@@ -418,8 +425,9 @@ function isThinkingModeToolChoiceRejection(
418
425
  if (!isClientErrorStatus(error)) {
419
426
  return false;
420
427
  }
421
- return /does not support this tool_choice/i.test(
422
- openaiCompatErrorHaystack(error),
428
+ const haystack = openaiCompatErrorHaystack(error);
429
+ return THINKING_MODE_TOOL_CHOICE_REJECTION_PATTERNS.some((pattern) =>
430
+ pattern.test(haystack),
423
431
  );
424
432
  }
425
433
 
@@ -969,7 +977,18 @@ export class OpenAIChatCompletionsProvider implements Provider {
969
977
  const thinkingOn = isThinkingEnabledOnWire(params);
970
978
  const skipAutoDefault = thinkingOn && toolChoice === "auto";
971
979
  const skipAllChoices = thinkingOn && this.omitToolChoiceWhenReasoning;
972
- if (!skipAutoDefault && !skipAllChoices) {
980
+ const skipIncompatibleForcedChoice =
981
+ thinkingOn &&
982
+ !supportsForcedToolChoiceWithThinking(
983
+ this.name,
984
+ modelOverride ?? this.model,
985
+ ) &&
986
+ (toolChoice === "required" || typeof toolChoice === "object");
987
+ if (
988
+ !skipAutoDefault &&
989
+ !skipAllChoices &&
990
+ !skipIncompatibleForcedChoice
991
+ ) {
973
992
  params.tool_choice = toolChoice;
974
993
  }
975
994
  }