@velum-labs/routekit-gateway 0.9.4 → 0.9.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -22,6 +22,24 @@ type JsonObject = Record<string, unknown>;
22
22
  * job; the translation itself stays total.
23
23
  */
24
24
  export declare function isCursorChatBody(body: unknown): body is JsonObject;
25
+ /**
26
+ * Spell a namespaced model id the way Cursor's custom-model settings accept
27
+ * it. Cursor rejects ids containing "/" ("Model name is not valid"), so the
28
+ * cursor route advertises and answers to a dash-separated spelling
29
+ * (`claude-code/claude-fable-5` -> `claude-code-claude-fable-5`). Dashes are
30
+ * preferred over dots because dotted names can collide with Cursor's managed
31
+ * model catalog.
32
+ */
33
+ export declare function cursorModelAliasId(id: string): string;
34
+ /**
35
+ * Resolve Cursor's dash-separated spelling back to a served namespaced id.
36
+ *
37
+ * Resolution is an exact lookup over the gateway's served ids rather than a
38
+ * separator split, because namespaces themselves contain dashes. A rewrite
39
+ * only happens when the model is not served as spelled; if two served ids
40
+ * produce the same alias, the first one listed wins.
41
+ */
42
+ export declare function resolveCursorModelAlias(model: unknown, servedIds: readonly string[]): string | undefined;
25
43
  /**
26
44
  * Map a Cursor BYOK request body onto a Chat Completions body.
27
45
  *
@@ -40,6 +40,31 @@ function isObject(value) {
40
40
  export function isCursorChatBody(body) {
41
41
  return isObject(body) && ("messages" in body || "input" in body);
42
42
  }
43
+ /**
44
+ * Spell a namespaced model id the way Cursor's custom-model settings accept
45
+ * it. Cursor rejects ids containing "/" ("Model name is not valid"), so the
46
+ * cursor route advertises and answers to a dash-separated spelling
47
+ * (`claude-code/claude-fable-5` -> `claude-code-claude-fable-5`). Dashes are
48
+ * preferred over dots because dotted names can collide with Cursor's managed
49
+ * model catalog.
50
+ */
51
+ export function cursorModelAliasId(id) {
52
+ return id.replace("/", "-");
53
+ }
54
+ /**
55
+ * Resolve Cursor's dash-separated spelling back to a served namespaced id.
56
+ *
57
+ * Resolution is an exact lookup over the gateway's served ids rather than a
58
+ * separator split, because namespaces themselves contain dashes. A rewrite
59
+ * only happens when the model is not served as spelled; if two served ids
60
+ * produce the same alias, the first one listed wins.
61
+ */
62
+ export function resolveCursorModelAlias(model, servedIds) {
63
+ if (typeof model !== "string" || model.length === 0 || servedIds.includes(model)) {
64
+ return undefined;
65
+ }
66
+ return servedIds.find((id) => id.includes("/") && cursorModelAliasId(id) === model);
67
+ }
43
68
  /**
44
69
  * Map a Cursor BYOK request body onto a Chat Completions body.
45
70
  *
@@ -870,6 +870,26 @@ function responsesOutput(payload) {
870
870
  ...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {})
871
871
  };
872
872
  }
873
+ const CODEX_EMPTY_RESPONSE_ERROR = {
874
+ message: "Codex completed without assistant content or tool calls",
875
+ type: "upstream_empty_response"
876
+ };
877
+ function responsesItemText(item) {
878
+ if (item.type !== "message")
879
+ return "";
880
+ const content = item.content;
881
+ return (content ?? [])
882
+ .flatMap((part) => typeof part.text === "string" ? [part.text] : [])
883
+ .join("");
884
+ }
885
+ function codexCompletionResponse(model, payload) {
886
+ const message = responsesOutput(payload);
887
+ const hasOutput = (typeof message.content === "string" && message.content.length > 0) ||
888
+ (Array.isArray(message.tool_calls) && message.tool_calls.length > 0);
889
+ return hasOutput
890
+ ? jsonResponse(chatCompletion(model, message, normalizedOpenAiUsage(payload.usage)))
891
+ : jsonResponse({ error: CODEX_EMPTY_RESPONSE_ERROR }, 502);
892
+ }
873
893
  export class CodexResponsesBackend extends HttpProviderBackend {
874
894
  #accountId;
875
895
  #forceStream;
@@ -912,10 +932,31 @@ export class CodexResponsesBackend extends HttpProviderBackend {
912
932
  return copyFailure(response, await response.text());
913
933
  if (body.stream === true) {
914
934
  let hasToolCalls = false;
935
+ let hasAssistantContent = false;
936
+ const emittedText = new Map();
937
+ const contentChunk = (content) => ({
938
+ id: randomId(18, "chatcmpl_"),
939
+ object: "chat.completion.chunk",
940
+ model,
941
+ choices: [{ index: 0, delta: { content }, finish_reason: null }]
942
+ });
943
+ const recoverMessage = (output, outputIndex) => {
944
+ const complete = responsesItemText(output);
945
+ const previous = emittedText.get(outputIndex) ?? "";
946
+ const suffix = complete.startsWith(previous)
947
+ ? complete.slice(previous.length)
948
+ : "";
949
+ if (suffix.length === 0)
950
+ return [];
951
+ emittedText.set(outputIndex, complete);
952
+ hasAssistantContent = true;
953
+ return [contentChunk(suffix)];
954
+ };
915
955
  return mapSse(response, (event, data) => {
916
956
  const item = data;
917
- if (event === "response.reasoning_summary_text.delta" ||
918
- event === "response.reasoning_text.delta") {
957
+ const eventType = event === "message" && typeof item.type === "string" ? item.type : event;
958
+ if (eventType === "response.reasoning_summary_text.delta" ||
959
+ eventType === "response.reasoning_text.delta") {
919
960
  return [
920
961
  {
921
962
  id: randomId(18, "chatcmpl_"),
@@ -931,17 +972,16 @@ export class CodexResponsesBackend extends HttpProviderBackend {
931
972
  }
932
973
  ];
933
974
  }
934
- if (event === "response.output_text.delta") {
935
- return [
936
- {
937
- id: randomId(18, "chatcmpl_"),
938
- object: "chat.completion.chunk",
939
- model,
940
- choices: [{ index: 0, delta: { content: item.delta }, finish_reason: null }]
941
- }
942
- ];
975
+ if (eventType === "response.output_text.delta") {
976
+ const outputIndex = typeof item.output_index === "number" ? item.output_index : 0;
977
+ const delta = typeof item.delta === "string" ? item.delta : "";
978
+ emittedText.set(outputIndex, `${emittedText.get(outputIndex) ?? ""}${delta}`);
979
+ if (delta.length === 0)
980
+ return [];
981
+ hasAssistantContent = true;
982
+ return [contentChunk(delta)];
943
983
  }
944
- if (event === "response.function_call_arguments.delta") {
984
+ if (eventType === "response.function_call_arguments.delta") {
945
985
  return [
946
986
  {
947
987
  id: randomId(18, "chatcmpl_"),
@@ -964,7 +1004,7 @@ export class CodexResponsesBackend extends HttpProviderBackend {
964
1004
  }
965
1005
  ];
966
1006
  }
967
- if (event === "response.output_item.added") {
1007
+ if (eventType === "response.output_item.added") {
968
1008
  const output = item.item;
969
1009
  if (output?.type !== "function_call")
970
1010
  return [];
@@ -993,9 +1033,27 @@ export class CodexResponsesBackend extends HttpProviderBackend {
993
1033
  }
994
1034
  ];
995
1035
  }
996
- if (event === "response.completed") {
1036
+ if (eventType === "response.output_item.done") {
1037
+ const output = item.item;
1038
+ if (output?.type !== "message")
1039
+ return [];
1040
+ const outputIndex = typeof item.output_index === "number" ? item.output_index : 0;
1041
+ return recoverMessage(output, outputIndex);
1042
+ }
1043
+ if (eventType === "response.completed") {
997
1044
  const completed = item.response;
1045
+ const recovered = Array.isArray(completed?.output)
1046
+ ? completed.output.flatMap((output, outputIndex) => typeof output === "object" &&
1047
+ output !== null &&
1048
+ output.type === "message"
1049
+ ? recoverMessage(output, outputIndex)
1050
+ : [])
1051
+ : [];
1052
+ if (!hasAssistantContent && !hasToolCalls) {
1053
+ return [...recovered, { error: CODEX_EMPTY_RESPONSE_ERROR }];
1054
+ }
998
1055
  return [
1056
+ ...recovered,
999
1057
  {
1000
1058
  id: randomId(18, "chatcmpl_"),
1001
1059
  object: "chat.completion.chunk",
@@ -1013,6 +1071,18 @@ export class CodexResponsesBackend extends HttpProviderBackend {
1013
1071
  }
1014
1072
  ];
1015
1073
  }
1074
+ if (eventType === "response.failed" ||
1075
+ eventType === "response.incomplete" ||
1076
+ eventType === "error") {
1077
+ return [
1078
+ {
1079
+ error: {
1080
+ message: "Codex response did not complete",
1081
+ type: "upstream_error"
1082
+ }
1083
+ }
1084
+ ];
1085
+ }
1016
1086
  return [];
1017
1087
  });
1018
1088
  }
@@ -1022,7 +1092,7 @@ export class CodexResponsesBackend extends HttpProviderBackend {
1022
1092
  ...decoder.feed(await response.text()),
1023
1093
  ...decoder.flush()
1024
1094
  ];
1025
- const completedOutput = [];
1095
+ const completedOutput = new Map();
1026
1096
  let completedResponse;
1027
1097
  for (const event of events) {
1028
1098
  let payload;
@@ -1043,7 +1113,10 @@ export class CodexResponsesBackend extends HttpProviderBackend {
1043
1113
  if (eventType === "response.output_item.done" &&
1044
1114
  typeof record.item === "object" &&
1045
1115
  record.item !== null) {
1046
- completedOutput.push(record.item);
1116
+ const outputIndex = typeof record.output_index === "number"
1117
+ ? record.output_index
1118
+ : completedOutput.size;
1119
+ completedOutput.set(outputIndex, record.item);
1047
1120
  }
1048
1121
  if (eventType === "response.completed" &&
1049
1122
  typeof record.response === "object" &&
@@ -1052,16 +1125,20 @@ export class CodexResponsesBackend extends HttpProviderBackend {
1052
1125
  }
1053
1126
  }
1054
1127
  if (completedResponse !== undefined) {
1055
- const terminalOutput = completedResponse.output;
1056
- const payload = (!Array.isArray(terminalOutput) || terminalOutput.length === 0) &&
1057
- completedOutput.length > 0
1058
- ? { ...completedResponse, output: completedOutput }
1059
- : completedResponse;
1060
- return jsonResponse(chatCompletion(model, responsesOutput(payload), normalizedOpenAiUsage(payload.usage)));
1128
+ const terminalOutput = Array.isArray(completedResponse.output)
1129
+ ? [...completedResponse.output]
1130
+ : [];
1131
+ for (const [outputIndex, output] of completedOutput) {
1132
+ if (terminalOutput[outputIndex] === undefined) {
1133
+ terminalOutput[outputIndex] = output;
1134
+ }
1135
+ }
1136
+ const payload = { ...completedResponse, output: terminalOutput };
1137
+ return codexCompletionResponse(model, payload);
1061
1138
  }
1062
1139
  throw new SseParseError("provider SSE stream ended without response.completed");
1063
1140
  }
1064
1141
  const payload = (await response.json());
1065
- return jsonResponse(chatCompletion(model, responsesOutput(payload), normalizedOpenAiUsage(payload.usage)));
1142
+ return codexCompletionResponse(model, payload);
1066
1143
  }
1067
1144
  }
@@ -63,16 +63,52 @@ function reasoningWireShape(provider) {
63
63
  return undefined;
64
64
  }
65
65
  }
66
+ const ANTHROPIC_EFFORT_ORDER = [
67
+ "low",
68
+ "medium",
69
+ "high",
70
+ "xhigh",
71
+ "max"
72
+ ];
73
+ function capabilitySupported(value) {
74
+ if (typeof value === "boolean")
75
+ return value;
76
+ if (!isRecord(value) || typeof value.supported !== "boolean") {
77
+ return undefined;
78
+ }
79
+ return value.supported;
80
+ }
66
81
  export function parseReasoningCapabilities(entry, provider, refreshedAt = new Date().toISOString()) {
67
82
  if (!isRecord(entry))
68
83
  return undefined;
69
84
  const capabilities = isRecord(entry.capabilities) ? entry.capabilities : undefined;
70
85
  const nested = (isRecord(entry.reasoning) ? entry.reasoning : undefined) ??
71
86
  (isRecord(capabilities?.reasoning) ? capabilities.reasoning : undefined);
72
- const efforts = effortOptions(entry.supported_reasoning_levels ??
87
+ const discoveredEfforts = effortOptions(entry.supported_reasoning_levels ??
73
88
  entry.supported_reasoning_efforts ??
74
89
  nested?.efforts ??
75
90
  nested?.supported_efforts);
91
+ const anthropicEffort = provider === "anthropic" || provider === "claude-code"
92
+ ? isRecord(capabilities?.effort)
93
+ ? capabilities.effort
94
+ : undefined
95
+ : undefined;
96
+ const anthropicThinking = provider === "anthropic" || provider === "claude-code"
97
+ ? isRecord(capabilities?.thinking)
98
+ ? capabilities.thinking
99
+ : undefined
100
+ : undefined;
101
+ const thinkingTypes = isRecord(anthropicThinking?.types)
102
+ ? anthropicThinking.types
103
+ : undefined;
104
+ const effortSupported = capabilitySupported(anthropicEffort?.supported);
105
+ const thinkingSupported = capabilitySupported(anthropicThinking?.supported);
106
+ const adaptiveSupported = capabilitySupported(thinkingTypes?.adaptive);
107
+ const enabledSupported = capabilitySupported(thinkingTypes?.enabled);
108
+ const anthropicEfforts = effortSupported === true
109
+ ? ANTHROPIC_EFFORT_ORDER.flatMap((id) => capabilitySupported(anthropicEffort?.[id]) === true ? [{ id }] : [])
110
+ : [];
111
+ const efforts = discoveredEfforts.length > 0 ? discoveredEfforts : anthropicEfforts;
76
112
  const supportedParameters = Array.isArray(entry.supported_parameters)
77
113
  ? entry.supported_parameters.filter((parameter) => typeof parameter === "string")
78
114
  : [];
@@ -80,12 +116,20 @@ export function parseReasoningCapabilities(entry, provider, refreshedAt = new Da
80
116
  capabilities?.reasoning_controls ??
81
117
  entry.reasoning_controls;
82
118
  const supported = efforts.length > 0 ||
119
+ effortSupported === true ||
120
+ thinkingSupported === true ||
121
+ adaptiveSupported === true ||
122
+ enabledSupported === true ||
83
123
  supportedParameters.includes("reasoning") ||
84
124
  supportedParameters.includes("reasoning_effort") ||
85
125
  explicitStatus === "supported";
86
- const unsupported = explicitStatus === "unsupported" || nested?.supported === false;
87
- if (!supported && !unsupported && nested === undefined)
126
+ const unsupported = explicitStatus === "unsupported" ||
127
+ nested?.supported === false ||
128
+ (effortSupported === false && thinkingSupported === false);
129
+ const hasAnthropicMetadata = anthropicEffort !== undefined || anthropicThinking !== undefined;
130
+ if (!supported && !unsupported && nested === undefined && !hasAnthropicMetadata) {
88
131
  return undefined;
132
+ }
89
133
  const defaultEffort = typeof entry.default_reasoning_level === "string"
90
134
  ? entry.default_reasoning_level
91
135
  : typeof nested?.default_effort === "string"
@@ -94,7 +138,7 @@ export function parseReasoningCapabilities(entry, provider, refreshedAt = new Da
94
138
  ? nested.defaultEffort
95
139
  : undefined;
96
140
  const budgetSource = isRecord(nested?.budget) ? nested.budget : undefined;
97
- const budget = budgetSource === undefined
141
+ const nestedBudget = budgetSource === undefined
98
142
  ? undefined
99
143
  : {
100
144
  ...(typeof budgetSource.min_tokens === "number"
@@ -113,12 +157,16 @@ export function parseReasoningCapabilities(entry, provider, refreshedAt = new Da
113
157
  ? { defaultTokens: budgetSource.defaultTokens }
114
158
  : {})
115
159
  };
160
+ const budget = nestedBudget ?? (enabledSupported === true ? { minTokens: 1_024 } : undefined);
161
+ const adaptive = typeof nested?.adaptive === "boolean"
162
+ ? nested.adaptive
163
+ : adaptiveSupported;
116
164
  return {
117
165
  status: unsupported ? "unsupported" : supported ? "supported" : "unknown",
118
166
  ...(efforts.length > 0 ? { efforts } : {}),
119
167
  ...(defaultEffort !== undefined ? { defaultEffort } : {}),
120
168
  ...(budget !== undefined ? { budget } : {}),
121
- ...(typeof nested?.adaptive === "boolean" ? { adaptive: nested.adaptive } : {}),
169
+ ...(adaptive !== undefined ? { adaptive } : {}),
122
170
  ...(reasoningWireShape(provider) !== undefined
123
171
  ? { wireShape: reasoningWireShape(provider) }
124
172
  : {}),
package/dist/server.js CHANGED
@@ -4,7 +4,7 @@ import { ProviderFailureError } from "@velum-labs/routekit-contracts";
4
4
  import { anthropicModelsResponse, handleAnthropicMessages, handleCountTokens, resolveClaudeModelAlias } from "./adapters/anthropic.js";
5
5
  import { effectiveModel, isStream, withDefaultModel } from "./adapters/chat.js";
6
6
  import { authorizedRequest } from "./auth.js";
7
- import { isCursorChatBody, translateCursorRequest } from "./adapters/cursor.js";
7
+ import { cursorModelAliasId, isCursorChatBody, resolveCursorModelAlias, translateCursorRequest } from "./adapters/cursor.js";
8
8
  import { handleResponses } from "./adapters/responses.js";
9
9
  import { validateAnthropicRequest, validateChatRequest, validateCountTokensRequest, validateResponsesRequest } from "./adapters/validate.js";
10
10
  import { buildModelCallRecord, MODEL_CALL_ID_HEADER, modelCallId } from "./provenance.js";
@@ -273,9 +273,22 @@ export async function startGateway(options) {
273
273
  return;
274
274
  }
275
275
  // Cursor may probe the models list relative to its BYOK base URL
276
- // (`.../v1/cursor`); mirror /v1/models there.
276
+ // (`.../v1/cursor`); mirror /v1/models there. Namespaced ids are respelled
277
+ // with dashes because Cursor's custom-model settings reject "/" in names;
278
+ // the chat route below resolves the dashed spelling back.
277
279
  if (method === "GET" && path === "/v1/cursor/models") {
278
- await pipeUpstream(res, await backend.models());
280
+ const upstream = await backend.models();
281
+ if (!upstream.ok) {
282
+ await pipeUpstream(res, upstream);
283
+ return;
284
+ }
285
+ const payload = (await upstream.json());
286
+ writeJson(res, 200, {
287
+ ...payload,
288
+ data: (payload.data ?? []).map((entry) => typeof entry.id === "string"
289
+ ? { ...entry, id: cursorModelAliasId(entry.id) }
290
+ : entry)
291
+ });
279
292
  return;
280
293
  }
281
294
  // Anthropic single-model retrieve (`GET /v1/models/{id}`): Claude Code probes
@@ -349,6 +362,9 @@ export async function startGateway(options) {
349
362
  const translated = translateCursorRequest(raw);
350
363
  if (rejectInvalid(res, validateChatRequest(translated)))
351
364
  return;
365
+ const aliased = resolveCursorModelAlias(translated.model, backend.listModelIds?.() ?? []);
366
+ if (aliased !== undefined)
367
+ translated.model = aliased;
352
368
  const body = withDefaultModel(translated, backend.defaultModel);
353
369
  await handleModelCall(res, provenance, {
354
370
  dialect: "openai-chat",
@@ -1,6 +1,6 @@
1
1
  import assert from "node:assert/strict";
2
2
  import { test } from "node:test";
3
- import { isCursorChatBody, translateCursorRequest } from "../adapters/cursor.js";
3
+ import { cursorModelAliasId, isCursorChatBody, resolveCursorModelAlias, translateCursorRequest } from "../adapters/cursor.js";
4
4
  import { startGateway } from "../server.js";
5
5
  const cursorBody = {
6
6
  model: "route-primary",
@@ -47,6 +47,18 @@ test("Cursor hybrid requests translate to chat messages and tools", () => {
47
47
  ]);
48
48
  assert.equal(translated.tools[0]?.function.name, "read_file");
49
49
  });
50
+ test("Cursor model aliases respell namespaced ids with dashes", () => {
51
+ assert.equal(cursorModelAliasId("claude-code/claude-fable-5"), "claude-code-claude-fable-5");
52
+ assert.equal(cursorModelAliasId("openai/gpt-4o"), "openai-gpt-4o");
53
+ assert.equal(cursorModelAliasId("route-primary"), "route-primary");
54
+ const served = ["claude-code/claude-fable-5", "openai/gpt-4o", "route-primary"];
55
+ assert.equal(resolveCursorModelAlias("claude-code-claude-fable-5", served), "claude-code/claude-fable-5");
56
+ assert.equal(resolveCursorModelAlias("openai-gpt-4o", served), "openai/gpt-4o");
57
+ // Served-as-spelled ids and unknown names never rewrite.
58
+ assert.equal(resolveCursorModelAlias("route-primary", served), undefined);
59
+ assert.equal(resolveCursorModelAlias("claude-fable-5", served), undefined);
60
+ assert.equal(resolveCursorModelAlias(undefined, served), undefined);
61
+ });
50
62
  test("Cursor hybrid detection rejects unrelated bodies", () => {
51
63
  assert.equal(isCursorChatBody({ input: "hello" }), true);
52
64
  assert.equal(isCursorChatBody({ messages: [] }), true);
@@ -98,3 +110,53 @@ test("RouteKit serves the Cursor hybrid through its neutral HTTP boundary", asyn
98
110
  await gateway.close();
99
111
  }
100
112
  });
113
+ test("Cursor route resolves dashed model aliases to namespaced ids", async () => {
114
+ let received;
115
+ const backend = {
116
+ defaultModel: "claude-code/claude-fable-5",
117
+ chat(body) {
118
+ received = body;
119
+ return Promise.resolve(Response.json({
120
+ id: "chatcmpl_2",
121
+ object: "chat.completion",
122
+ model: "claude-code/claude-fable-5",
123
+ choices: [
124
+ {
125
+ index: 0,
126
+ message: { role: "assistant", content: "done" },
127
+ finish_reason: "stop"
128
+ }
129
+ ]
130
+ }));
131
+ },
132
+ models: () => Promise.resolve(Response.json({
133
+ object: "list",
134
+ data: [
135
+ { id: "claude-code/claude-fable-5", object: "model" },
136
+ { id: "openai/gpt-4o", object: "model" }
137
+ ]
138
+ })),
139
+ listModelIds: () => ["claude-code/claude-fable-5", "openai/gpt-4o"],
140
+ embeddings: () => Promise.resolve(new Response(null, { status: 501 }))
141
+ };
142
+ const gateway = await startGateway({ backend });
143
+ try {
144
+ const response = await fetch(`${gateway.url()}/v1/cursor/chat/completions`, {
145
+ method: "POST",
146
+ headers: { "content-type": "application/json" },
147
+ body: JSON.stringify({
148
+ model: "claude-code-claude-fable-5",
149
+ messages: [{ role: "user", content: "hi" }]
150
+ })
151
+ });
152
+ assert.equal(response.status, 200);
153
+ assert.equal(received?.model, "claude-code/claude-fable-5");
154
+ // The models mirror advertises the dashed spelling Cursor accepts.
155
+ const models = await fetch(`${gateway.url()}/v1/cursor/models`);
156
+ assert.equal(models.status, 200);
157
+ assert.deepEqual((await models.json()).data.map((model) => model.id), ["claude-code-claude-fable-5", "openai-gpt-4o"]);
158
+ }
159
+ finally {
160
+ await gateway.close();
161
+ }
162
+ });
@@ -3,7 +3,8 @@ import { test } from "node:test";
3
3
  import { AnthropicBackend, CodexResponsesBackend, GoogleGenAiBackend } from "../provider-backends.js";
4
4
  import { anthropicToChat } from "../adapters/anthropic.js";
5
5
  import { attachReasoningSelection } from "../adapters/openai-chat-wire.js";
6
- import { SseParseError } from "../sse/parse.js";
6
+ import { ChatStreamAssembler } from "../sse/chat-assembler.js";
7
+ import { SseDecoder, SseParseError } from "../sse/parse.js";
7
8
  function sse(events, includeDone = false) {
8
9
  const body = events
9
10
  .map(({ event, data }) => `${event === undefined ? "" : `event: ${event}\n`}data: ${JSON.stringify(data)}\n\n`)
@@ -437,6 +438,294 @@ test("Codex subscription egress recovers output from completed stream items", as
437
438
  globalThis.fetch = original;
438
439
  }
439
440
  });
441
+ test("Codex subscription egress merges completed items into partial terminal output", async () => {
442
+ const original = globalThis.fetch;
443
+ globalThis.fetch = async () => sse([
444
+ {
445
+ event: "response.output_item.done",
446
+ data: {
447
+ item: {
448
+ type: "reasoning",
449
+ summary: [{ type: "summary_text", text: "brief reasoning" }]
450
+ },
451
+ output_index: 0
452
+ }
453
+ },
454
+ {
455
+ event: "response.output_item.done",
456
+ data: {
457
+ item: {
458
+ type: "message",
459
+ role: "assistant",
460
+ content: [{ type: "output_text", text: "RouteKit works" }]
461
+ },
462
+ output_index: 1
463
+ }
464
+ },
465
+ {
466
+ event: "response.completed",
467
+ data: {
468
+ response: {
469
+ output: [
470
+ {
471
+ type: "reasoning",
472
+ summary: [{ type: "summary_text", text: "brief reasoning" }]
473
+ }
474
+ ],
475
+ usage: { input_tokens: 8, output_tokens: 8, total_tokens: 16 }
476
+ }
477
+ }
478
+ }
479
+ ]);
480
+ try {
481
+ const backend = new CodexResponsesBackend({
482
+ baseUrl: "https://chatgpt.test/backend-api/codex",
483
+ apiKey: "oauth",
484
+ defaultModel: "gpt-5.5",
485
+ forceStream: true,
486
+ omitSampling: true
487
+ });
488
+ const response = await backend.chat({
489
+ stream: false,
490
+ messages: [{ role: "user", content: "Reply with: RouteKit works" }]
491
+ });
492
+ const body = (await response.json());
493
+ assert.equal(response.status, 200);
494
+ assert.equal(body.choices[0]?.message.content, "RouteKit works");
495
+ assert.equal(body.choices[0]?.message.reasoning, "brief reasoning");
496
+ }
497
+ finally {
498
+ globalThis.fetch = original;
499
+ }
500
+ });
501
+ test("Codex subscription streaming recovers text when only the completed item carries it", async () => {
502
+ const original = globalThis.fetch;
503
+ globalThis.fetch = async () => sse([
504
+ {
505
+ data: {
506
+ type: "response.output_item.done",
507
+ item: {
508
+ type: "message",
509
+ role: "assistant",
510
+ content: [{ type: "output_text", text: "RouteKit works" }]
511
+ },
512
+ output_index: 0
513
+ }
514
+ },
515
+ {
516
+ data: {
517
+ type: "response.completed",
518
+ response: {
519
+ output: [],
520
+ usage: { input_tokens: 8, output_tokens: 5, total_tokens: 13 }
521
+ }
522
+ }
523
+ }
524
+ ]);
525
+ try {
526
+ const backend = new CodexResponsesBackend({
527
+ baseUrl: "https://chatgpt.test/backend-api/codex",
528
+ apiKey: "oauth",
529
+ defaultModel: "gpt-5.5",
530
+ forceStream: true,
531
+ omitSampling: true
532
+ });
533
+ const response = await backend.chat({
534
+ stream: true,
535
+ messages: [{ role: "user", content: "Reply with: RouteKit works" }]
536
+ });
537
+ const text = await response.text();
538
+ assert.match(text, /"content":"RouteKit works"/);
539
+ assert.match(text, /"finish_reason":"stop"/);
540
+ assert.match(text, /"completion_tokens":5/);
541
+ }
542
+ finally {
543
+ globalThis.fetch = original;
544
+ }
545
+ });
546
+ test("Codex subscription streaming does not duplicate delta and completed-item text", async () => {
547
+ const original = globalThis.fetch;
548
+ globalThis.fetch = async () => sse([
549
+ {
550
+ event: "response.output_text.delta",
551
+ data: { output_index: 0, delta: "RouteKit " }
552
+ },
553
+ {
554
+ event: "response.output_item.done",
555
+ data: {
556
+ item: {
557
+ type: "message",
558
+ role: "assistant",
559
+ content: [{ type: "output_text", text: "RouteKit works" }]
560
+ },
561
+ output_index: 0
562
+ }
563
+ },
564
+ {
565
+ event: "response.completed",
566
+ data: {
567
+ response: {
568
+ output: [
569
+ {
570
+ type: "message",
571
+ role: "assistant",
572
+ content: [{ type: "output_text", text: "RouteKit works" }]
573
+ }
574
+ ],
575
+ usage: { input_tokens: 8, output_tokens: 5, total_tokens: 13 }
576
+ }
577
+ }
578
+ }
579
+ ]);
580
+ try {
581
+ const backend = new CodexResponsesBackend({
582
+ baseUrl: "https://chatgpt.test/backend-api/codex",
583
+ apiKey: "oauth",
584
+ defaultModel: "gpt-5.5"
585
+ });
586
+ const response = await backend.chat({
587
+ stream: true,
588
+ messages: [{ role: "user", content: "Reply with: RouteKit works" }]
589
+ });
590
+ const decoder = new SseDecoder();
591
+ const events = [
592
+ ...decoder.feed(await response.text()),
593
+ ...decoder.flush()
594
+ ];
595
+ const assembler = new ChatStreamAssembler();
596
+ for (const event of events)
597
+ assembler.push(event);
598
+ assert.equal(assembler.result().content, "RouteKit works");
599
+ assert.equal(assembler.result().finishReason, "stop");
600
+ }
601
+ finally {
602
+ globalThis.fetch = original;
603
+ }
604
+ });
605
+ test("Codex subscription egress rejects a silent reasoning-only completion", async () => {
606
+ const original = globalThis.fetch;
607
+ globalThis.fetch = async () => sse([
608
+ {
609
+ data: {
610
+ type: "response.output_item.done",
611
+ item: {
612
+ type: "reasoning",
613
+ summary: [{ type: "summary_text", text: "internal reasoning" }]
614
+ },
615
+ output_index: 0
616
+ }
617
+ },
618
+ {
619
+ data: {
620
+ type: "response.completed",
621
+ response: {
622
+ output: [],
623
+ usage: {
624
+ input_tokens: 22,
625
+ output_tokens: 31,
626
+ output_tokens_details: { reasoning_tokens: 22 },
627
+ total_tokens: 53
628
+ }
629
+ }
630
+ }
631
+ }
632
+ ]);
633
+ try {
634
+ const backend = new CodexResponsesBackend({
635
+ baseUrl: "https://chatgpt.test/backend-api/codex",
636
+ apiKey: "oauth",
637
+ defaultModel: "gpt-5.5",
638
+ forceStream: true,
639
+ omitSampling: true
640
+ });
641
+ const response = await backend.chat({
642
+ stream: false,
643
+ messages: [{ role: "user", content: "Reply with: RouteKit works" }]
644
+ });
645
+ assert.equal(response.status, 502);
646
+ assert.deepEqual(await response.json(), {
647
+ error: {
648
+ message: "Codex completed without assistant content or tool calls",
649
+ type: "upstream_empty_response"
650
+ }
651
+ });
652
+ }
653
+ finally {
654
+ globalThis.fetch = original;
655
+ }
656
+ });
657
+ test("Codex subscription streaming surfaces a silent reasoning-only completion as an error", async () => {
658
+ const original = globalThis.fetch;
659
+ globalThis.fetch = async () => sse([
660
+ {
661
+ event: "response.reasoning_summary_text.delta",
662
+ data: { output_index: 0, delta: "internal reasoning" }
663
+ },
664
+ {
665
+ event: "response.completed",
666
+ data: {
667
+ response: {
668
+ output: [],
669
+ usage: {
670
+ input_tokens: 22,
671
+ output_tokens: 31,
672
+ output_tokens_details: { reasoning_tokens: 22 },
673
+ total_tokens: 53
674
+ }
675
+ }
676
+ }
677
+ }
678
+ ]);
679
+ try {
680
+ const backend = new CodexResponsesBackend({
681
+ baseUrl: "https://chatgpt.test/backend-api/codex",
682
+ apiKey: "oauth",
683
+ defaultModel: "gpt-5.5"
684
+ });
685
+ const response = await backend.chat({
686
+ stream: true,
687
+ messages: [{ role: "user", content: "Reply with: RouteKit works" }]
688
+ });
689
+ const text = await response.text();
690
+ assert.match(text, /"type":"upstream_empty_response"/);
691
+ assert.doesNotMatch(text, /"finish_reason":"stop"/);
692
+ }
693
+ finally {
694
+ globalThis.fetch = original;
695
+ }
696
+ });
697
+ test("Codex subscription streaming surfaces terminal failure events", async () => {
698
+ const original = globalThis.fetch;
699
+ const backend = new CodexResponsesBackend({
700
+ baseUrl: "https://chatgpt.test/backend-api/codex",
701
+ apiKey: "oauth",
702
+ defaultModel: "gpt-5.5"
703
+ });
704
+ try {
705
+ for (const terminal of [
706
+ { event: "response.failed", data: { response: { status: "failed" } } },
707
+ {
708
+ event: "response.incomplete",
709
+ data: { response: { status: "incomplete" } }
710
+ },
711
+ {
712
+ data: { type: "response.failed", response: { status: "failed" } }
713
+ }
714
+ ]) {
715
+ globalThis.fetch = async () => sse([terminal]);
716
+ const response = await backend.chat({
717
+ stream: true,
718
+ messages: [{ role: "user", content: "Reply with: RouteKit works" }]
719
+ });
720
+ const text = await response.text();
721
+ assert.match(text, /"type":"upstream_error"/);
722
+ assert.doesNotMatch(text, /"finish_reason":"stop"/);
723
+ }
724
+ }
725
+ finally {
726
+ globalThis.fetch = original;
727
+ }
728
+ });
440
729
  test("Anthropic streaming egress preserves tool calls and terminal usage", async () => {
441
730
  const original = globalThis.fetch;
442
731
  globalThis.fetch = async () => sse([
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1,67 @@
1
+ import assert from "node:assert/strict";
2
+ import { test } from "node:test";
3
+ import { parseDiscoveredModels } from "../provider-source.js";
4
+ test("Anthropic discovery projects authoritative effort and thinking capabilities", () => {
5
+ const [model] = parseDiscoveredModels("anthropic", {
6
+ data: [
7
+ {
8
+ id: "claude-fable-5",
9
+ capabilities: {
10
+ effort: {
11
+ supported: true,
12
+ low: { supported: true },
13
+ medium: { supported: false },
14
+ high: { supported: true },
15
+ xhigh: null,
16
+ max: { supported: true }
17
+ },
18
+ thinking: {
19
+ supported: true,
20
+ types: {
21
+ adaptive: { supported: true },
22
+ enabled: { supported: true }
23
+ }
24
+ }
25
+ }
26
+ }
27
+ ]
28
+ }, "claude-code");
29
+ assert.deepEqual(model?.reasoning, {
30
+ status: "supported",
31
+ efforts: [{ id: "low" }, { id: "high" }, { id: "max" }],
32
+ budget: { minTokens: 1_024 },
33
+ adaptive: true,
34
+ wireShape: "anthropic",
35
+ provenance: "provider",
36
+ refreshedAt: model?.reasoning?.refreshedAt
37
+ });
38
+ assert.equal(model?.reasoning?.defaultEffort, undefined);
39
+ });
40
+ test("Anthropic discovery preserves explicit unsupported and missing capabilities", () => {
41
+ const models = parseDiscoveredModels("anthropic", {
42
+ data: [
43
+ {
44
+ id: "claude-no-reasoning",
45
+ capabilities: {
46
+ effort: { supported: false },
47
+ thinking: {
48
+ supported: false,
49
+ types: {
50
+ adaptive: { supported: false },
51
+ enabled: { supported: false }
52
+ }
53
+ }
54
+ }
55
+ },
56
+ { id: "claude-unknown" }
57
+ ]
58
+ }, "claude-code");
59
+ assert.deepEqual(models[0]?.reasoning, {
60
+ status: "unsupported",
61
+ adaptive: false,
62
+ wireShape: "anthropic",
63
+ provenance: "provider",
64
+ refreshedAt: models[0]?.reasoning?.refreshedAt
65
+ });
66
+ assert.equal(models[1]?.reasoning, undefined);
67
+ });
@@ -2,6 +2,7 @@ import assert from "node:assert/strict";
2
2
  import { createServer } from "node:http";
3
3
  import { test } from "node:test";
4
4
  import { OpenAiBackend } from "../backend.js";
5
+ import { AnthropicBackend } from "../provider-backends.js";
5
6
  import { MODEL_CALL_ID_HEADER } from "../provenance.js";
6
7
  import { CatalogBackend } from "../router.js";
7
8
  import { chatToResponses, customToolNames, openAiSseToResponses, responsesToChat, responsesToolRegistry } from "../adapters/responses.js";
@@ -225,6 +226,86 @@ test("serves a Responses request carrying reasoning: null end to end", async ()
225
226
  await mock.close();
226
227
  }
227
228
  });
229
+ test("Responses routes discovered Claude efforts to adaptive Anthropic egress", async () => {
230
+ const requests = [];
231
+ const anthropic = new AnthropicBackend({
232
+ baseUrl: "https://api.anthropic.test/v1",
233
+ apiKey: "unused",
234
+ transport: async (input, init) => {
235
+ requests.push(new Request(input, init));
236
+ return Response.json({
237
+ id: "msg_fable",
238
+ type: "message",
239
+ role: "assistant",
240
+ model: "claude-fable-5",
241
+ content: [{ type: "text", text: "FABLE_OK" }],
242
+ stop_reason: "end_turn",
243
+ usage: { input_tokens: 1, output_tokens: 1 }
244
+ });
245
+ }
246
+ });
247
+ const backend = await CatalogBackend.create({
248
+ config: {
249
+ providers: { "claude-code": {} },
250
+ defaultModel: "claude-code/claude-fable-5"
251
+ },
252
+ sources: {
253
+ "claude-code": {
254
+ sourceId: "claude-code",
255
+ async discoverModels() {
256
+ return [
257
+ {
258
+ id: "claude-fable-5",
259
+ reasoning: {
260
+ status: "supported",
261
+ efforts: [{ id: "low" }, { id: "high" }],
262
+ budget: { minTokens: 1_024 },
263
+ adaptive: true,
264
+ wireShape: "anthropic",
265
+ provenance: "provider"
266
+ }
267
+ }
268
+ ];
269
+ },
270
+ chat: (body, signal, options) => anthropic.chat(body, signal, options),
271
+ embeddings: async () => Response.json({})
272
+ }
273
+ }
274
+ });
275
+ const gateway = await startGateway({ backend });
276
+ try {
277
+ const supported = await fetch(`${gateway.url()}/v1/responses`, {
278
+ method: "POST",
279
+ headers: { "content-type": "application/json" },
280
+ body: JSON.stringify({
281
+ model: "claude-code/claude-fable-5",
282
+ input: "hi",
283
+ max_output_tokens: 64,
284
+ reasoning: { effort: "high" }
285
+ })
286
+ });
287
+ assert.equal(supported.status, 200);
288
+ const outbound = (await requests[0]?.json());
289
+ assert.deepEqual(outbound.thinking, { type: "adaptive" });
290
+ assert.deepEqual(outbound.output_config, { effort: "high" });
291
+ const unsupported = await fetch(`${gateway.url()}/v1/responses`, {
292
+ method: "POST",
293
+ headers: { "content-type": "application/json" },
294
+ body: JSON.stringify({
295
+ model: "claude-code/claude-fable-5",
296
+ input: "hi",
297
+ reasoning: { effort: "max" }
298
+ })
299
+ });
300
+ assert.equal(unsupported.status, 400);
301
+ assert.equal((await unsupported.json()).error
302
+ ?.message, 'reasoning effort "max" is not supported by model "claude-code/claude-fable-5"');
303
+ assert.equal(requests.length, 1);
304
+ }
305
+ finally {
306
+ await gateway.close();
307
+ }
308
+ });
228
309
  test("serves a Responses request with null optional fields end to end", async () => {
229
310
  const mock = await startMock();
230
311
  const gateway = await startGateway({
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@velum-labs/routekit-gateway",
3
3
  "private": false,
4
- "version": "0.9.4",
4
+ "version": "0.9.6",
5
5
  "repository": {
6
6
  "type": "git",
7
7
  "url": "git+https://github.com/velum-labs/handoffkit.git",
@@ -27,10 +27,10 @@
27
27
  },
28
28
  "dependencies": {
29
29
  "zod": "4.4.3",
30
- "@velum-labs/routekit-registry": "0.9.4",
31
- "@velum-labs/routekit-runtime": "0.9.4",
32
- "@velum-labs/routekit-contracts": "0.9.4",
33
- "@velum-labs/routekit-tracing": "0.9.4"
30
+ "@velum-labs/routekit-contracts": "0.9.6",
31
+ "@velum-labs/routekit-registry": "0.9.6",
32
+ "@velum-labs/routekit-runtime": "0.9.6",
33
+ "@velum-labs/routekit-tracing": "0.9.6"
34
34
  },
35
35
  "keywords": [
36
36
  "routekit",