@velum-labs/routekit-gateway 0.16.4 → 0.16.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,6 +8,7 @@
8
8
  * server pipes straight to the client (JSON or SSE).
9
9
  */
10
10
  import type { Backend, BackendRequestOptions } from "../backend.js";
11
+ import type { ModelReasoningCapabilities, ReasoningSelection } from "@velum-labs/routekit-contracts";
11
12
  import { type AnthropicThinkingConfig, type OpenAiChoice } from "./openai-chat-wire.js";
12
13
  import type { ExecutedSearch, ServerToolLoopEvent } from "./server-tool-loop.js";
13
14
  type AnthropicTextBlock = {
@@ -109,6 +110,20 @@ export type ClaudePickerModelRoute = {
109
110
  publicId: string;
110
111
  nativeId: string;
111
112
  provider: string;
113
+ reasoning?: ModelReasoningCapabilities;
114
+ };
115
+ export type ClaudeModelSelection = {
116
+ status: "resolved";
117
+ model: string;
118
+ clientModel: string;
119
+ selection: ReasoningSelection;
120
+ } | {
121
+ status: "unsupported_effort";
122
+ model: string;
123
+ message: string;
124
+ } | {
125
+ status: "passthrough";
126
+ model: string;
112
127
  };
113
128
  /**
114
129
  * The id a model is advertised under in Claude Code's `/model` picker. Claude
@@ -119,13 +134,32 @@ export type ClaudePickerModelRoute = {
119
134
  * identifier we control end-to-end, so any model can be made selectable.
120
135
  */
121
136
  export declare function claudeModelAlias(id: string): string;
122
- export declare function resolveClaudeModelAlias(requested: string | undefined, modelIds?: readonly string[]): string | undefined;
137
+ /** Claude Code picker spelling for one catalog route. */
138
+ export declare function claudePickerClientModel(route: ClaudePickerModelRoute): string;
139
+ export declare function resolveClaudeModelAlias(requested: string | undefined, modelIds?: readonly string[], modelRoutes?: readonly ClaudePickerModelRoute[]): string | undefined;
140
+ /**
141
+ * Resolve a Claude Code picker id (base or effort-qualified) to the served
142
+ * model and request-scoped reasoning selection.
143
+ */
144
+ export declare function resolveClaudeModelSelection(requested: string | undefined, modelIds?: readonly string[], modelRoutes?: readonly ClaudePickerModelRoute[]): ClaudeModelSelection;
145
+ /**
146
+ * Apply a request-scoped effort selection onto an Anthropic Messages body.
147
+ *
148
+ * Effort selections use adaptive thinking plus `output_config.effort`. Base /
149
+ * auto selections leave thinking and output_config untouched so provider
150
+ * defaults remain in force.
151
+ */
152
+ export declare function withClaudeReasoningSelection(body: AnthropicRequest, selection: ReasoningSelection): AnthropicRequest;
153
+ /** Qualify a Claude picker base id with a launch-time effort selection. */
154
+ export declare function claudeEffortQualifiedModel(baseClientModel: string, selection: ReasoningSelection | undefined): string;
123
155
  /**
124
156
  * Anthropic-shaped `/v1/models` discovery response. Every advertised model is
125
157
  * listed so it appears in Claude Code's `/model` picker: Anthropic-family ids
126
158
  * as-is, others under a `claude-`prefixed alias with the real id as
127
- * `display_name`. `modelIds` is the full advertised set (default model first);
128
- * when absent we fall back to the single backend default.
159
+ * `display_name`. Reasoning-capable models also advertise one
160
+ * `<base>:<effort>` entry per discovered effort. `modelIds` is the full
161
+ * advertised set (default model first); when absent we fall back to the
162
+ * single backend default.
129
163
  */
130
164
  export declare function anthropicModelsResponse(backendModel: string | undefined, modelIds?: readonly string[], modelRoutes?: readonly ClaudePickerModelRoute[]): Response;
131
165
  export {};
@@ -7,6 +7,7 @@
7
7
  * the request handler wires them to a `Backend` and returns a `Response` the
8
8
  * server pipes straight to the client (JSON or SSE).
9
9
  */
10
+ import { EFFORT_QUALIFIED_MODEL_CODEC, effortQualifiedClientModel, enumerateModelEffortVariants, resolveModelEffortVariant } from "@velum-labs/routekit-contracts";
10
11
  import { estimateTokens, randomId } from "@velum-labs/routekit-runtime";
11
12
  import { SseDecoder, SseParseError } from "../sse/parse.js";
12
13
  import { attachAnthropicMessageContent, attachAnthropicRequestMetadata, attachReasoningSelection, attachReasoningSelectionError, anthropicReasoningDetailsOf } from "./openai-chat-wire.js";
@@ -1136,22 +1137,112 @@ export const CLAUDE_ALIAS_PREFIX = "claude-";
1136
1137
  export function claudeModelAlias(id) {
1137
1138
  return isAnthropicFamilyId(id) ? id : `${CLAUDE_ALIAS_PREFIX}${id}`;
1138
1139
  }
1139
- export function resolveClaudeModelAlias(requested, modelIds = []) {
1140
- if (requested === undefined || modelIds.includes(requested))
1141
- return requested;
1142
- if (!requested.startsWith(CLAUDE_ALIAS_PREFIX))
1143
- return requested;
1140
+ /** Claude Code picker spelling for one catalog route. */
1141
+ export function claudePickerClientModel(route) {
1142
+ const displayName = route.provider === "claude-code" ? route.nativeId : route.publicId;
1143
+ return claudeModelAlias(displayName);
1144
+ }
1145
+ function claudeVariantEntries(modelIds, modelRoutes) {
1146
+ const routes = new Map(modelRoutes.map((route) => [route.publicId, route]));
1147
+ return modelIds.map((publicId) => {
1148
+ const route = routes.get(publicId) ?? {
1149
+ publicId,
1150
+ nativeId: publicId,
1151
+ provider: "unknown"
1152
+ };
1153
+ return {
1154
+ model: publicId,
1155
+ clientModel: claudePickerClientModel(route),
1156
+ ...(route.reasoning !== undefined ? { reasoning: route.reasoning } : {})
1157
+ };
1158
+ });
1159
+ }
1160
+ export function resolveClaudeModelAlias(requested, modelIds = [], modelRoutes = []) {
1161
+ if (requested === undefined)
1162
+ return undefined;
1163
+ const selection = resolveClaudeModelSelection(requested, modelIds, modelRoutes);
1164
+ return selection.status === "unsupported_effort" ? undefined : selection.model;
1165
+ }
1166
+ /**
1167
+ * Resolve a Claude Code picker id (base or effort-qualified) to the served
1168
+ * model and request-scoped reasoning selection.
1169
+ */
1170
+ export function resolveClaudeModelSelection(requested, modelIds = [], modelRoutes = []) {
1171
+ if (requested === undefined) {
1172
+ return { status: "passthrough", model: "" };
1173
+ }
1174
+ if (modelIds.includes(requested)) {
1175
+ const route = modelRoutes.find((entry) => entry.publicId === requested);
1176
+ return {
1177
+ status: "resolved",
1178
+ model: requested,
1179
+ clientModel: route === undefined ? claudeModelAlias(requested) : claudePickerClientModel(route),
1180
+ selection: { mode: "auto" }
1181
+ };
1182
+ }
1183
+ const resolved = resolveModelEffortVariant(requested, claudeVariantEntries(modelIds, modelRoutes), EFFORT_QUALIFIED_MODEL_CODEC);
1184
+ if (resolved.ok) {
1185
+ return {
1186
+ status: "resolved",
1187
+ model: resolved.model,
1188
+ clientModel: resolved.clientModel,
1189
+ selection: resolved.selection
1190
+ };
1191
+ }
1192
+ if (resolved.code === "unsupported_effort") {
1193
+ return {
1194
+ status: "unsupported_effort",
1195
+ model: requested,
1196
+ message: resolved.message
1197
+ };
1198
+ }
1199
+ // Preserve the historical base-alias fallback for callers that only pass
1200
+ // model ids (no route metadata) and for unknown ids deferred to Anthropic.
1201
+ if (!requested.startsWith(CLAUDE_ALIAS_PREFIX)) {
1202
+ return { status: "passthrough", model: requested };
1203
+ }
1144
1204
  const candidate = requested.slice(CLAUDE_ALIAS_PREFIX.length);
1145
- return modelIds.includes(candidate) && claudeModelAlias(candidate) === requested
1146
- ? candidate
1147
- : requested;
1205
+ if (modelIds.includes(candidate) && claudeModelAlias(candidate) === requested) {
1206
+ return {
1207
+ status: "resolved",
1208
+ model: candidate,
1209
+ clientModel: requested,
1210
+ selection: { mode: "auto" }
1211
+ };
1212
+ }
1213
+ return { status: "passthrough", model: requested };
1214
+ }
1215
+ /**
1216
+ * Apply a request-scoped effort selection onto an Anthropic Messages body.
1217
+ *
1218
+ * Effort selections use adaptive thinking plus `output_config.effort`. Base /
1219
+ * auto selections leave thinking and output_config untouched so provider
1220
+ * defaults remain in force.
1221
+ */
1222
+ export function withClaudeReasoningSelection(body, selection) {
1223
+ if (selection.mode !== "effort")
1224
+ return body;
1225
+ const outputConfig = body.output_config === null || body.output_config === undefined
1226
+ ? { effort: selection.effort }
1227
+ : { ...body.output_config, effort: selection.effort };
1228
+ return {
1229
+ ...body,
1230
+ thinking: { type: "adaptive" },
1231
+ output_config: outputConfig
1232
+ };
1233
+ }
1234
+ /** Qualify a Claude picker base id with a launch-time effort selection. */
1235
+ export function claudeEffortQualifiedModel(baseClientModel, selection) {
1236
+ return effortQualifiedClientModel(baseClientModel, selection, EFFORT_QUALIFIED_MODEL_CODEC);
1148
1237
  }
1149
1238
  /**
1150
1239
  * Anthropic-shaped `/v1/models` discovery response. Every advertised model is
1151
1240
  * listed so it appears in Claude Code's `/model` picker: Anthropic-family ids
1152
1241
  * as-is, others under a `claude-`prefixed alias with the real id as
1153
- * `display_name`. `modelIds` is the full advertised set (default model first);
1154
- * when absent we fall back to the single backend default.
1242
+ * `display_name`. Reasoning-capable models also advertise one
1243
+ * `<base>:<effort>` entry per discovered effort. `modelIds` is the full
1244
+ * advertised set (default model first); when absent we fall back to the
1245
+ * single backend default.
1155
1246
  */
1156
1247
  export function anthropicModelsResponse(backendModel, modelIds, modelRoutes = []) {
1157
1248
  const source = modelIds !== undefined && modelIds.length > 0
@@ -1163,18 +1254,29 @@ export function anthropicModelsResponse(backendModel, modelIds, modelRoutes = []
1163
1254
  const routes = new Map(modelRoutes.map((route) => [route.publicId, route]));
1164
1255
  const models = [];
1165
1256
  for (const realId of source) {
1166
- const route = routes.get(realId);
1167
- const displayName = route?.provider === "claude-code" ? route.nativeId : realId;
1168
- const id = claudeModelAlias(displayName);
1169
- if (seen.has(id))
1170
- continue;
1171
- seen.add(id);
1172
- models.push({
1173
- type: "model",
1174
- id,
1175
- display_name: displayName,
1176
- created_at: new Date(0).toISOString()
1177
- });
1257
+ const route = routes.get(realId) ?? {
1258
+ publicId: realId,
1259
+ nativeId: realId,
1260
+ provider: "unknown"
1261
+ };
1262
+ const displayName = route.provider === "claude-code" ? route.nativeId : realId;
1263
+ for (const variant of enumerateModelEffortVariants({
1264
+ model: realId,
1265
+ clientModel: claudePickerClientModel(route),
1266
+ ...(route.reasoning !== undefined ? { reasoning: route.reasoning } : {})
1267
+ }, EFFORT_QUALIFIED_MODEL_CODEC)) {
1268
+ if (seen.has(variant.id))
1269
+ continue;
1270
+ seen.add(variant.id);
1271
+ models.push({
1272
+ type: "model",
1273
+ id: variant.id,
1274
+ display_name: variant.selection.mode === "effort"
1275
+ ? `${displayName} (${variant.selection.effort})`
1276
+ : displayName,
1277
+ created_at: new Date(0).toISOString()
1278
+ });
1279
+ }
1178
1280
  }
1179
1281
  const ids = models.map((model) => model.id);
1180
1282
  return new Response(JSON.stringify({
@@ -47,7 +47,7 @@ export declare function cursorModelVariants(id: string, reasoning: unknown): Cur
47
47
  /**
48
48
  * Resolve a Cursor-facing model variant back to its served model and effort.
49
49
  *
50
- * Exact served ids win before suffix parsing, so a provider model whose real
50
+ * Exact served ids win before qualification, so a provider model whose real
51
51
  * id contains a colon remains addressable. Effort aliases are accepted but
52
52
  * normalized to the provider's canonical id.
53
53
  */
@@ -14,7 +14,7 @@
14
14
  * never a reason to throw — unknown item and tool types are dropped so the
15
15
  * boundary stays defensive without 4xx-ing on new shapes.
16
16
  */
17
- import { cursorModelName, resolveReasoningEffort, stripCursorNamespace } from "@velum-labs/routekit-contracts";
17
+ import { cursorModelName, EFFORT_QUALIFIED_MODEL_CODEC, enumerateModelEffortVariants, resolveModelEffortVariant, stripCursorNamespace } from "@velum-labs/routekit-contracts";
18
18
  import { droppedField } from "./dropped.js";
19
19
  import { attachReasoningSelection, attachReasoningSelectionError, hasExplicitReasoningSelection, reasoningSelectionErrorOf, reasoningSelectionOf } from "./openai-chat-wire.js";
20
20
  /** Fields copied through unchanged when present and non-null. */
@@ -53,6 +53,20 @@ export function isCursorChatBody(body) {
53
53
  export function cursorModelAliasId(id) {
54
54
  return id.replaceAll("/", "-");
55
55
  }
56
+ function cursorVariantEntries(servedIds, reasoningCapabilities) {
57
+ return servedIds.map((id) => ({
58
+ model: id,
59
+ clientModel: cursorModelName(id),
60
+ ...(reasoningCapabilities?.(id) !== undefined
61
+ ? { reasoning: reasoningCapabilities(id) }
62
+ : {})
63
+ }));
64
+ }
65
+ function cursorSelectionOf(model, selection) {
66
+ return selection.mode === "effort" && typeof selection.effort === "string"
67
+ ? { model, reasoningEffort: selection.effort }
68
+ : { model };
69
+ }
56
70
  /**
57
71
  * Expand one served model into the opaque ids Cursor can put in its picker.
58
72
  *
@@ -60,30 +74,22 @@ export function cursorModelAliasId(id) {
60
74
  * so each discovered effort is represented as a model-name variant instead.
61
75
  */
62
76
  export function cursorModelVariants(id, reasoning) {
63
- const variants = [{ model: cursorModelName(id) }];
64
- if (!isObject(reasoning) || reasoning.status !== "supported")
65
- return variants;
66
- if (!Array.isArray(reasoning.efforts))
67
- return variants;
68
- const seen = new Set();
69
- for (const option of reasoning.efforts) {
70
- if (!isObject(option) || typeof option.id !== "string" || option.id.length === 0) {
71
- continue;
72
- }
73
- if (seen.has(option.id))
74
- continue;
75
- seen.add(option.id);
76
- variants.push({
77
- model: cursorModelName(`${id}:${option.id}`),
78
- reasoningEffort: option.id
79
- });
80
- }
81
- return variants;
77
+ const capabilities = isObject(reasoning) &&
78
+ (reasoning.status === "supported" ||
79
+ reasoning.status === "unsupported" ||
80
+ reasoning.status === "unknown")
81
+ ? reasoning
82
+ : undefined;
83
+ return enumerateModelEffortVariants({
84
+ model: id,
85
+ clientModel: cursorModelName(id),
86
+ ...(capabilities !== undefined ? { reasoning: capabilities } : {})
87
+ }, EFFORT_QUALIFIED_MODEL_CODEC).map((variant) => cursorSelectionOf(variant.id, variant.selection));
82
88
  }
83
89
  /**
84
90
  * Resolve a Cursor-facing model variant back to its served model and effort.
85
91
  *
86
- * Exact served ids win before suffix parsing, so a provider model whose real
92
+ * Exact served ids win before qualification, so a provider model whose real
87
93
  * id contains a colon remains addressable. Effort aliases are accepted but
88
94
  * normalized to the provider's canonical id.
89
95
  */
@@ -95,13 +101,25 @@ export function resolveCursorModelSelection(model, servedIds, reasoningCapabilit
95
101
  const candidate = stripped ?? model;
96
102
  if (servedIds.includes(candidate))
97
103
  return { model: candidate };
98
- const suffixed = resolveCursorReasoningSuffix(candidate, servedIds, reasoningCapabilities);
99
- if (suffixed !== undefined)
100
- return suffixed;
104
+ const resolved = resolveModelEffortVariant(candidate, cursorVariantEntries(servedIds, reasoningCapabilities), EFFORT_QUALIFIED_MODEL_CODEC);
105
+ if (resolved.ok)
106
+ return cursorSelectionOf(resolved.model, resolved.selection);
101
107
  const legacy = servedIds.find((id) => id.includes("/") && cursorModelAliasId(id) === candidate);
102
108
  if (legacy !== undefined)
103
109
  return { model: legacy };
104
- return resolveLegacyCursorReasoningSuffix(candidate, servedIds, reasoningCapabilities);
110
+ const legacyEntries = servedIds
111
+ .filter((id) => id.includes("/"))
112
+ .map((id) => ({
113
+ model: id,
114
+ clientModel: cursorModelAliasId(id),
115
+ ...(reasoningCapabilities?.(id) !== undefined
116
+ ? { reasoning: reasoningCapabilities(id) }
117
+ : {})
118
+ }));
119
+ const legacyResolved = resolveModelEffortVariant(candidate, legacyEntries, EFFORT_QUALIFIED_MODEL_CODEC);
120
+ return legacyResolved.ok
121
+ ? cursorSelectionOf(legacyResolved.model, legacyResolved.selection)
122
+ : undefined;
105
123
  }
106
124
  /**
107
125
  * Resolve a Cursor-facing model name back to a served id.
@@ -113,44 +131,6 @@ export function resolveCursorModelSelection(model, servedIds, reasoningCapabilit
113
131
  export function resolveCursorModelAlias(model, servedIds) {
114
132
  return resolveCursorModelSelection(model, servedIds)?.model;
115
133
  }
116
- function resolveCursorReasoningSuffix(candidate, servedIds, reasoningCapabilities) {
117
- if (reasoningCapabilities === undefined)
118
- return undefined;
119
- for (const id of [...servedIds].sort((left, right) => right.length - left.length)) {
120
- const prefix = `${id}:`;
121
- if (!candidate.startsWith(prefix))
122
- continue;
123
- const requested = candidate.slice(prefix.length);
124
- if (requested.length === 0)
125
- continue;
126
- const capabilities = reasoningCapabilities(id);
127
- if (capabilities === undefined || capabilities.status !== "supported")
128
- continue;
129
- const effort = resolveReasoningEffort(capabilities, requested);
130
- if (effort !== undefined)
131
- return { model: id, reasoningEffort: effort };
132
- }
133
- return undefined;
134
- }
135
- function resolveLegacyCursorReasoningSuffix(candidate, servedIds, reasoningCapabilities) {
136
- if (reasoningCapabilities === undefined)
137
- return undefined;
138
- for (const id of servedIds) {
139
- if (!id.includes("/"))
140
- continue;
141
- const prefix = `${cursorModelAliasId(id)}:`;
142
- if (!candidate.startsWith(prefix))
143
- continue;
144
- const requested = candidate.slice(prefix.length);
145
- const capabilities = reasoningCapabilities(id);
146
- if (capabilities === undefined || capabilities.status !== "supported")
147
- continue;
148
- const effort = resolveReasoningEffort(capabilities, requested);
149
- if (effort !== undefined)
150
- return { model: id, reasoningEffort: effort };
151
- }
152
- return undefined;
153
- }
154
134
  /**
155
135
  * Map a Cursor BYOK request body onto a Chat Completions body.
156
136
  *
package/dist/index.d.ts CHANGED
@@ -21,8 +21,8 @@ export { effectiveModel, isStream, withDefaultModel } from "./adapters/chat.js";
21
21
  export { ANTHROPIC_MESSAGE_CONTENT, ANTHROPIC_REQUEST_METADATA, REASONING_SELECTION, ROUTEKIT_EXTENSION_KEY, attachAnthropicMessageContent, attachAnthropicRequestMetadata, attachReasoningSelection, anthropicMessageContentOf, anthropicRequestMetadataOf, routeKitRequestValidationErrorOf, reasoningSelectionErrorOf, reasoningSelectionOf, responsesReasoningMetadataErrorOf, withoutRouteKitExtensions } from "./adapters/openai-chat-wire.js";
22
22
  export type { AnthropicNativeContentBlock, AnthropicRequestMetadata, RouteKitMessageEnvelope, RouteKitReasoningEnvelope } from "./adapters/openai-chat-wire.js";
23
23
  export { isCursorChatBody, translateCursorRequest } from "./adapters/cursor.js";
24
- export { anthropicModelsResponse, anthropicToChat, CLAUDE_ALIAS_PREFIX, chatToAnthropicMessage, claudeModelAlias, countTokensEstimate, handleAnthropicMessages, handleCountTokens, mapStopReason, openAiSseToAnthropic } from "./adapters/anthropic.js";
25
- export type { AnthropicRequest } from "./adapters/anthropic.js";
24
+ export { anthropicModelsResponse, anthropicToChat, CLAUDE_ALIAS_PREFIX, chatToAnthropicMessage, claudeEffortQualifiedModel, claudeModelAlias, claudePickerClientModel, countTokensEstimate, handleAnthropicMessages, handleCountTokens, mapStopReason, openAiSseToAnthropic, resolveClaudeModelAlias, resolveClaudeModelSelection, withClaudeReasoningSelection } from "./adapters/anthropic.js";
25
+ export type { AnthropicRequest, ClaudeModelSelection, ClaudePickerModelRoute } from "./adapters/anthropic.js";
26
26
  export { chatToResponses, customToolNames, handleResponses, openAiSseToResponses, responsesToChat, responsesToolRegistry } from "./adapters/responses.js";
27
27
  export type { ResponsesRequest, ResponsesToolKind, ResponsesToolRegistry } from "./adapters/responses.js";
28
28
  export { MAX_WEB_SEARCHES_PER_TURN, resolveWebSearchExecutor } from "./adapters/web-search.js";
package/dist/index.js CHANGED
@@ -11,7 +11,7 @@ export { CapacityPool } from "./capacity-pool.js";
11
11
  export { effectiveModel, isStream, withDefaultModel } from "./adapters/chat.js";
12
12
  export { ANTHROPIC_MESSAGE_CONTENT, ANTHROPIC_REQUEST_METADATA, REASONING_SELECTION, ROUTEKIT_EXTENSION_KEY, attachAnthropicMessageContent, attachAnthropicRequestMetadata, attachReasoningSelection, anthropicMessageContentOf, anthropicRequestMetadataOf, routeKitRequestValidationErrorOf, reasoningSelectionErrorOf, reasoningSelectionOf, responsesReasoningMetadataErrorOf, withoutRouteKitExtensions } from "./adapters/openai-chat-wire.js";
13
13
  export { isCursorChatBody, translateCursorRequest } from "./adapters/cursor.js";
14
- export { anthropicModelsResponse, anthropicToChat, CLAUDE_ALIAS_PREFIX, chatToAnthropicMessage, claudeModelAlias, countTokensEstimate, handleAnthropicMessages, handleCountTokens, mapStopReason, openAiSseToAnthropic } from "./adapters/anthropic.js";
14
+ export { anthropicModelsResponse, anthropicToChat, CLAUDE_ALIAS_PREFIX, chatToAnthropicMessage, claudeEffortQualifiedModel, claudeModelAlias, claudePickerClientModel, countTokensEstimate, handleAnthropicMessages, handleCountTokens, mapStopReason, openAiSseToAnthropic, resolveClaudeModelAlias, resolveClaudeModelSelection, withClaudeReasoningSelection } from "./adapters/anthropic.js";
15
15
  export { chatToResponses, customToolNames, handleResponses, openAiSseToResponses, responsesToChat, responsesToolRegistry } from "./adapters/responses.js";
16
16
  export { MAX_WEB_SEARCHES_PER_TURN, resolveWebSearchExecutor } from "./adapters/web-search.js";
17
17
  export { DIALECT_DROPPED_ATTRIBUTE, droppedField, resetDroppedFieldWarnings, withDroppedFieldSpan } from "./adapters/dropped.js";
package/dist/router.js CHANGED
@@ -1,4 +1,4 @@
1
- import { resolveReasoningEffort } from "@velum-labs/routekit-contracts";
1
+ import { resolveReasoningSelection } from "@velum-labs/routekit-contracts";
2
2
  import { z } from "zod";
3
3
  import { attachReasoningSelection, reasoningSelectionOf, routeKitRequestValidationErrorOf } from "./adapters/openai-chat-wire.js";
4
4
  import { BedrockProviderSource } from "./bedrock-source.js";
@@ -686,9 +686,6 @@ export class CatalogBackend {
686
686
  };
687
687
  }
688
688
  #validatedReasoning(entry, selection) {
689
- if (selection.mode === "auto" || selection.mode === "disabled") {
690
- return selection;
691
- }
692
689
  const capability = entry.reasoning;
693
690
  if (selection.mode === "effort" &&
694
691
  selection.effort === "none" &&
@@ -697,34 +694,25 @@ export class CatalogBackend {
697
694
  capability.status === "unsupported")) {
698
695
  return { mode: "disabled" };
699
696
  }
700
- if (capability === undefined || capability.status === "unknown") {
697
+ const resolved = resolveReasoningSelection(capability, selection);
698
+ if (resolved.ok)
699
+ return resolved.selection;
700
+ if (resolved.code === "unknown_capability") {
701
701
  return `model "${entry.publicId}" has no discovered reasoning controls`;
702
702
  }
703
- if (capability.status === "unsupported") {
703
+ if (resolved.code === "unsupported") {
704
704
  return `model "${entry.publicId}" does not support reasoning controls`;
705
705
  }
706
- if (selection.mode === "effort") {
707
- const effort = resolveReasoningEffort(capability, selection.effort);
708
- return effort === undefined
709
- ? `reasoning effort "${selection.effort}" is not supported by model "${entry.publicId}"`
710
- : { mode: "effort", effort };
706
+ if (resolved.code === "unsupported_effort") {
707
+ return `reasoning effort "${selection.mode === "effort" ? selection.effort : ""}" is not supported by model "${entry.publicId}"`;
711
708
  }
712
- if (selection.mode === "adaptive") {
713
- return capability.adaptive === true
714
- ? selection
715
- : `adaptive reasoning is not supported by model "${entry.publicId}"`;
709
+ if (resolved.code === "unsupported_adaptive") {
710
+ return `adaptive reasoning is not supported by model "${entry.publicId}"`;
716
711
  }
717
- const budget = capability.budget;
718
- if (budget === undefined) {
712
+ if (resolved.code === "unsupported_budget") {
719
713
  return `reasoning token budgets are not supported by model "${entry.publicId}"`;
720
714
  }
721
- if (budget.minTokens !== undefined && selection.budgetTokens < budget.minTokens) {
722
- return `reasoning budget must be at least ${budget.minTokens} tokens`;
723
- }
724
- if (budget.maxTokens !== undefined && selection.budgetTokens > budget.maxTokens) {
725
- return `reasoning budget must be at most ${budget.maxTokens} tokens`;
726
- }
727
- return selection;
715
+ return resolved.message;
728
716
  }
729
717
  }
730
718
  export function isSubscriptionProvider(provider) {
package/dist/server.js CHANGED
@@ -1,6 +1,6 @@
1
1
  import { createServer } from "node:http";
2
- import { isCodexPickerEligibleModel, ProviderFailureError } from "@velum-labs/routekit-contracts";
3
- import { anthropicModelsResponse, handleAnthropicMessages, handleCountTokens, resolveClaudeModelAlias } from "./adapters/anthropic.js";
2
+ import { isCodexPickerEligibleModel, ProviderFailureError, reasoningEffortDescriptors } from "@velum-labs/routekit-contracts";
3
+ import { anthropicModelsResponse, handleAnthropicMessages, handleCountTokens, resolveClaudeModelSelection, withClaudeReasoningSelection } from "./adapters/anthropic.js";
4
4
  import { effectiveModel, isStream, withDefaultModel } from "./adapters/chat.js";
5
5
  import { authorizedRequest, parsePrincipalHeader, ROUTEKIT_PRINCIPAL_HEADER } from "./auth.js";
6
6
  import { cursorModelVariants, isCursorChatBody, resolveCursorModelSelection, translateCursorRequest } from "./adapters/cursor.js";
@@ -12,9 +12,9 @@ import { buildModelCallRecord, MODEL_CALL_ID_HEADER, modelCallId } from "./prove
12
12
  import { waitForDrainOrClose } from "./http-response.js";
13
13
  import { NoModelAvailableError, UnknownModelError } from "./router.js";
14
14
  function codexModelInfo(id, priority, reasoning) {
15
- const levels = (reasoning?.efforts ?? []).map((effort) => ({
15
+ const levels = reasoningEffortDescriptors(reasoning).map((effort) => ({
16
16
  effort: effort.id,
17
- description: effort.description ?? effort.label ?? effort.id
17
+ description: effort.label
18
18
  }));
19
19
  return {
20
20
  slug: id,
@@ -67,6 +67,18 @@ function catalogModelRoutes(backend) {
67
67
  return route === undefined ? [] : [route];
68
68
  });
69
69
  }
70
+ function resolveClaudeSelection(backend, requested) {
71
+ return resolveClaudeModelSelection(requested, backend.listModelIds?.() ?? [], catalogModelRoutes(backend));
72
+ }
73
+ function writeClaudeSelectionError(res, selection) {
74
+ writeJson(res, 400, {
75
+ type: "error",
76
+ error: {
77
+ type: "invalid_request_error",
78
+ message: selection.message
79
+ }
80
+ });
81
+ }
70
82
  function resolveNativeModelRoute(backend, provider, requested) {
71
83
  if (backend.resolveModelRoute === undefined)
72
84
  return undefined;
@@ -314,13 +326,19 @@ export async function startGateway(options) {
314
326
  }
315
327
  // Anthropic single-model retrieve (`GET /v1/models/{id}`): Claude Code probes
316
328
  // this to validate a selected model before its first turn. Echo the id back
317
- // so any advertised/aliased id validates; routing is decided at chat time.
329
+ // so any advertised/aliased/effort-qualified id validates; routing is decided
330
+ // at chat time.
318
331
  if (method === "GET" && path.startsWith("/v1/models/")) {
319
332
  const id = decodeURIComponent(path.slice("/v1/models/".length));
320
- const alias = resolveClaudeModelAlias(id, backend.listModelIds?.());
333
+ const selection = resolveClaudeSelection(backend, id);
334
+ if (selection.status === "unsupported_effort") {
335
+ writeClaudeSelectionError(res, selection);
336
+ return;
337
+ }
338
+ const alias = selection.model;
321
339
  const route = backend.resolveModelRoute?.(alias, "claude-code");
322
340
  const resolved = route?.publicId ?? alias;
323
- if (resolved === undefined ||
341
+ if (resolved.length === 0 ||
324
342
  (backend.resolveModelRoute !== undefined && route === undefined) ||
325
343
  (backend.resolveModelRoute === undefined &&
326
344
  !(backend.servesModel?.(resolved) ?? false) &&
@@ -450,7 +468,12 @@ export async function startGateway(options) {
450
468
  if (rejectInvalid(res, validateCountTokensRequest(raw)))
451
469
  return;
452
470
  const rawBody = raw;
453
- const alias = resolveClaudeModelAlias(rawBody.model, backend.listModelIds?.());
471
+ const selection = resolveClaudeSelection(backend, rawBody.model);
472
+ if (selection.status === "unsupported_effort") {
473
+ writeClaudeSelectionError(res, selection);
474
+ return;
475
+ }
476
+ const alias = selection.model.length > 0 ? selection.model : undefined;
454
477
  const route = backend.resolveModelRoute?.(alias, "claude-code");
455
478
  if (alias !== undefined &&
456
479
  backend.resolveModelRoute !== undefined &&
@@ -464,16 +487,21 @@ export async function startGateway(options) {
464
487
  });
465
488
  return;
466
489
  }
490
+ const normalizedBody = selection.status === "resolved" &&
491
+ alias !== undefined &&
492
+ alias !== rawBody.model
493
+ ? withModel(rawBody, alias)
494
+ : rawBody;
467
495
  if (anthropicRelay?.countTokens !== undefined &&
468
496
  (route?.provider === "claude-code" ||
469
497
  backend.resolveModelRoute === undefined)) {
470
498
  const relayBody = route?.provider === "claude-code"
471
- ? withModel(rawBody, route.nativeId)
472
- : rawBody;
499
+ ? withModel(normalizedBody, route.nativeId)
500
+ : normalizedBody;
473
501
  await pipeUpstream(res, await anthropicRelay.countTokens(req.headers, relayBody));
474
502
  return;
475
503
  }
476
- await pipeUpstream(res, handleCountTokens(rawBody));
504
+ await pipeUpstream(res, handleCountTokens(normalizedBody));
477
505
  return;
478
506
  }
479
507
  if (method === "POST" && path === "/v1/messages") {
@@ -483,19 +511,33 @@ export async function startGateway(options) {
483
511
  if (rejectInvalid(res, validateAnthropicRequest(raw)))
484
512
  return;
485
513
  const rawBody = raw;
486
- const resolvedModel = resolveClaudeModelAlias(rawBody.model, backend.listModelIds?.());
514
+ const selection = resolveClaudeSelection(backend, rawBody.model);
515
+ if (selection.status === "unsupported_effort") {
516
+ writeClaudeSelectionError(res, selection);
517
+ return;
518
+ }
519
+ const resolvedModel = selection.status === "resolved"
520
+ ? selection.model
521
+ : selection.model.length > 0
522
+ ? selection.model
523
+ : undefined;
487
524
  // Defer an unknown Claude model to the Anthropic adapter so it can emit
488
525
  // the native Anthropic error envelope. The later handleModelCall wrapper
489
526
  // still records attribution and provenance for the rejected request.
490
527
  const route = backend.resolveModelRoute?.(resolvedModel, "claude-code");
491
528
  const canonicalModel = route?.publicId ?? resolvedModel;
492
- const body = canonicalModel === rawBody.model || canonicalModel === undefined
493
- ? rawBody
494
- : withModel(rawBody, canonicalModel);
529
+ const selectedBody = selection.status === "resolved"
530
+ ? withClaudeReasoningSelection(canonicalModel === rawBody.model || canonicalModel === undefined
531
+ ? rawBody
532
+ : withModel(rawBody, canonicalModel), selection.selection)
533
+ : canonicalModel === rawBody.model || canonicalModel === undefined
534
+ ? rawBody
535
+ : withModel(rawBody, canonicalModel);
536
+ const body = selectedBody;
495
537
  const requestedModel = typeof body.model === "string" ? body.model : undefined;
496
538
  if (anthropicRelay !== undefined &&
497
539
  route?.provider === "claude-code") {
498
- const relayBody = withModel(rawBody, route.nativeId);
540
+ const relayBody = withClaudeReasoningSelection(withModel(rawBody, route.nativeId), selection.status === "resolved" ? selection.selection : { mode: "auto" });
499
541
  await dispatchModelCall({
500
542
  dialect: "anthropic-messages",
501
543
  body,
@@ -1,7 +1,7 @@
1
1
  import assert from "node:assert/strict";
2
2
  import { createServer } from "node:http";
3
3
  import { test } from "node:test";
4
- import { anthropicModelsResponse, anthropicToChat, chatToAnthropicMessage, claudeModelAlias, mapStopReason, openAiSseToAnthropic, resolveClaudeModelAlias } from "../adapters/anthropic.js";
4
+ import { anthropicModelsResponse, anthropicToChat, chatToAnthropicMessage, claudeModelAlias, mapStopReason, openAiSseToAnthropic, resolveClaudeModelAlias, resolveClaudeModelSelection, withClaudeReasoningSelection } from "../adapters/anthropic.js";
5
5
  import { OpenAiBackend } from "../backend.js";
6
6
  import { CatalogBackend } from "../router.js";
7
7
  import { MODEL_CALL_ID_HEADER } from "../provenance.js";
@@ -83,6 +83,74 @@ test("anthropicModelsResponse exposes Claude subscription models as bare native
83
83
  }
84
84
  ]);
85
85
  });
86
+ test("anthropicModelsResponse emits base plus discovered effort variants", async () => {
87
+ const reasoning = {
88
+ status: "supported",
89
+ efforts: [
90
+ { id: "low" },
91
+ { id: "high", aliases: ["max"] },
92
+ { id: "high" }
93
+ ],
94
+ provenance: "provider"
95
+ };
96
+ const response = anthropicModelsResponse("claude-code/claude-sonnet-4-6", ["claude-code/claude-sonnet-4-6", "codex/gpt-5.5", "openai/gpt-4o"], [
97
+ {
98
+ publicId: "claude-code/claude-sonnet-4-6",
99
+ nativeId: "claude-sonnet-4-6",
100
+ provider: "claude-code",
101
+ reasoning
102
+ },
103
+ {
104
+ publicId: "codex/gpt-5.5",
105
+ nativeId: "gpt-5.5",
106
+ provider: "codex",
107
+ reasoning
108
+ },
109
+ {
110
+ publicId: "openai/gpt-4o",
111
+ nativeId: "gpt-4o",
112
+ provider: "openai"
113
+ }
114
+ ]);
115
+ const body = (await response.json());
116
+ assert.deepEqual(body.data.map((model) => model.id), [
117
+ "claude-sonnet-4-6",
118
+ "claude-sonnet-4-6:low",
119
+ "claude-sonnet-4-6:high",
120
+ "claude-codex/gpt-5.5",
121
+ "claude-codex/gpt-5.5:low",
122
+ "claude-codex/gpt-5.5:high",
123
+ "claude-openai/gpt-4o"
124
+ ]);
125
+ assert.equal(body.data.find((model) => model.id === "claude-sonnet-4-6:high")?.display_name, "claude-sonnet-4-6 (high)");
126
+ assert.deepEqual(resolveClaudeModelSelection("claude-codex/gpt-5.5:max", ["claude-code/claude-sonnet-4-6", "codex/gpt-5.5", "openai/gpt-4o"], [
127
+ {
128
+ publicId: "codex/gpt-5.5",
129
+ nativeId: "gpt-5.5",
130
+ provider: "codex",
131
+ reasoning
132
+ }
133
+ ]), {
134
+ status: "resolved",
135
+ model: "codex/gpt-5.5",
136
+ clientModel: "claude-codex/gpt-5.5",
137
+ selection: { mode: "effort", effort: "high" }
138
+ });
139
+ assert.equal(resolveClaudeModelSelection("claude-codex/gpt-5.5:unknown", ["codex/gpt-5.5"], [
140
+ {
141
+ publicId: "codex/gpt-5.5",
142
+ nativeId: "gpt-5.5",
143
+ provider: "codex",
144
+ reasoning
145
+ }
146
+ ]).status, "unsupported_effort");
147
+ assert.deepEqual(withClaudeReasoningSelection({ model: "claude-x", messages: [{ role: "user", content: "hi" }] }, { mode: "effort", effort: "high" }), {
148
+ model: "claude-x",
149
+ messages: [{ role: "user", content: "hi" }],
150
+ thinking: { type: "adaptive" },
151
+ output_config: { effort: "high" }
152
+ });
153
+ });
86
154
  test("anthropicToChat tolerates thinking: null (same failure class as Responses reasoning: null)", () => {
87
155
  const chat = anthropicToChat({ model: "claude-x", messages: [{ role: "user", content: "hi" }], thinking: null }, "claude-x");
88
156
  assert.equal(chat.reasoning_effort, undefined);
@@ -791,3 +859,151 @@ test("Claude picker aliases use the canonical catalog and pooled native relay",
791
859
  await gateway.close();
792
860
  }
793
861
  });
862
+ test("Claude effort variants apply request-scoped effort on native and translated routes", async () => {
863
+ const reasoning = {
864
+ status: "supported",
865
+ efforts: [{ id: "low" }, { id: "high", aliases: ["max"] }],
866
+ provenance: "provider"
867
+ };
868
+ const sourceCalls = [];
869
+ const source = (sourceId) => ({
870
+ sourceId,
871
+ discoverModels: async () => [
872
+ {
873
+ id: sourceId === "claude-code" ? "claude-sonnet-4-6" : "gpt-5.5",
874
+ reasoning
875
+ }
876
+ ],
877
+ chat: async (body) => {
878
+ sourceCalls.push(body);
879
+ return Response.json({
880
+ id: "chatcmpl_effort",
881
+ choices: [
882
+ {
883
+ index: 0,
884
+ message: { role: "assistant", content: "TRANSLATED_OK" },
885
+ finish_reason: "stop"
886
+ }
887
+ ],
888
+ usage: { prompt_tokens: 1, completion_tokens: 1 }
889
+ });
890
+ },
891
+ embeddings: async () => Response.json({})
892
+ });
893
+ const backend = await CatalogBackend.create({
894
+ config: {
895
+ providers: { "claude-code": {}, codex: {} },
896
+ defaultModel: "claude-code/claude-sonnet-4-6"
897
+ },
898
+ sources: {
899
+ "claude-code": source("claude-code"),
900
+ codex: source("codex")
901
+ }
902
+ });
903
+ const relayedBodies = [];
904
+ const relay = {
905
+ dialect: "anthropic",
906
+ shouldRelay: () => false,
907
+ relay: async (_headers, body) => {
908
+ relayedBodies.push(body);
909
+ return Response.json({
910
+ id: "msg_effort",
911
+ type: "message",
912
+ role: "assistant",
913
+ model: body.model,
914
+ content: [{ type: "text", text: "NATIVE_OK" }],
915
+ stop_reason: "end_turn",
916
+ stop_sequence: null,
917
+ usage: { input_tokens: 1, output_tokens: 1 }
918
+ });
919
+ }
920
+ };
921
+ const gateway = await startGateway({
922
+ backend,
923
+ providerRelays: { anthropic: relay }
924
+ });
925
+ try {
926
+ const catalog = (await (await fetch(`${gateway.url()}/v1/models`, {
927
+ headers: { "anthropic-version": "2023-06-01" }
928
+ })).json());
929
+ assert.ok(catalog.data.some((model) => model.id === "claude-sonnet-4-6"));
930
+ assert.ok(catalog.data.some((model) => model.id === "claude-sonnet-4-6:high"));
931
+ assert.ok(catalog.data.some((model) => model.id === "claude-codex/gpt-5.5:low"));
932
+ for (const effort of ["low", "high"]) {
933
+ const response = await fetch(`${gateway.url()}/v1/messages`, {
934
+ method: "POST",
935
+ headers: {
936
+ "content-type": "application/json",
937
+ "anthropic-version": "2023-06-01"
938
+ },
939
+ body: JSON.stringify({
940
+ model: `claude-sonnet-4-6:${effort}`,
941
+ max_tokens: 32,
942
+ messages: [{ role: "user", content: "hi" }]
943
+ })
944
+ });
945
+ assert.equal(response.status, 200);
946
+ }
947
+ assert.deepEqual(relayedBodies.map((body) => [
948
+ body.model,
949
+ body.thinking?.type,
950
+ body.output_config?.effort
951
+ ]), [
952
+ ["claude-sonnet-4-6", "adaptive", "low"],
953
+ ["claude-sonnet-4-6", "adaptive", "high"]
954
+ ]);
955
+ const base = await fetch(`${gateway.url()}/v1/messages`, {
956
+ method: "POST",
957
+ headers: {
958
+ "content-type": "application/json",
959
+ "anthropic-version": "2023-06-01"
960
+ },
961
+ body: JSON.stringify({
962
+ model: "claude-sonnet-4-6",
963
+ max_tokens: 32,
964
+ messages: [{ role: "user", content: "hi" }]
965
+ })
966
+ });
967
+ assert.equal(base.status, 200);
968
+ assert.equal(relayedBodies.at(-1)?.output_config, undefined);
969
+ assert.equal(relayedBodies.at(-1)?.thinking, undefined);
970
+ const translated = await fetch(`${gateway.url()}/v1/messages`, {
971
+ method: "POST",
972
+ headers: {
973
+ "content-type": "application/json",
974
+ "anthropic-version": "2023-06-01"
975
+ },
976
+ body: JSON.stringify({
977
+ model: "claude-codex/gpt-5.5:max",
978
+ max_tokens: 32,
979
+ messages: [{ role: "user", content: "hi" }]
980
+ })
981
+ });
982
+ assert.equal(translated.status, 200);
983
+ assert.equal(sourceCalls.length, 1);
984
+ assert.equal(sourceCalls[0]?.model, "gpt-5.5");
985
+ assert.equal(sourceCalls[0]?.reasoning_effort, "high");
986
+ const rejected = await fetch(`${gateway.url()}/v1/messages`, {
987
+ method: "POST",
988
+ headers: {
989
+ "content-type": "application/json",
990
+ "anthropic-version": "2023-06-01"
991
+ },
992
+ body: JSON.stringify({
993
+ model: "claude-sonnet-4-6:bogus",
994
+ max_tokens: 32,
995
+ messages: [{ role: "user", content: "hi" }]
996
+ })
997
+ });
998
+ assert.equal(rejected.status, 400);
999
+ assert.match(await rejected.text(), /not supported/);
1000
+ assert.equal(relayedBodies.length, 3);
1001
+ assert.equal(sourceCalls.length, 1);
1002
+ const retrieve = await fetch(`${gateway.url()}/v1/models/${encodeURIComponent("claude-sonnet-4-6:high")}`, { headers: { "anthropic-version": "2023-06-01" } });
1003
+ assert.equal(retrieve.status, 200);
1004
+ assert.equal((await retrieve.json()).id, "claude-sonnet-4-6:high");
1005
+ }
1006
+ finally {
1007
+ await gateway.close();
1008
+ }
1009
+ });
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@velum-labs/routekit-gateway",
3
3
  "private": false,
4
- "version": "0.16.4",
4
+ "version": "0.16.6",
5
5
  "repository": {
6
6
  "type": "git",
7
7
  "url": "git+https://github.com/velum-labs/routekit.git",
@@ -29,10 +29,10 @@
29
29
  "@aws-sdk/client-bedrock": "3.1095.0",
30
30
  "@aws-sdk/client-bedrock-runtime": "3.1095.0",
31
31
  "zod": "4.4.3",
32
- "@velum-labs/routekit-contracts": "0.16.4",
33
- "@velum-labs/routekit-registry": "0.16.4",
34
- "@velum-labs/routekit-runtime": "0.16.4",
35
- "@velum-labs/routekit-tracing": "0.16.4"
32
+ "@velum-labs/routekit-contracts": "0.16.6",
33
+ "@velum-labs/routekit-registry": "0.16.6",
34
+ "@velum-labs/routekit-runtime": "0.16.6",
35
+ "@velum-labs/routekit-tracing": "0.16.6"
36
36
  },
37
37
  "keywords": [
38
38
  "routekit",