@velum-labs/routekit-gateway 1.0.0 → 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -179,7 +179,9 @@ export function handleResponses(backend, body, modelCallId, signal, backendOptio
179
179
  const upstream = yield* backend.chat(chat, signal, requestOptions);
180
180
  if (!upstream.ok) {
181
181
  const detail = yield* gatewayTryPromise(() => upstream.text());
182
- return jsonResponse(upstream.status, { error: unwrapUpstreamError(detail) });
182
+ return jsonResponse(upstream.status, {
183
+ error: unwrapUpstreamError(detail, { preserveMetadata: true })
184
+ });
183
185
  }
184
186
  if (executor !== undefined) {
185
187
  const loopOptions = {
@@ -8,7 +8,11 @@
8
8
  * `invalid_request_error` from the backend must stay an
9
9
  * `invalid_request_error` on the door.
10
10
  */
11
- export declare function unwrapUpstreamError(detail: string): {
11
+ export declare function unwrapUpstreamError(detail: string, options?: {
12
+ readonly preserveMetadata?: boolean;
13
+ }): {
12
14
  type: string;
13
15
  message: string;
16
+ code?: string;
17
+ param?: string;
14
18
  };
@@ -8,13 +8,23 @@
8
8
  * `invalid_request_error` from the backend must stay an
9
9
  * `invalid_request_error` on the door.
10
10
  */
11
- export function unwrapUpstreamError(detail) {
11
+ export function unwrapUpstreamError(detail, options = {}) {
12
12
  try {
13
13
  const parsed = JSON.parse(detail);
14
14
  if (typeof parsed.error?.message === "string") {
15
15
  return {
16
16
  type: typeof parsed.error.type === "string" ? parsed.error.type : "api_error",
17
- message: parsed.error.message
17
+ message: parsed.error.message,
18
+ ...(options.preserveMetadata === true &&
19
+ typeof parsed.error.code === "string" &&
20
+ parsed.error.code.length > 0
21
+ ? { code: parsed.error.code }
22
+ : {}),
23
+ ...(options.preserveMetadata === true &&
24
+ typeof parsed.error.param === "string" &&
25
+ parsed.error.param.length > 0
26
+ ? { param: parsed.error.param }
27
+ : {})
18
28
  };
19
29
  }
20
30
  }
@@ -7,6 +7,7 @@ import { Effect } from "effect";
7
7
  import { REASONING_SELECTION, reasoningSelectionOf, routeKitRequestValidationErrorOf, withoutRouteKitExtensions } from "../adapters/openai-chat-wire.js";
8
8
  import { normalizeOpenAiResponsesCallIds } from "../adapters/openai-responses-wire.js";
9
9
  import { joinPath, staticBackendModelPort } from "./backend.js";
10
+ import { openaiChatUsesMaxCompletionTokens } from "./openai-reasoning.js";
10
11
  function invalidReasoningControlResponse(message, metadata = false, path) {
11
12
  return Response.json({
12
13
  error: {
@@ -17,6 +18,23 @@ function invalidReasoningControlResponse(message, metadata = false, path) {
17
18
  }
18
19
  }, { status: 400 });
19
20
  }
21
+ function withChatCompletionTokenBudget(payload) {
22
+ if (payload === null || typeof payload !== "object" || Array.isArray(payload)) {
23
+ return payload;
24
+ }
25
+ const body = payload;
26
+ const model = typeof body.model === "string" ? body.model : "";
27
+ if (!openaiChatUsesMaxCompletionTokens(model))
28
+ return payload;
29
+ if (!Object.hasOwn(body, "max_tokens"))
30
+ return payload;
31
+ const next = { ...body };
32
+ if (!Object.hasOwn(next, "max_completion_tokens")) {
33
+ next.max_completion_tokens = next.max_tokens;
34
+ }
35
+ delete next.max_tokens;
36
+ return next;
37
+ }
20
38
  /** An OpenAI HTTP backend supporting Chat Completions and native Responses. */
21
39
  export class OpenAiBackend {
22
40
  #baseUrl;
@@ -97,7 +115,7 @@ export class OpenAiBackend {
97
115
  !Array.isArray(selectedPayload)
98
116
  ? this.#openRouterReasoning(selectedPayload, selection)
99
117
  : selectedPayload;
100
- const providerPayload = withoutRouteKitExtensions(payload);
118
+ const providerPayload = withChatCompletionTokenBudget(withoutRouteKitExtensions(payload));
101
119
  return this.#request("/chat/completions", {
102
120
  method: "POST",
103
121
  headers: this.#headers(options),
@@ -5,3 +5,5 @@
5
5
  */
6
6
  import type { ModelReasoningCapabilities } from "@velum-labs/routekit-contracts";
7
7
  export declare function openaiReasoningCapabilities(modelId: string): ModelReasoningCapabilities | undefined;
8
+ /** GPT-5.x Chat Completions reject legacy `max_tokens`; use `max_completion_tokens`. */
9
+ export declare function openaiChatUsesMaxCompletionTokens(modelId: string): boolean;
@@ -23,3 +23,7 @@ export function openaiReasoningCapabilities(modelId) {
23
23
  }
24
24
  return undefined;
25
25
  }
26
+ /** GPT-5.x Chat Completions reject legacy `max_tokens`; use `max_completion_tokens`. */
27
+ export function openaiChatUsesMaxCompletionTokens(modelId) {
28
+ return /(?:^|[./])gpt-5(?:[.:-]|$)/.test(modelId);
29
+ }
@@ -332,6 +332,7 @@ export class RoutingBackend {
332
332
  error: {
333
333
  type: "invalid_request_error",
334
334
  code: "unsupported_reasoning_control",
335
+ param: requestedSelection.mode === "effort" ? "reasoning.effort" : "reasoning",
335
336
  message: selection
336
337
  }
337
338
  }, { status: 400 }));
@@ -422,6 +423,7 @@ export class RoutingBackend {
422
423
  error: {
423
424
  type: "invalid_request_error",
424
425
  code: "unsupported_reasoning_control",
426
+ param: requestedSelection.mode === "effort" ? "reasoning.effort" : "reasoning",
425
427
  message: selection
426
428
  }
427
429
  }, { status: 400 }));
@@ -126,6 +126,64 @@ test("injects the default model and pipes the completion back", async () => {
126
126
  await mock.close();
127
127
  }
128
128
  });
129
+ test("GPT-5 Chat Completions translate max_tokens to max_completion_tokens", async () => {
130
+ const mock = await startMock();
131
+ const backend = new OpenAiBackend({ baseUrl: `${mock.url}/v1` });
132
+ const gateway = await startGateway({ backend });
133
+ try {
134
+ for (const model of [
135
+ "gpt-5.6-sol",
136
+ "openai.gpt-5.6-luna",
137
+ "bedrock/openai.gpt-5.5",
138
+ "routekit/bedrock/us.openai.gpt-5.7"
139
+ ]) {
140
+ const response = await fetch(`${gateway.url()}/v1/chat/completions`, {
141
+ method: "POST",
142
+ headers: { "content-type": "application/json" },
143
+ body: JSON.stringify({
144
+ model,
145
+ max_tokens: 128,
146
+ messages: [{ role: "user", content: "hi" }]
147
+ })
148
+ });
149
+ assert.equal(response.status, 200, model);
150
+ const body = mock.lastChatBody();
151
+ assert.equal(body?.max_completion_tokens, 128, model);
152
+ assert.equal("max_tokens" in (body ?? {}), false, model);
153
+ }
154
+ const modern = await fetch(`${gateway.url()}/v1/chat/completions`, {
155
+ method: "POST",
156
+ headers: { "content-type": "application/json" },
157
+ body: JSON.stringify({
158
+ model: "bedrock/openai.gpt-5.6-terra",
159
+ max_tokens: 128,
160
+ max_completion_tokens: 256,
161
+ messages: [{ role: "user", content: "hi" }]
162
+ })
163
+ });
164
+ assert.equal(modern.status, 200);
165
+ const modernBody = mock.lastChatBody();
166
+ assert.equal(modernBody?.max_completion_tokens, 256);
167
+ assert.equal("max_tokens" in (modernBody ?? {}), false);
168
+ const legacy = await fetch(`${gateway.url()}/v1/chat/completions`, {
169
+ method: "POST",
170
+ headers: { "content-type": "application/json" },
171
+ body: JSON.stringify({
172
+ model: "gpt-4o",
173
+ max_tokens: 64,
174
+ messages: [{ role: "user", content: "hi" }]
175
+ })
176
+ });
177
+ assert.equal(legacy.status, 200);
178
+ const legacyBody = mock.lastChatBody();
179
+ assert.equal(legacyBody?.max_tokens, 64);
180
+ assert.equal("max_completion_tokens" in (legacyBody ?? {}), false);
181
+ }
182
+ finally {
183
+ await Effect.runPromise(gateway.close);
184
+ await mock.close();
185
+ }
186
+ });
129
187
  test("compound provider operations are not counted as retries", () => {
130
188
  const attribution = collectAttribution({
131
189
  effective_model: "codex/gpt-test",
@@ -1,6 +1,6 @@
1
1
  import assert from "node:assert/strict";
2
2
  import { test } from "node:test";
3
- import { openaiReasoningCapabilities } from "../providers/openai-reasoning.js";
3
+ import { openaiChatUsesMaxCompletionTokens, openaiReasoningCapabilities } from "../providers/openai-reasoning.js";
4
4
  test("OpenAI source authors verified GPT-5.5 and GPT-5.6 reasoning controls", () => {
5
5
  assert.deepEqual(openaiReasoningCapabilities("gpt-5.6-sol"), {
6
6
  status: "supported",
@@ -25,3 +25,23 @@ test("OpenAI source authors verified GPT-5.5 and GPT-5.6 reasoning controls", ()
25
25
  assert.equal(openaiReasoningCapabilities("gpt-4o"), undefined);
26
26
  assert.equal(openaiReasoningCapabilities("gpt-5.4"), undefined);
27
27
  });
28
+ test("GPT-5 Chat Completions family uses max_completion_tokens", () => {
29
+ for (const model of [
30
+ "gpt-5",
31
+ "gpt-5.5",
32
+ "gpt-5.6",
33
+ "gpt-5.6-sol",
34
+ "gpt-5.6-terra",
35
+ "gpt-5.6-luna",
36
+ "gpt-5-mini",
37
+ "openai/gpt-5.5",
38
+ "openai.gpt-5.6-sol",
39
+ "bedrock/openai.gpt-5.6-sol",
40
+ "routekit/bedrock/us.openai.gpt-5.6-sol"
41
+ ]) {
42
+ assert.equal(openaiChatUsesMaxCompletionTokens(model), true, model);
43
+ }
44
+ for (const model of ["gpt-4o", "gpt-4.1", "o3", "claude-sonnet-4-6", "not-gpt-5.6"]) {
45
+ assert.equal(openaiChatUsesMaxCompletionTokens(model), false, model);
46
+ }
47
+ });
@@ -320,7 +320,9 @@ test("Responses routes discovered Claude efforts to adaptive Anthropic egress",
320
320
  })
321
321
  });
322
322
  assert.equal(unsupported.status, 400);
323
- assert.equal((await unsupported.json()).error?.message, 'reasoning effort "max" is not supported by model "claude-code/claude-fable-5"');
323
+ const error = (await unsupported.json());
324
+ assert.equal(error.error?.param, "reasoning.effort");
325
+ assert.equal(error.error?.message, 'reasoning effort "max" is not supported by model "claude-code/claude-fable-5"');
324
326
  assert.equal(requests.length, 1);
325
327
  }
326
328
  finally {
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@velum-labs/routekit-gateway",
3
3
  "private": false,
4
- "version": "1.0.0",
4
+ "version": "1.0.2",
5
5
  "repository": {
6
6
  "type": "git",
7
7
  "url": "git+https://github.com/velum-labs/routekit.git",
@@ -45,12 +45,12 @@
45
45
  "@aws-sdk/client-bedrock": "3.1095.0",
46
46
  "@aws-sdk/client-bedrock-runtime": "3.1095.0",
47
47
  "effect": "4.0.0-rc.108",
48
- "@velum-labs/routekit-config-core": "1.0.0",
49
- "@velum-labs/routekit-contracts": "1.0.0",
50
- "@velum-labs/routekit-eval-contracts": "1.0.0",
51
- "@velum-labs/routekit-eval-core": "1.0.0",
52
- "@velum-labs/routekit-registry": "1.0.0",
53
- "@velum-labs/routekit-runtime": "1.0.0"
48
+ "@velum-labs/routekit-config-core": "1.0.2",
49
+ "@velum-labs/routekit-contracts": "1.0.2",
50
+ "@velum-labs/routekit-eval-contracts": "1.0.2",
51
+ "@velum-labs/routekit-eval-core": "1.0.2",
52
+ "@velum-labs/routekit-registry": "1.0.2",
53
+ "@velum-labs/routekit-runtime": "1.0.2"
54
54
  },
55
55
  "keywords": [
56
56
  "routekit",