@velum-labs/routekit-gateway 1.0.0 → 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/responses.js +3 -1
- package/dist/adapters/upstream-error.d.ts +5 -1
- package/dist/adapters/upstream-error.js +12 -2
- package/dist/providers/openai-backend.js +19 -1
- package/dist/providers/openai-reasoning.d.ts +2 -0
- package/dist/providers/openai-reasoning.js +4 -0
- package/dist/routing/router.js +2 -0
- package/dist/test/chat.test.js +58 -0
- package/dist/test/openai-reasoning.test.js +21 -1
- package/dist/test/responses-reasoning.test.js +3 -1
- package/package.json +7 -7
|
@@ -179,7 +179,9 @@ export function handleResponses(backend, body, modelCallId, signal, backendOptio
|
|
|
179
179
|
const upstream = yield* backend.chat(chat, signal, requestOptions);
|
|
180
180
|
if (!upstream.ok) {
|
|
181
181
|
const detail = yield* gatewayTryPromise(() => upstream.text());
|
|
182
|
-
return jsonResponse(upstream.status, {
|
|
182
|
+
return jsonResponse(upstream.status, {
|
|
183
|
+
error: unwrapUpstreamError(detail, { preserveMetadata: true })
|
|
184
|
+
});
|
|
183
185
|
}
|
|
184
186
|
if (executor !== undefined) {
|
|
185
187
|
const loopOptions = {
|
|
@@ -8,7 +8,11 @@
|
|
|
8
8
|
* `invalid_request_error` from the backend must stay an
|
|
9
9
|
* `invalid_request_error` on the door.
|
|
10
10
|
*/
|
|
11
|
-
export declare function unwrapUpstreamError(detail: string
|
|
11
|
+
export declare function unwrapUpstreamError(detail: string, options?: {
|
|
12
|
+
readonly preserveMetadata?: boolean;
|
|
13
|
+
}): {
|
|
12
14
|
type: string;
|
|
13
15
|
message: string;
|
|
16
|
+
code?: string;
|
|
17
|
+
param?: string;
|
|
14
18
|
};
|
|
@@ -8,13 +8,23 @@
|
|
|
8
8
|
* `invalid_request_error` from the backend must stay an
|
|
9
9
|
* `invalid_request_error` on the door.
|
|
10
10
|
*/
|
|
11
|
-
export function unwrapUpstreamError(detail) {
|
|
11
|
+
export function unwrapUpstreamError(detail, options = {}) {
|
|
12
12
|
try {
|
|
13
13
|
const parsed = JSON.parse(detail);
|
|
14
14
|
if (typeof parsed.error?.message === "string") {
|
|
15
15
|
return {
|
|
16
16
|
type: typeof parsed.error.type === "string" ? parsed.error.type : "api_error",
|
|
17
|
-
message: parsed.error.message
|
|
17
|
+
message: parsed.error.message,
|
|
18
|
+
...(options.preserveMetadata === true &&
|
|
19
|
+
typeof parsed.error.code === "string" &&
|
|
20
|
+
parsed.error.code.length > 0
|
|
21
|
+
? { code: parsed.error.code }
|
|
22
|
+
: {}),
|
|
23
|
+
...(options.preserveMetadata === true &&
|
|
24
|
+
typeof parsed.error.param === "string" &&
|
|
25
|
+
parsed.error.param.length > 0
|
|
26
|
+
? { param: parsed.error.param }
|
|
27
|
+
: {})
|
|
18
28
|
};
|
|
19
29
|
}
|
|
20
30
|
}
|
|
@@ -7,6 +7,7 @@ import { Effect } from "effect";
|
|
|
7
7
|
import { REASONING_SELECTION, reasoningSelectionOf, routeKitRequestValidationErrorOf, withoutRouteKitExtensions } from "../adapters/openai-chat-wire.js";
|
|
8
8
|
import { normalizeOpenAiResponsesCallIds } from "../adapters/openai-responses-wire.js";
|
|
9
9
|
import { joinPath, staticBackendModelPort } from "./backend.js";
|
|
10
|
+
import { openaiChatUsesMaxCompletionTokens } from "./openai-reasoning.js";
|
|
10
11
|
function invalidReasoningControlResponse(message, metadata = false, path) {
|
|
11
12
|
return Response.json({
|
|
12
13
|
error: {
|
|
@@ -17,6 +18,23 @@ function invalidReasoningControlResponse(message, metadata = false, path) {
|
|
|
17
18
|
}
|
|
18
19
|
}, { status: 400 });
|
|
19
20
|
}
|
|
21
|
+
function withChatCompletionTokenBudget(payload) {
|
|
22
|
+
if (payload === null || typeof payload !== "object" || Array.isArray(payload)) {
|
|
23
|
+
return payload;
|
|
24
|
+
}
|
|
25
|
+
const body = payload;
|
|
26
|
+
const model = typeof body.model === "string" ? body.model : "";
|
|
27
|
+
if (!openaiChatUsesMaxCompletionTokens(model))
|
|
28
|
+
return payload;
|
|
29
|
+
if (!Object.hasOwn(body, "max_tokens"))
|
|
30
|
+
return payload;
|
|
31
|
+
const next = { ...body };
|
|
32
|
+
if (!Object.hasOwn(next, "max_completion_tokens")) {
|
|
33
|
+
next.max_completion_tokens = next.max_tokens;
|
|
34
|
+
}
|
|
35
|
+
delete next.max_tokens;
|
|
36
|
+
return next;
|
|
37
|
+
}
|
|
20
38
|
/** An OpenAI HTTP backend supporting Chat Completions and native Responses. */
|
|
21
39
|
export class OpenAiBackend {
|
|
22
40
|
#baseUrl;
|
|
@@ -97,7 +115,7 @@ export class OpenAiBackend {
|
|
|
97
115
|
!Array.isArray(selectedPayload)
|
|
98
116
|
? this.#openRouterReasoning(selectedPayload, selection)
|
|
99
117
|
: selectedPayload;
|
|
100
|
-
const providerPayload = withoutRouteKitExtensions(payload);
|
|
118
|
+
const providerPayload = withChatCompletionTokenBudget(withoutRouteKitExtensions(payload));
|
|
101
119
|
return this.#request("/chat/completions", {
|
|
102
120
|
method: "POST",
|
|
103
121
|
headers: this.#headers(options),
|
|
@@ -5,3 +5,5 @@
|
|
|
5
5
|
*/
|
|
6
6
|
import type { ModelReasoningCapabilities } from "@velum-labs/routekit-contracts";
|
|
7
7
|
export declare function openaiReasoningCapabilities(modelId: string): ModelReasoningCapabilities | undefined;
|
|
8
|
+
/** GPT-5.x Chat Completions reject legacy `max_tokens`; use `max_completion_tokens`. */
|
|
9
|
+
export declare function openaiChatUsesMaxCompletionTokens(modelId: string): boolean;
|
|
@@ -23,3 +23,7 @@ export function openaiReasoningCapabilities(modelId) {
|
|
|
23
23
|
}
|
|
24
24
|
return undefined;
|
|
25
25
|
}
|
|
26
|
+
/** GPT-5.x Chat Completions reject legacy `max_tokens`; use `max_completion_tokens`. */
|
|
27
|
+
export function openaiChatUsesMaxCompletionTokens(modelId) {
|
|
28
|
+
return /(?:^|[./])gpt-5(?:[.:-]|$)/.test(modelId);
|
|
29
|
+
}
|
package/dist/routing/router.js
CHANGED
|
@@ -332,6 +332,7 @@ export class RoutingBackend {
|
|
|
332
332
|
error: {
|
|
333
333
|
type: "invalid_request_error",
|
|
334
334
|
code: "unsupported_reasoning_control",
|
|
335
|
+
param: requestedSelection.mode === "effort" ? "reasoning.effort" : "reasoning",
|
|
335
336
|
message: selection
|
|
336
337
|
}
|
|
337
338
|
}, { status: 400 }));
|
|
@@ -422,6 +423,7 @@ export class RoutingBackend {
|
|
|
422
423
|
error: {
|
|
423
424
|
type: "invalid_request_error",
|
|
424
425
|
code: "unsupported_reasoning_control",
|
|
426
|
+
param: requestedSelection.mode === "effort" ? "reasoning.effort" : "reasoning",
|
|
425
427
|
message: selection
|
|
426
428
|
}
|
|
427
429
|
}, { status: 400 }));
|
package/dist/test/chat.test.js
CHANGED
|
@@ -126,6 +126,64 @@ test("injects the default model and pipes the completion back", async () => {
|
|
|
126
126
|
await mock.close();
|
|
127
127
|
}
|
|
128
128
|
});
|
|
129
|
+
test("GPT-5 Chat Completions translate max_tokens to max_completion_tokens", async () => {
|
|
130
|
+
const mock = await startMock();
|
|
131
|
+
const backend = new OpenAiBackend({ baseUrl: `${mock.url}/v1` });
|
|
132
|
+
const gateway = await startGateway({ backend });
|
|
133
|
+
try {
|
|
134
|
+
for (const model of [
|
|
135
|
+
"gpt-5.6-sol",
|
|
136
|
+
"openai.gpt-5.6-luna",
|
|
137
|
+
"bedrock/openai.gpt-5.5",
|
|
138
|
+
"routekit/bedrock/us.openai.gpt-5.7"
|
|
139
|
+
]) {
|
|
140
|
+
const response = await fetch(`${gateway.url()}/v1/chat/completions`, {
|
|
141
|
+
method: "POST",
|
|
142
|
+
headers: { "content-type": "application/json" },
|
|
143
|
+
body: JSON.stringify({
|
|
144
|
+
model,
|
|
145
|
+
max_tokens: 128,
|
|
146
|
+
messages: [{ role: "user", content: "hi" }]
|
|
147
|
+
})
|
|
148
|
+
});
|
|
149
|
+
assert.equal(response.status, 200, model);
|
|
150
|
+
const body = mock.lastChatBody();
|
|
151
|
+
assert.equal(body?.max_completion_tokens, 128, model);
|
|
152
|
+
assert.equal("max_tokens" in (body ?? {}), false, model);
|
|
153
|
+
}
|
|
154
|
+
const modern = await fetch(`${gateway.url()}/v1/chat/completions`, {
|
|
155
|
+
method: "POST",
|
|
156
|
+
headers: { "content-type": "application/json" },
|
|
157
|
+
body: JSON.stringify({
|
|
158
|
+
model: "bedrock/openai.gpt-5.6-terra",
|
|
159
|
+
max_tokens: 128,
|
|
160
|
+
max_completion_tokens: 256,
|
|
161
|
+
messages: [{ role: "user", content: "hi" }]
|
|
162
|
+
})
|
|
163
|
+
});
|
|
164
|
+
assert.equal(modern.status, 200);
|
|
165
|
+
const modernBody = mock.lastChatBody();
|
|
166
|
+
assert.equal(modernBody?.max_completion_tokens, 256);
|
|
167
|
+
assert.equal("max_tokens" in (modernBody ?? {}), false);
|
|
168
|
+
const legacy = await fetch(`${gateway.url()}/v1/chat/completions`, {
|
|
169
|
+
method: "POST",
|
|
170
|
+
headers: { "content-type": "application/json" },
|
|
171
|
+
body: JSON.stringify({
|
|
172
|
+
model: "gpt-4o",
|
|
173
|
+
max_tokens: 64,
|
|
174
|
+
messages: [{ role: "user", content: "hi" }]
|
|
175
|
+
})
|
|
176
|
+
});
|
|
177
|
+
assert.equal(legacy.status, 200);
|
|
178
|
+
const legacyBody = mock.lastChatBody();
|
|
179
|
+
assert.equal(legacyBody?.max_tokens, 64);
|
|
180
|
+
assert.equal("max_completion_tokens" in (legacyBody ?? {}), false);
|
|
181
|
+
}
|
|
182
|
+
finally {
|
|
183
|
+
await Effect.runPromise(gateway.close);
|
|
184
|
+
await mock.close();
|
|
185
|
+
}
|
|
186
|
+
});
|
|
129
187
|
test("compound provider operations are not counted as retries", () => {
|
|
130
188
|
const attribution = collectAttribution({
|
|
131
189
|
effective_model: "codex/gpt-test",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import assert from "node:assert/strict";
|
|
2
2
|
import { test } from "node:test";
|
|
3
|
-
import { openaiReasoningCapabilities } from "../providers/openai-reasoning.js";
|
|
3
|
+
import { openaiChatUsesMaxCompletionTokens, openaiReasoningCapabilities } from "../providers/openai-reasoning.js";
|
|
4
4
|
test("OpenAI source authors verified GPT-5.5 and GPT-5.6 reasoning controls", () => {
|
|
5
5
|
assert.deepEqual(openaiReasoningCapabilities("gpt-5.6-sol"), {
|
|
6
6
|
status: "supported",
|
|
@@ -25,3 +25,23 @@ test("OpenAI source authors verified GPT-5.5 and GPT-5.6 reasoning controls", ()
|
|
|
25
25
|
assert.equal(openaiReasoningCapabilities("gpt-4o"), undefined);
|
|
26
26
|
assert.equal(openaiReasoningCapabilities("gpt-5.4"), undefined);
|
|
27
27
|
});
|
|
28
|
+
test("GPT-5 Chat Completions family uses max_completion_tokens", () => {
|
|
29
|
+
for (const model of [
|
|
30
|
+
"gpt-5",
|
|
31
|
+
"gpt-5.5",
|
|
32
|
+
"gpt-5.6",
|
|
33
|
+
"gpt-5.6-sol",
|
|
34
|
+
"gpt-5.6-terra",
|
|
35
|
+
"gpt-5.6-luna",
|
|
36
|
+
"gpt-5-mini",
|
|
37
|
+
"openai/gpt-5.5",
|
|
38
|
+
"openai.gpt-5.6-sol",
|
|
39
|
+
"bedrock/openai.gpt-5.6-sol",
|
|
40
|
+
"routekit/bedrock/us.openai.gpt-5.6-sol"
|
|
41
|
+
]) {
|
|
42
|
+
assert.equal(openaiChatUsesMaxCompletionTokens(model), true, model);
|
|
43
|
+
}
|
|
44
|
+
for (const model of ["gpt-4o", "gpt-4.1", "o3", "claude-sonnet-4-6", "not-gpt-5.6"]) {
|
|
45
|
+
assert.equal(openaiChatUsesMaxCompletionTokens(model), false, model);
|
|
46
|
+
}
|
|
47
|
+
});
|
|
@@ -320,7 +320,9 @@ test("Responses routes discovered Claude efforts to adaptive Anthropic egress",
|
|
|
320
320
|
})
|
|
321
321
|
});
|
|
322
322
|
assert.equal(unsupported.status, 400);
|
|
323
|
-
|
|
323
|
+
const error = (await unsupported.json());
|
|
324
|
+
assert.equal(error.error?.param, "reasoning.effort");
|
|
325
|
+
assert.equal(error.error?.message, 'reasoning effort "max" is not supported by model "claude-code/claude-fable-5"');
|
|
324
326
|
assert.equal(requests.length, 1);
|
|
325
327
|
}
|
|
326
328
|
finally {
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@velum-labs/routekit-gateway",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "1.0.
|
|
4
|
+
"version": "1.0.2",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
7
7
|
"url": "git+https://github.com/velum-labs/routekit.git",
|
|
@@ -45,12 +45,12 @@
|
|
|
45
45
|
"@aws-sdk/client-bedrock": "3.1095.0",
|
|
46
46
|
"@aws-sdk/client-bedrock-runtime": "3.1095.0",
|
|
47
47
|
"effect": "4.0.0-rc.108",
|
|
48
|
-
"@velum-labs/routekit-config-core": "1.0.
|
|
49
|
-
"@velum-labs/routekit-contracts": "1.0.
|
|
50
|
-
"@velum-labs/routekit-eval-contracts": "1.0.
|
|
51
|
-
"@velum-labs/routekit-eval-core": "1.0.
|
|
52
|
-
"@velum-labs/routekit-registry": "1.0.
|
|
53
|
-
"@velum-labs/routekit-runtime": "1.0.
|
|
48
|
+
"@velum-labs/routekit-config-core": "1.0.2",
|
|
49
|
+
"@velum-labs/routekit-contracts": "1.0.2",
|
|
50
|
+
"@velum-labs/routekit-eval-contracts": "1.0.2",
|
|
51
|
+
"@velum-labs/routekit-eval-core": "1.0.2",
|
|
52
|
+
"@velum-labs/routekit-registry": "1.0.2",
|
|
53
|
+
"@velum-labs/routekit-runtime": "1.0.2"
|
|
54
54
|
},
|
|
55
55
|
"keywords": [
|
|
56
56
|
"routekit",
|