@velum-labs/routekit-gateway 1.0.8 → 1.0.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -740,17 +740,26 @@ function buildOutput(message, toolRegistry) {
|
|
|
740
740
|
return output;
|
|
741
741
|
}
|
|
742
742
|
export function chatToResponses(openai, model, toolRegistry = EMPTY_TOOL_REGISTRY, searches = []) {
|
|
743
|
-
const
|
|
743
|
+
const choice = openai.choices?.[0];
|
|
744
|
+
const message = choice?.message;
|
|
745
|
+
const incompleteReason = choice?.finish_reason === "length" ||
|
|
746
|
+
choice?.finish_reason === "max_tokens" ||
|
|
747
|
+
choice?.finish_reason === "max_output_tokens"
|
|
748
|
+
? "max_output_tokens"
|
|
749
|
+
: choice?.finish_reason === "content_filter"
|
|
750
|
+
? "content_filter"
|
|
751
|
+
: undefined;
|
|
744
752
|
// Gateway-executed searches happened before the terminal step's output.
|
|
745
753
|
const output = [...searches.map(executedSearchItem), ...buildOutput(message, toolRegistry)];
|
|
746
754
|
return {
|
|
747
755
|
id: `resp_${openai.id ?? randomId()}`,
|
|
748
756
|
object: "response",
|
|
749
757
|
created_at: Math.floor(Date.now() / 1000),
|
|
750
|
-
status: "completed",
|
|
758
|
+
status: incompleteReason === undefined ? "completed" : "incomplete",
|
|
751
759
|
model,
|
|
752
760
|
output,
|
|
753
761
|
usage: chatUsageToResponses(openai.usage),
|
|
762
|
+
...(incompleteReason === undefined ? {} : { incomplete_details: { reason: incompleteReason } }),
|
|
754
763
|
...(openai.provider_cost !== undefined ? { provider_cost: openai.provider_cost } : {})
|
|
755
764
|
};
|
|
756
765
|
}
|
|
@@ -123,7 +123,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
123
123
|
let nextOutputIndex = 0;
|
|
124
124
|
let messageOutputIndex = -1;
|
|
125
125
|
let finished = false;
|
|
126
|
-
let
|
|
126
|
+
let finishReason;
|
|
127
127
|
let usage;
|
|
128
128
|
let providerCost;
|
|
129
129
|
let sequenceNumber = 0;
|
|
@@ -136,7 +136,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
136
136
|
// search, completed items collected for the terminal response payload.
|
|
137
137
|
const openSearches = new Map();
|
|
138
138
|
const completedSearchItems = [];
|
|
139
|
-
const baseResponse = (status, output) => ({
|
|
139
|
+
const baseResponse = (status, output, incompleteReason) => ({
|
|
140
140
|
id: responseId,
|
|
141
141
|
object: "response",
|
|
142
142
|
created_at: Math.floor(Date.now() / 1000),
|
|
@@ -144,6 +144,9 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
144
144
|
model,
|
|
145
145
|
output,
|
|
146
146
|
usage: status === "completed" ? chatUsageToResponses(usage) : null,
|
|
147
|
+
...(status === "incomplete" && incompleteReason !== undefined
|
|
148
|
+
? { incomplete_details: { reason: incompleteReason } }
|
|
149
|
+
: {}),
|
|
147
150
|
...(status === "completed" && providerCost !== undefined ? { provider_cost: providerCost } : {})
|
|
148
151
|
});
|
|
149
152
|
const ensureCreated = (controller) => {
|
|
@@ -341,7 +344,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
341
344
|
}
|
|
342
345
|
return indexed.sort((a, b) => a.outputIndex - b.outputIndex).map(({ item }) => item);
|
|
343
346
|
};
|
|
344
|
-
const finalize = (controller, terminal = "completed") => {
|
|
347
|
+
const finalize = (controller, terminal = "completed", incompleteReason) => {
|
|
345
348
|
if (finished)
|
|
346
349
|
return;
|
|
347
350
|
finished = true;
|
|
@@ -410,7 +413,22 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
410
413
|
// as incomplete.
|
|
411
414
|
controller.enqueue(terminal === "completed"
|
|
412
415
|
? emit("response.completed", { response: baseResponse("completed", assembleOutput()) })
|
|
413
|
-
: emit("response.incomplete", {
|
|
416
|
+
: emit("response.incomplete", {
|
|
417
|
+
response: baseResponse("incomplete", assembleOutput(), incompleteReason)
|
|
418
|
+
}));
|
|
419
|
+
};
|
|
420
|
+
const finalizeFromFinishReason = (controller) => {
|
|
421
|
+
if (finishReason === "length" ||
|
|
422
|
+
finishReason === "max_tokens" ||
|
|
423
|
+
finishReason === "max_output_tokens") {
|
|
424
|
+
finalize(controller, "incomplete", "max_output_tokens");
|
|
425
|
+
return;
|
|
426
|
+
}
|
|
427
|
+
if (finishReason === "content_filter") {
|
|
428
|
+
finalize(controller, "incomplete", "content_filter");
|
|
429
|
+
return;
|
|
430
|
+
}
|
|
431
|
+
finalize(controller, finishReason === undefined ? "incomplete" : "completed");
|
|
414
432
|
};
|
|
415
433
|
// A mid-stream provider failure (`data: {"error": {...}}` — e.g. the
|
|
416
434
|
// router's classified provider_error) becomes a `response.failed` event
|
|
@@ -590,9 +608,9 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
590
608
|
}
|
|
591
609
|
if (choice.finish_reason !== null && choice.finish_reason !== undefined) {
|
|
592
610
|
// OpenAI can emit token usage/provider cost in a later choices:[] chunk.
|
|
593
|
-
// Record
|
|
594
|
-
//
|
|
595
|
-
|
|
611
|
+
// Record how the turn ended, but wait for [DONE] or EOF before emitting
|
|
612
|
+
// the one terminal Responses event.
|
|
613
|
+
finishReason = choice.finish_reason;
|
|
596
614
|
}
|
|
597
615
|
};
|
|
598
616
|
const handleEvent = (controller, data) => {
|
|
@@ -601,7 +619,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
601
619
|
if (data === "[DONE]") {
|
|
602
620
|
// A `[DONE]` with no prior finish_reason is truncation, not a clean stop.
|
|
603
621
|
if (!finished)
|
|
604
|
-
|
|
622
|
+
finalizeFromFinishReason(controller);
|
|
605
623
|
return;
|
|
606
624
|
}
|
|
607
625
|
let chunk;
|
|
@@ -639,7 +657,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
639
657
|
},
|
|
640
658
|
onEnd(controller) {
|
|
641
659
|
if (!finished)
|
|
642
|
-
|
|
660
|
+
finalizeFromFinishReason(controller);
|
|
643
661
|
}
|
|
644
662
|
});
|
|
645
663
|
}
|
|
@@ -165,19 +165,55 @@ function streamedProviderError(body) {
|
|
|
165
165
|
const response = asRecord(payload?.response);
|
|
166
166
|
return (asRecord(payload?.error) ??
|
|
167
167
|
asRecord(response?.error) ??
|
|
168
|
-
asRecord(response?.incomplete_details)
|
|
168
|
+
asRecord(response?.incomplete_details) ?? {
|
|
169
|
+
type: eventType === "response.incomplete" ? "response_incomplete" : "response_failed"
|
|
170
|
+
});
|
|
169
171
|
}
|
|
170
172
|
return undefined;
|
|
171
173
|
}
|
|
172
|
-
function
|
|
173
|
-
const
|
|
174
|
+
function bufferedProviderError(body) {
|
|
175
|
+
const response = asRecord(parseJson(body));
|
|
176
|
+
if (response?.status === "incomplete") {
|
|
177
|
+
return asRecord(response.incomplete_details) ?? { type: "response_incomplete" };
|
|
178
|
+
}
|
|
179
|
+
if (response?.status === "failed") {
|
|
180
|
+
return asRecord(response.error) ?? { type: "response_failed" };
|
|
181
|
+
}
|
|
182
|
+
return undefined;
|
|
183
|
+
}
|
|
184
|
+
function requestedMaximumOutputTokens(body) {
|
|
185
|
+
const request = asRecord(body);
|
|
186
|
+
for (const field of ["max_output_tokens", "max_completion_tokens", "max_tokens"]) {
|
|
187
|
+
const value = request?.[field];
|
|
188
|
+
if (typeof value === "number" && Number.isInteger(value) && value > 0)
|
|
189
|
+
return value;
|
|
190
|
+
}
|
|
191
|
+
return undefined;
|
|
192
|
+
}
|
|
193
|
+
function terminalStopReason(context, result, usage) {
|
|
194
|
+
const terminalError = bufferedProviderError(result.responseBody) ?? streamedProviderError(result.responseBody);
|
|
195
|
+
const reason = terminalError?.reason ?? terminalError?.code;
|
|
196
|
+
if (typeof reason === "string" && reason.length > 0)
|
|
197
|
+
return reason;
|
|
198
|
+
const maximumOutputTokens = requestedMaximumOutputTokens(context.requestBody);
|
|
199
|
+
return maximumOutputTokens !== undefined &&
|
|
200
|
+
usage?.completion_tokens !== undefined &&
|
|
201
|
+
usage.completion_tokens >= maximumOutputTokens
|
|
202
|
+
? "max_output_tokens"
|
|
203
|
+
: undefined;
|
|
204
|
+
}
|
|
205
|
+
function providerError(result, stopReason) {
|
|
206
|
+
const terminalError = bufferedProviderError(result.responseBody) ?? streamedProviderError(result.responseBody);
|
|
174
207
|
if (result.error === undefined &&
|
|
175
208
|
result.statusCode >= 200 &&
|
|
176
209
|
result.statusCode < 400 &&
|
|
177
|
-
|
|
210
|
+
terminalError === undefined &&
|
|
211
|
+
stopReason === undefined) {
|
|
178
212
|
return undefined;
|
|
179
213
|
}
|
|
180
|
-
const responseError = asRecord(asRecord(parseJson(result.responseBody))?.error) ??
|
|
214
|
+
const responseError = asRecord(asRecord(parseJson(result.responseBody))?.error) ??
|
|
215
|
+
terminalError ??
|
|
216
|
+
(stopReason === undefined ? undefined : { reason: stopReason });
|
|
181
217
|
const noModelAvailable = result.statusCode === 503 &&
|
|
182
218
|
responseError?.type === "unavailable" &&
|
|
183
219
|
responseError.message === "no model is available; configure a provider";
|
|
@@ -196,13 +232,15 @@ function providerError(result) {
|
|
|
196
232
|
: "provider_error";
|
|
197
233
|
const message = kind === "capability_missing"
|
|
198
234
|
? "no model route is configured"
|
|
199
|
-
:
|
|
200
|
-
? "provider
|
|
201
|
-
: kind === "
|
|
202
|
-
? "provider
|
|
203
|
-
: kind === "
|
|
204
|
-
? "provider
|
|
205
|
-
:
|
|
235
|
+
: responseError?.reason === "max_output_tokens"
|
|
236
|
+
? "provider response reached the maximum output token limit"
|
|
237
|
+
: kind === "timeout"
|
|
238
|
+
? "provider request timed out"
|
|
239
|
+
: kind === "rate_limited"
|
|
240
|
+
? "provider rate limited the request"
|
|
241
|
+
: kind === "validation_error"
|
|
242
|
+
? "provider rejected the request"
|
|
243
|
+
: "provider request failed";
|
|
206
244
|
return {
|
|
207
245
|
kind,
|
|
208
246
|
message,
|
|
@@ -225,13 +263,15 @@ export function buildModelCallRecord(context, result) {
|
|
|
225
263
|
totalTokens: usage.total_tokens
|
|
226
264
|
}
|
|
227
265
|
});
|
|
228
|
-
const
|
|
266
|
+
const stopReason = terminalStopReason(context, result, usage);
|
|
267
|
+
const error = providerError(result, stopReason);
|
|
229
268
|
const metadata = {
|
|
230
269
|
dialect: context.dialect,
|
|
231
270
|
stream: context.stream,
|
|
232
271
|
http_status: result.statusCode,
|
|
233
272
|
duration_ms: result.durationMs,
|
|
234
273
|
requested_model: context.requestedModel ?? null,
|
|
274
|
+
...(stopReason === undefined ? {} : { stop_reason: stopReason }),
|
|
235
275
|
unknown_usage: callCost.unknownUsage,
|
|
236
276
|
unknown_cost: callCost.unknownCost,
|
|
237
277
|
...(context.attribution !== undefined
|
|
@@ -261,3 +261,65 @@ test("HTTP 200 Codex terminal SSE quota failure is rate-limited provenance", ()
|
|
|
261
261
|
assert.equal(record.error?.kind, "rate_limited");
|
|
262
262
|
assert.equal(record.error?.retryable, true);
|
|
263
263
|
});
|
|
264
|
+
test("HTTP 200 incomplete Responses output is failed provenance", () => {
|
|
265
|
+
const record = buildModelCallRecord({
|
|
266
|
+
callId: "call_response_incomplete",
|
|
267
|
+
dialect: "openai-responses",
|
|
268
|
+
requestedModel: "claude-code/claude-opus-5",
|
|
269
|
+
model: "claude-code/claude-opus-5",
|
|
270
|
+
stream: false,
|
|
271
|
+
requestBody: { input: "author twenty evaluation cases" },
|
|
272
|
+
startedAt: "2026-08-20T13:00:00.000Z"
|
|
273
|
+
}, {
|
|
274
|
+
statusCode: 200,
|
|
275
|
+
durationMs: 5,
|
|
276
|
+
responseBody: Buffer.from(JSON.stringify({
|
|
277
|
+
status: "incomplete",
|
|
278
|
+
incomplete_details: { reason: "max_output_tokens" },
|
|
279
|
+
usage: { input_tokens: 31_533, output_tokens: 16_384 }
|
|
280
|
+
}))
|
|
281
|
+
});
|
|
282
|
+
assert.equal(record.status, "failed");
|
|
283
|
+
assert.equal(record.error?.message, "provider response reached the maximum output token limit");
|
|
284
|
+
assert.equal(record.usage?.completion_tokens, 16_384);
|
|
285
|
+
assert.equal(record.metadata?.stop_reason, "max_output_tokens");
|
|
286
|
+
});
|
|
287
|
+
test("an exact requested output-token cap is failed provenance", () => {
|
|
288
|
+
const record = buildModelCallRecord({
|
|
289
|
+
callId: "call_response_exact_cap",
|
|
290
|
+
dialect: "openai-responses",
|
|
291
|
+
requestedModel: "claude-code/claude-opus-5",
|
|
292
|
+
model: "claude-code/claude-opus-5",
|
|
293
|
+
stream: false,
|
|
294
|
+
requestBody: {
|
|
295
|
+
input: "author twenty evaluation cases",
|
|
296
|
+
max_output_tokens: 32_768
|
|
297
|
+
},
|
|
298
|
+
startedAt: "2026-08-20T13:00:00.000Z"
|
|
299
|
+
}, {
|
|
300
|
+
statusCode: 200,
|
|
301
|
+
durationMs: 5,
|
|
302
|
+
responseBody: Buffer.from(JSON.stringify({
|
|
303
|
+
status: "completed",
|
|
304
|
+
usage: { input_tokens: 31_533, output_tokens: 32_768 }
|
|
305
|
+
}))
|
|
306
|
+
});
|
|
307
|
+
assert.equal(record.status, "failed");
|
|
308
|
+
assert.equal(record.metadata?.stop_reason, "max_output_tokens");
|
|
309
|
+
});
|
|
310
|
+
test("HTTP 200 Responses stream without a finish reason is failed provenance", () => {
|
|
311
|
+
const record = buildModelCallRecord({
|
|
312
|
+
callId: "call_response_stream_truncated",
|
|
313
|
+
dialect: "openai-responses",
|
|
314
|
+
requestedModel: "openai/model",
|
|
315
|
+
model: "openai/model",
|
|
316
|
+
stream: true,
|
|
317
|
+
requestBody: { input: "answer", stream: true },
|
|
318
|
+
startedAt: "2026-08-20T13:00:00.000Z"
|
|
319
|
+
}, {
|
|
320
|
+
statusCode: 200,
|
|
321
|
+
durationMs: 5,
|
|
322
|
+
responseBody: Buffer.from('event: response.incomplete\ndata: {"type":"response.incomplete","response":{"status":"incomplete"}}\n\n')
|
|
323
|
+
});
|
|
324
|
+
assert.equal(record.status, "failed");
|
|
325
|
+
});
|
|
@@ -4,7 +4,7 @@ import { runRouteKitEffect } from "@velum-labs/routekit-runtime/effect";
|
|
|
4
4
|
import { Effect } from "effect";
|
|
5
5
|
import { reasoningSelectionOf } from "../adapters/openai-chat-wire.js";
|
|
6
6
|
import { parseResponsesEncryptedContent, wrapResponsesEncryptedContent } from "../adapters/openai-responses-wire.js";
|
|
7
|
-
import { openAiSseToResponses } from "../adapters/responses.js";
|
|
7
|
+
import { chatToResponses, openAiSseToResponses } from "../adapters/responses.js";
|
|
8
8
|
import { RoutingBackend } from "../routing/router.js";
|
|
9
9
|
import { startGateway } from "../gateway-service.js";
|
|
10
10
|
import { testProviderSource } from "./provider-source-fixture.js";
|
|
@@ -26,6 +26,37 @@ test("a mid-stream provider error event becomes response.failed with the upstrea
|
|
|
26
26
|
assert.ok(text.includes("openrouter call failed (unknown)"));
|
|
27
27
|
assert.ok(!text.includes("event: response.completed"));
|
|
28
28
|
});
|
|
29
|
+
test("max-token Chat completion becomes an incomplete Responses result", () => {
|
|
30
|
+
for (const finishReason of ["length", "max_tokens"]) {
|
|
31
|
+
const response = chatToResponses({
|
|
32
|
+
choices: [
|
|
33
|
+
{
|
|
34
|
+
message: { role: "assistant", content: '{"cases":[{"id":"truncated' },
|
|
35
|
+
finish_reason: finishReason
|
|
36
|
+
}
|
|
37
|
+
]
|
|
38
|
+
}, "claude-code/claude-opus-5");
|
|
39
|
+
assert.equal(response.status, "incomplete");
|
|
40
|
+
assert.deepEqual(response.incomplete_details, { reason: "max_output_tokens" });
|
|
41
|
+
}
|
|
42
|
+
});
|
|
43
|
+
test("max-token Chat stream becomes response.incomplete with its reason", async () => {
|
|
44
|
+
const stream = openAiSseToResponses(sseStream(`data: ${JSON.stringify({
|
|
45
|
+
choices: [
|
|
46
|
+
{
|
|
47
|
+
index: 0,
|
|
48
|
+
delta: { content: '{"cases":[{"id":"truncated' },
|
|
49
|
+
finish_reason: null
|
|
50
|
+
}
|
|
51
|
+
]
|
|
52
|
+
})}\n\n`, `data: ${JSON.stringify({
|
|
53
|
+
choices: [{ index: 0, delta: {}, finish_reason: "length" }]
|
|
54
|
+
})}\n\n`, "data: [DONE]\n\n"), "claude-code/claude-opus-5");
|
|
55
|
+
const text = await new Response(stream).text();
|
|
56
|
+
assert.match(text, /event: response\.incomplete/u);
|
|
57
|
+
assert.match(text, /"incomplete_details":\{"reason":"max_output_tokens"\}/u);
|
|
58
|
+
assert.doesNotMatch(text, /event: response\.completed/u);
|
|
59
|
+
});
|
|
29
60
|
test("translates a streamed Responses event sequence", async () => {
|
|
30
61
|
const mock = await startMock();
|
|
31
62
|
const gateway = await startGateway({
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@velum-labs/routekit-gateway",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "1.0.
|
|
4
|
+
"version": "1.0.9",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
7
7
|
"url": "git+https://github.com/velum-labs/routekit.git",
|
|
@@ -45,12 +45,12 @@
|
|
|
45
45
|
"@aws-sdk/client-bedrock": "3.1095.0",
|
|
46
46
|
"@aws-sdk/client-bedrock-runtime": "3.1095.0",
|
|
47
47
|
"effect": "4.0.0-rc.108",
|
|
48
|
-
"@velum-labs/routekit-config-core": "1.0.
|
|
49
|
-
"@velum-labs/routekit-contracts": "1.0.
|
|
50
|
-
"@velum-labs/routekit-eval-contracts": "1.0.
|
|
51
|
-
"@velum-labs/routekit-eval-core": "1.0.
|
|
52
|
-
"@velum-labs/routekit-registry": "1.0.
|
|
53
|
-
"@velum-labs/routekit-runtime": "1.0.
|
|
48
|
+
"@velum-labs/routekit-config-core": "1.0.9",
|
|
49
|
+
"@velum-labs/routekit-contracts": "1.0.9",
|
|
50
|
+
"@velum-labs/routekit-eval-contracts": "1.0.9",
|
|
51
|
+
"@velum-labs/routekit-eval-core": "1.0.9",
|
|
52
|
+
"@velum-labs/routekit-registry": "1.0.9",
|
|
53
|
+
"@velum-labs/routekit-runtime": "1.0.9"
|
|
54
54
|
},
|
|
55
55
|
"keywords": [
|
|
56
56
|
"routekit",
|