@velum-labs/routekit-gateway 0.16.6 → 0.16.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/backend.d.ts +3 -0
- package/dist/backend.js +29 -11
- package/dist/index.d.ts +2 -2
- package/dist/provenance.js +47 -10
- package/dist/provider-backends.js +28 -8
- package/dist/server.d.ts +1 -1
- package/dist/server.js +9 -0
- package/dist/test/provenance.test.js +15 -0
- package/dist/test/provider-backends.test.js +25 -0
- package/package.json +5 -5
package/dist/backend.d.ts
CHANGED
|
@@ -72,7 +72,10 @@ export type Backend = {
|
|
|
72
72
|
/** Release any owned resources (e.g. a managed model process). Optional. */
|
|
73
73
|
close?(): Promise<void> | void;
|
|
74
74
|
};
|
|
75
|
+
export type BackendResponseMode = "buffered" | "streaming";
|
|
75
76
|
export type BackendRequestOptions = {
|
|
77
|
+
/** Original downstream response mode, before any provider forces upstream SSE. */
|
|
78
|
+
responseMode?: BackendResponseMode;
|
|
76
79
|
modelCallId?: string;
|
|
77
80
|
reasoningCapabilities?: ModelReasoningCapabilities;
|
|
78
81
|
/** Request-local, sanitized attribution updates from routing/backends. */
|
package/dist/backend.js
CHANGED
|
@@ -7,7 +7,14 @@ export function joinPath(baseUrl, path) {
|
|
|
7
7
|
return `${base}${suffix}`;
|
|
8
8
|
}
|
|
9
9
|
function invalidReasoningControlResponse(message, metadata = false, path) {
|
|
10
|
-
return Response.json({
|
|
10
|
+
return Response.json({
|
|
11
|
+
error: {
|
|
12
|
+
type: "invalid_request_error",
|
|
13
|
+
code: metadata ? "invalid_reasoning_metadata" : "invalid_reasoning_control",
|
|
14
|
+
...(path !== undefined ? { param: path } : {}),
|
|
15
|
+
message
|
|
16
|
+
}
|
|
17
|
+
}, { status: 400 });
|
|
11
18
|
}
|
|
12
19
|
/** An OpenAI HTTP backend supporting Chat Completions and native Responses. */
|
|
13
20
|
export class OpenAiBackend {
|
|
@@ -32,7 +39,10 @@ export class OpenAiBackend {
|
|
|
32
39
|
};
|
|
33
40
|
}
|
|
34
41
|
chat(body, signal, options = {}) {
|
|
35
|
-
const routed = this.#forceModel !== undefined &&
|
|
42
|
+
const routed = this.#forceModel !== undefined &&
|
|
43
|
+
typeof body === "object" &&
|
|
44
|
+
body !== null &&
|
|
45
|
+
!Array.isArray(body)
|
|
36
46
|
? { ...body, model: this.#forceModel }
|
|
37
47
|
: body;
|
|
38
48
|
const validationError = routeKitRequestValidationErrorOf(routed);
|
|
@@ -49,18 +59,22 @@ export class OpenAiBackend {
|
|
|
49
59
|
}
|
|
50
60
|
}, { status: 400 }));
|
|
51
61
|
}
|
|
52
|
-
const canonicalSelection = routed !== null &&
|
|
62
|
+
const canonicalSelection = routed !== null &&
|
|
63
|
+
typeof routed === "object" &&
|
|
64
|
+
!Array.isArray(routed) &&
|
|
53
65
|
(routed[REASONING_SELECTION] !== undefined ||
|
|
54
66
|
routed.x_routekit?.selection !== undefined);
|
|
55
|
-
const selectedPayload = canonicalSelection && routed !== null &&
|
|
56
|
-
typeof routed === "object" && !Array.isArray(routed)
|
|
67
|
+
const selectedPayload = canonicalSelection && routed !== null && typeof routed === "object" && !Array.isArray(routed)
|
|
57
68
|
? {
|
|
58
69
|
...routed,
|
|
59
70
|
...(selection.mode === "effort" ? { reasoning_effort: selection.effort } : {})
|
|
60
71
|
}
|
|
61
72
|
: routed;
|
|
62
|
-
if (canonicalSelection &&
|
|
63
|
-
|
|
73
|
+
if (canonicalSelection &&
|
|
74
|
+
selection.mode !== "effort" &&
|
|
75
|
+
selectedPayload !== null &&
|
|
76
|
+
typeof selectedPayload === "object" &&
|
|
77
|
+
!Array.isArray(selectedPayload)) {
|
|
64
78
|
delete selectedPayload.reasoning_effort;
|
|
65
79
|
}
|
|
66
80
|
const payload = options.reasoningCapabilities?.wireShape === "openrouter" &&
|
|
@@ -148,7 +162,10 @@ export class ModelRoutedBackend {
|
|
|
148
162
|
return model !== undefined && this.#routedIds.has(model) ? this.#routed : this.#primary;
|
|
149
163
|
}
|
|
150
164
|
listModelIds() {
|
|
151
|
-
const ids = [
|
|
165
|
+
const ids = [
|
|
166
|
+
...(this.#primary.listModelIds?.() ??
|
|
167
|
+
(this.defaultModel !== undefined ? [this.defaultModel] : []))
|
|
168
|
+
];
|
|
152
169
|
for (const id of this.#routedIds) {
|
|
153
170
|
if (!ids.includes(id))
|
|
154
171
|
ids.push(id);
|
|
@@ -168,7 +185,9 @@ export class ModelRoutedBackend {
|
|
|
168
185
|
return backend.reasoningWireShape?.(delegatedModel);
|
|
169
186
|
}
|
|
170
187
|
chat(body, signal, options = {}) {
|
|
171
|
-
const model = typeof body === "object" &&
|
|
188
|
+
const model = typeof body === "object" &&
|
|
189
|
+
body !== null &&
|
|
190
|
+
typeof body.model === "string"
|
|
172
191
|
? body.model
|
|
173
192
|
: undefined;
|
|
174
193
|
return this.#backendFor(model).chat(body, signal, options);
|
|
@@ -176,8 +195,7 @@ export class ModelRoutedBackend {
|
|
|
176
195
|
supportsResponses(model) {
|
|
177
196
|
const backend = this.#backendFor(model);
|
|
178
197
|
const delegatedModel = backend.resolveModel?.(model) ?? backend.defaultModel ?? model;
|
|
179
|
-
return
|
|
180
|
-
(backend.supportsResponses?.(delegatedModel) ?? true));
|
|
198
|
+
return backend.responses !== undefined && (backend.supportsResponses?.(delegatedModel) ?? true);
|
|
181
199
|
}
|
|
182
200
|
responses(body, signal, options = {}) {
|
|
183
201
|
const model = typeof body === "object" &&
|
package/dist/index.d.ts
CHANGED
|
@@ -4,13 +4,13 @@ export type { Gateway, GatewayOptions, ProviderRelay, ProviderRelayDialect } fro
|
|
|
4
4
|
export { startSwitchingGatewayProxy } from "./switching-proxy.js";
|
|
5
5
|
export type { SwitchingGatewayProxy } from "./switching-proxy.js";
|
|
6
6
|
export { joinPath, ModelRoutedBackend, OpenAiBackend } from "./backend.js";
|
|
7
|
-
export type { Backend, BackendModelRoute, BackendRequestOptions, RequestAttributionUpdate, ModelRoutedBackendOptions, OpenAiBackendOptions } from "./backend.js";
|
|
7
|
+
export type { Backend, BackendModelRoute, BackendRequestOptions, BackendResponseMode, RequestAttributionUpdate, ModelRoutedBackendOptions, OpenAiBackendOptions } from "./backend.js";
|
|
8
8
|
export { AnthropicBackend, CodexResponsesBackend, GoogleGenAiBackend } from "./provider-backends.js";
|
|
9
9
|
export type { ProviderBackendOptions, ProviderTransport } from "./provider-backends.js";
|
|
10
10
|
export { BedrockProviderSource, fromBedrockConverseOutput, toBedrockConverseInput } from "./bedrock-source.js";
|
|
11
11
|
export type { BedrockControlClient, BedrockProviderSourceOptions, BedrockRuntime } from "./bedrock-source.js";
|
|
12
12
|
export { CatalogBackend, DEFAULT_LEADERBOARD_DURABLE_RETENTION_DAYS, DEFAULT_LEADERBOARD_LIVE_LIMIT, DEFAULT_LEADERBOARD_LIVE_TTL_HOURS, isSubscriptionProvider, leaderboardConfigSchema, NoModelAvailableError, modelPolicyAllowsModel, modelPolicyRuleMatches, normalizeRouterConfigAliases, parseRouterConfig, resolveLeaderboardConfig, routerConfigSchema, splitNamespacedModel, UnknownModelError } from "./router.js";
|
|
13
|
-
export type { CatalogBackendOptions, CatalogModelInfo, LeaderboardConfig, ModelPolicy, ProviderPolicy, RouterConfig
|
|
13
|
+
export type { CatalogBackendOptions, CatalogModelInfo, LeaderboardConfig, ModelPolicy, ProviderPolicy, RouterConfig } from "./router.js";
|
|
14
14
|
export { API_PROVIDER_IDS, ApiProviderSource, parseDiscoveredModels, parseReasoningCapabilities, PROVIDER_IDS, SUBSCRIPTION_PROVIDER_IDS } from "./provider-source.js";
|
|
15
15
|
export type { ApiProviderId, ApiProviderSourceOptions, DiscoveredModel, ProviderId, ProviderSource, ProviderSourceTransport, SubscriptionProviderId } from "./provider-source.js";
|
|
16
16
|
export { endpointHealthProbe, probeEndpointHealth, providerAuthHeaders } from "./endpoint-health.js";
|
package/dist/provenance.js
CHANGED
|
@@ -5,6 +5,7 @@ import { dirname, join } from "node:path";
|
|
|
5
5
|
import { fileURLToPath } from "node:url";
|
|
6
6
|
import { artifactHash, requestHash, responseHash } from "@velum-labs/routekit-contracts";
|
|
7
7
|
import { meterCall, parseUsage, parseUsageFromSse } from "./cost.js";
|
|
8
|
+
import { decodeBufferedSse } from "./sse/parse.js";
|
|
8
9
|
export const MODEL_CALL_ID_HEADER = "x-routekit-model-call-id";
|
|
9
10
|
export const UNKNOWN_GIT_SHA = "unknown";
|
|
10
11
|
const GIT_SHA_PATTERN = /^[a-f0-9]{40}$/;
|
|
@@ -87,23 +88,56 @@ function providerRequestId(body) {
|
|
|
87
88
|
const id = asRecord(parseJson(body))?.id;
|
|
88
89
|
return typeof id === "string" && id.length > 0 ? id : undefined;
|
|
89
90
|
}
|
|
91
|
+
function streamedProviderError(body) {
|
|
92
|
+
const text = responseText(body);
|
|
93
|
+
if (!text.includes("response.failed") && !text.includes("response.incomplete"))
|
|
94
|
+
return undefined;
|
|
95
|
+
for (const event of decodeBufferedSse(text)) {
|
|
96
|
+
let payload;
|
|
97
|
+
try {
|
|
98
|
+
payload = asRecord(JSON.parse(event.data));
|
|
99
|
+
}
|
|
100
|
+
catch {
|
|
101
|
+
continue;
|
|
102
|
+
}
|
|
103
|
+
const eventType = event.event ?? payload?.type;
|
|
104
|
+
if (eventType !== "response.failed" &&
|
|
105
|
+
eventType !== "response.incomplete" &&
|
|
106
|
+
eventType !== "error") {
|
|
107
|
+
continue;
|
|
108
|
+
}
|
|
109
|
+
const response = asRecord(payload?.response);
|
|
110
|
+
return (asRecord(payload?.error) ??
|
|
111
|
+
asRecord(response?.error) ??
|
|
112
|
+
asRecord(response?.incomplete_details));
|
|
113
|
+
}
|
|
114
|
+
return undefined;
|
|
115
|
+
}
|
|
90
116
|
function providerError(result) {
|
|
91
|
-
|
|
117
|
+
const streamError = streamedProviderError(result.responseBody);
|
|
118
|
+
if (result.error === undefined &&
|
|
119
|
+
result.statusCode >= 200 &&
|
|
120
|
+
result.statusCode < 400 &&
|
|
121
|
+
streamError === undefined) {
|
|
92
122
|
return undefined;
|
|
93
123
|
}
|
|
94
|
-
const responseError = asRecord(asRecord(parseJson(result.responseBody))?.error);
|
|
124
|
+
const responseError = asRecord(asRecord(parseJson(result.responseBody))?.error) ?? streamError;
|
|
95
125
|
const noModelAvailable = result.statusCode === 503 &&
|
|
96
126
|
responseError?.type === "unavailable" &&
|
|
97
127
|
responseError.message === "no model is available; configure a provider";
|
|
128
|
+
const streamErrorIdentity = `${responseError?.type ?? ""} ${responseError?.error_type ?? ""} ${responseError?.code ?? ""}`.toLowerCase();
|
|
129
|
+
const streamRateLimited = /usage[_ ]?limit|rate[_ ]?limit|quota|insufficient_quota/.test(streamErrorIdentity);
|
|
98
130
|
const kind = noModelAvailable
|
|
99
131
|
? "capability_missing"
|
|
100
|
-
:
|
|
101
|
-
? "
|
|
102
|
-
: result.statusCode ===
|
|
103
|
-
? "
|
|
104
|
-
: result.statusCode ===
|
|
105
|
-
? "
|
|
106
|
-
:
|
|
132
|
+
: streamRateLimited
|
|
133
|
+
? "rate_limited"
|
|
134
|
+
: result.statusCode === 408
|
|
135
|
+
? "timeout"
|
|
136
|
+
: result.statusCode === 429
|
|
137
|
+
? "rate_limited"
|
|
138
|
+
: result.statusCode === 400 || result.statusCode === 422
|
|
139
|
+
? "validation_error"
|
|
140
|
+
: "provider_error";
|
|
107
141
|
const message = kind === "capability_missing"
|
|
108
142
|
? "no model route is configured"
|
|
109
143
|
: kind === "timeout"
|
|
@@ -117,7 +151,10 @@ function providerError(result) {
|
|
|
117
151
|
kind,
|
|
118
152
|
message,
|
|
119
153
|
retryable: !noModelAvailable &&
|
|
120
|
-
(
|
|
154
|
+
(streamRateLimited ||
|
|
155
|
+
result.statusCode === 408 ||
|
|
156
|
+
result.statusCode === 429 ||
|
|
157
|
+
result.statusCode >= 500)
|
|
121
158
|
};
|
|
122
159
|
}
|
|
123
160
|
export function buildModelCallRecord(context, result) {
|
|
@@ -1120,6 +1120,27 @@ function codexCompletionResponse(model, payload) {
|
|
|
1120
1120
|
? jsonResponse(chatCompletion(model, message, normalizedOpenAiUsage(payload.usage)))
|
|
1121
1121
|
: jsonResponse({ error: CODEX_EMPTY_RESPONSE_ERROR }, 502);
|
|
1122
1122
|
}
|
|
1123
|
+
function codexTerminalError(payload) {
|
|
1124
|
+
const record = typeof payload === "object" && payload !== null
|
|
1125
|
+
? payload
|
|
1126
|
+
: {};
|
|
1127
|
+
const response = typeof record.response === "object" && record.response !== null
|
|
1128
|
+
? record.response
|
|
1129
|
+
: undefined;
|
|
1130
|
+
const raw = typeof record.error === "object" && record.error !== null
|
|
1131
|
+
? record.error
|
|
1132
|
+
: typeof response?.error === "object" && response.error !== null
|
|
1133
|
+
? response.error
|
|
1134
|
+
: undefined;
|
|
1135
|
+
const details = typeof response?.incomplete_details === "object" && response.incomplete_details !== null
|
|
1136
|
+
? response.incomplete_details
|
|
1137
|
+
: undefined;
|
|
1138
|
+
return {
|
|
1139
|
+
...(raw ?? {}),
|
|
1140
|
+
type: raw?.type ?? raw?.error_type ?? details?.reason ?? "upstream_error",
|
|
1141
|
+
message: typeof raw?.message === "string" ? raw.message : "Codex response did not complete"
|
|
1142
|
+
};
|
|
1143
|
+
}
|
|
1123
1144
|
export class CodexResponsesBackend extends HttpProviderBackend {
|
|
1124
1145
|
#accountId;
|
|
1125
1146
|
#forceStream;
|
|
@@ -1326,14 +1347,7 @@ export class CodexResponsesBackend extends HttpProviderBackend {
|
|
|
1326
1347
|
if (eventType === "response.failed" ||
|
|
1327
1348
|
eventType === "response.incomplete" ||
|
|
1328
1349
|
eventType === "error") {
|
|
1329
|
-
return [
|
|
1330
|
-
{
|
|
1331
|
-
error: {
|
|
1332
|
-
message: "Codex response did not complete",
|
|
1333
|
-
type: "upstream_error"
|
|
1334
|
-
}
|
|
1335
|
-
}
|
|
1336
|
-
];
|
|
1350
|
+
return [{ error: codexTerminalError(item) }];
|
|
1337
1351
|
}
|
|
1338
1352
|
return [];
|
|
1339
1353
|
});
|
|
@@ -1346,6 +1360,7 @@ export class CodexResponsesBackend extends HttpProviderBackend {
|
|
|
1346
1360
|
];
|
|
1347
1361
|
const completedOutput = new Map();
|
|
1348
1362
|
let completedResponse;
|
|
1363
|
+
let terminalFailure;
|
|
1349
1364
|
for (const event of events) {
|
|
1350
1365
|
let payload;
|
|
1351
1366
|
try {
|
|
@@ -1375,6 +1390,9 @@ export class CodexResponsesBackend extends HttpProviderBackend {
|
|
|
1375
1390
|
record.response !== null) {
|
|
1376
1391
|
completedResponse = record.response;
|
|
1377
1392
|
}
|
|
1393
|
+
if (eventType === "response.failed" || eventType === "response.incomplete" || eventType === "error") {
|
|
1394
|
+
terminalFailure = codexTerminalError(record);
|
|
1395
|
+
}
|
|
1378
1396
|
}
|
|
1379
1397
|
if (completedResponse !== undefined) {
|
|
1380
1398
|
const terminalOutput = Array.isArray(completedResponse.output)
|
|
@@ -1388,6 +1406,8 @@ export class CodexResponsesBackend extends HttpProviderBackend {
|
|
|
1388
1406
|
const payload = { ...completedResponse, output: terminalOutput };
|
|
1389
1407
|
return codexCompletionResponse(model, payload);
|
|
1390
1408
|
}
|
|
1409
|
+
if (terminalFailure !== undefined)
|
|
1410
|
+
return jsonResponse({ error: terminalFailure }, 502);
|
|
1391
1411
|
throw new SseParseError("provider SSE stream ended without response.completed");
|
|
1392
1412
|
}
|
|
1393
1413
|
const payload = (await response.json());
|
package/dist/server.d.ts
CHANGED
|
@@ -31,7 +31,7 @@ export type ProviderRelayDialect = "anthropic" | "codex";
|
|
|
31
31
|
export type ProviderRelay = {
|
|
32
32
|
readonly dialect: ProviderRelayDialect;
|
|
33
33
|
shouldRelay(headers: IncomingMessage["headers"], model: string | undefined, servesLocally: (model: string) => boolean): boolean;
|
|
34
|
-
relay(headers: IncomingMessage["headers"], body: AnthropicRequest | ResponsesRequest, signal?: AbortSignal, options?: Pick<BackendRequestOptions, "onAttribution">): Promise<Response>;
|
|
34
|
+
relay(headers: IncomingMessage["headers"], body: AnthropicRequest | ResponsesRequest, signal?: AbortSignal, options?: Pick<BackendRequestOptions, "onAttribution" | "responseMode">): Promise<Response>;
|
|
35
35
|
models?(headers: IncomingMessage["headers"], search: string, signal?: AbortSignal): Promise<Response>;
|
|
36
36
|
countTokens?(headers: IncomingMessage["headers"], body: AnthropicRequest, signal?: AbortSignal): Promise<Response>;
|
|
37
37
|
mergedCatalog?(headers: IncomingMessage["headers"], search: string): Promise<{
|
package/dist/server.js
CHANGED
|
@@ -382,6 +382,7 @@ export async function startGateway(options) {
|
|
|
382
382
|
invoke: (callId, signal, onAttribution) => backend.chat(body, signal, {
|
|
383
383
|
modelCallId: callId,
|
|
384
384
|
requestContext,
|
|
385
|
+
responseMode: isStream(body) ? "streaming" : "buffered",
|
|
385
386
|
onAttribution
|
|
386
387
|
})
|
|
387
388
|
});
|
|
@@ -438,6 +439,7 @@ export async function startGateway(options) {
|
|
|
438
439
|
invoke: (callId, signal, onAttribution) => backend.chat(body, signal, {
|
|
439
440
|
modelCallId: callId,
|
|
440
441
|
requestContext,
|
|
442
|
+
responseMode: isStream(body) ? "streaming" : "buffered",
|
|
441
443
|
onAttribution
|
|
442
444
|
})
|
|
443
445
|
});
|
|
@@ -456,6 +458,7 @@ export async function startGateway(options) {
|
|
|
456
458
|
invoke: (callId, signal, onAttribution) => backend.embeddings(body, signal, {
|
|
457
459
|
modelCallId: callId,
|
|
458
460
|
requestContext,
|
|
461
|
+
responseMode: isStream(body) ? "streaming" : "buffered",
|
|
459
462
|
onAttribution
|
|
460
463
|
})
|
|
461
464
|
});
|
|
@@ -549,6 +552,7 @@ export async function startGateway(options) {
|
|
|
549
552
|
billing_mode: "subscription"
|
|
550
553
|
},
|
|
551
554
|
invoke: (_callId, signal, onAttribution) => anthropicRelay.relay(req.headers, relayBody, signal, {
|
|
555
|
+
responseMode: isStream(body) ? "streaming" : "buffered",
|
|
552
556
|
onAttribution
|
|
553
557
|
})
|
|
554
558
|
});
|
|
@@ -570,6 +574,7 @@ export async function startGateway(options) {
|
|
|
570
574
|
billing_mode: "subscription"
|
|
571
575
|
},
|
|
572
576
|
invoke: (_callId, signal, onAttribution) => anthropicRelay.relay(req.headers, body, signal, {
|
|
577
|
+
responseMode: isStream(body) ? "streaming" : "buffered",
|
|
573
578
|
onAttribution
|
|
574
579
|
})
|
|
575
580
|
});
|
|
@@ -582,6 +587,7 @@ export async function startGateway(options) {
|
|
|
582
587
|
attribution: initialAttribution(backend, resolvedModel, "claude-code"),
|
|
583
588
|
invoke: (callId, signal, onAttribution) => handleAnthropicMessages(backend, body, callId, signal, {
|
|
584
589
|
requestContext,
|
|
590
|
+
responseMode: isStream(body) ? "streaming" : "buffered",
|
|
585
591
|
onAttribution
|
|
586
592
|
})
|
|
587
593
|
});
|
|
@@ -637,6 +643,7 @@ export async function startGateway(options) {
|
|
|
637
643
|
billing_mode: "subscription"
|
|
638
644
|
},
|
|
639
645
|
invoke: (_callId, signal, onAttribution) => codexProviderRelay.relay(req.headers, relayBody, signal, {
|
|
646
|
+
responseMode: isStream(body) ? "streaming" : "buffered",
|
|
640
647
|
onAttribution
|
|
641
648
|
})
|
|
642
649
|
});
|
|
@@ -660,6 +667,7 @@ export async function startGateway(options) {
|
|
|
660
667
|
billing_mode: "client_auth"
|
|
661
668
|
},
|
|
662
669
|
invoke: (_callId, signal, onAttribution) => codexRequestRelay.relay(req.headers, body, signal, {
|
|
670
|
+
responseMode: isStream(body) ? "streaming" : "buffered",
|
|
663
671
|
onAttribution
|
|
664
672
|
})
|
|
665
673
|
});
|
|
@@ -672,6 +680,7 @@ export async function startGateway(options) {
|
|
|
672
680
|
attribution: initialAttribution(backend, requestedModel, codexProviderRelay !== undefined ? "codex" : undefined),
|
|
673
681
|
invoke: (callId, signal, onAttribution) => handleResponses(backend, canonicalBody, callId, signal, {
|
|
674
682
|
requestContext,
|
|
683
|
+
responseMode: isStream(body) ? "streaming" : "buffered",
|
|
675
684
|
onAttribution
|
|
676
685
|
})
|
|
677
686
|
});
|
|
@@ -174,3 +174,18 @@ test("model-call provenance meters aggregate buffered and Responses SSE usage",
|
|
|
174
174
|
assert.deepEqual(streamed.usage, buffered.usage);
|
|
175
175
|
assert.equal(streamed.metadata?.cost_estimate_usd, 0.0001575);
|
|
176
176
|
});
|
|
177
|
+
test("HTTP 200 Codex terminal SSE quota failure is rate-limited provenance", () => {
|
|
178
|
+
const body = Buffer.from('event: response.failed\ndata: {"response":{"error":{"type":"usage_limit_reached","code":"weekly","message":"spent"}}}\n\n');
|
|
179
|
+
const record = buildModelCallRecord({
|
|
180
|
+
callId: "call_sse_quota",
|
|
181
|
+
dialect: "openai-responses",
|
|
182
|
+
requestedModel: "codex/gpt-5.5",
|
|
183
|
+
model: "codex/gpt-5.5",
|
|
184
|
+
stream: true,
|
|
185
|
+
requestBody: { model: "codex/gpt-5.5", input: "hi", stream: true },
|
|
186
|
+
startedAt: "2026-07-29T00:00:00.000Z"
|
|
187
|
+
}, { statusCode: 200, durationMs: 5, responseBody: body });
|
|
188
|
+
assert.equal(record.status, "failed");
|
|
189
|
+
assert.equal(record.error?.kind, "rate_limited");
|
|
190
|
+
assert.equal(record.error?.retryable, true);
|
|
191
|
+
});
|
|
@@ -1848,3 +1848,28 @@ test("ordinary OpenAI Chat egress strips RouteKit provider-only envelopes", asyn
|
|
|
1848
1848
|
globalThis.fetch = original;
|
|
1849
1849
|
}
|
|
1850
1850
|
});
|
|
1851
|
+
test("Codex backend preserves structured forced-stream terminal failure", async () => {
|
|
1852
|
+
const backend = new CodexResponsesBackend({
|
|
1853
|
+
baseUrl: "https://codex.test", apiKey: "x", defaultModel: "m", forceStream: true,
|
|
1854
|
+
transport: async () => sse([{ event: "response.failed", data: { response: { error: {
|
|
1855
|
+
type: "usage_limit_reached", code: "weekly", message: "spent", resets_at: 1775000000
|
|
1856
|
+
} } } }])
|
|
1857
|
+
});
|
|
1858
|
+
const response = await backend.chat({ model: "m", messages: [] });
|
|
1859
|
+
assert.equal(response.status, 502);
|
|
1860
|
+
assert.deepEqual(await response.json(), { error: {
|
|
1861
|
+
type: "usage_limit_reached", code: "weekly", message: "spent", resets_at: 1775000000
|
|
1862
|
+
} });
|
|
1863
|
+
});
|
|
1864
|
+
test("Codex streaming backend preserves terminal provider error fields", async () => {
|
|
1865
|
+
const backend = new CodexResponsesBackend({
|
|
1866
|
+
baseUrl: "https://codex.test", apiKey: "x", defaultModel: "m",
|
|
1867
|
+
transport: async () => sse([{ event: "response.failed", data: { response: { error: {
|
|
1868
|
+
type: "usage_limit_reached", code: "weekly", message: "spent", resets_at: 1775000000
|
|
1869
|
+
} } } }])
|
|
1870
|
+
});
|
|
1871
|
+
const text = await (await backend.chat({ model: "m", messages: [], stream: true })).text();
|
|
1872
|
+
assert.match(text, /usage_limit_reached/);
|
|
1873
|
+
assert.match(text, /weekly/);
|
|
1874
|
+
assert.match(text, /1775000000/);
|
|
1875
|
+
});
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@velum-labs/routekit-gateway",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "0.16.
|
|
4
|
+
"version": "0.16.7",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
7
7
|
"url": "git+https://github.com/velum-labs/routekit.git",
|
|
@@ -29,10 +29,10 @@
|
|
|
29
29
|
"@aws-sdk/client-bedrock": "3.1095.0",
|
|
30
30
|
"@aws-sdk/client-bedrock-runtime": "3.1095.0",
|
|
31
31
|
"zod": "4.4.3",
|
|
32
|
-
"@velum-labs/routekit-contracts": "0.16.
|
|
33
|
-
"@velum-labs/routekit-registry": "0.16.
|
|
34
|
-
"@velum-labs/routekit-runtime": "0.16.
|
|
35
|
-
"@velum-labs/routekit-tracing": "0.16.
|
|
32
|
+
"@velum-labs/routekit-contracts": "0.16.7",
|
|
33
|
+
"@velum-labs/routekit-registry": "0.16.7",
|
|
34
|
+
"@velum-labs/routekit-runtime": "0.16.7",
|
|
35
|
+
"@velum-labs/routekit-tracing": "0.16.7"
|
|
36
36
|
},
|
|
37
37
|
"keywords": [
|
|
38
38
|
"routekit",
|