@velum-labs/routekit-gateway 0.16.6 → 0.16.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/backend.d.ts CHANGED
@@ -72,7 +72,10 @@ export type Backend = {
72
72
  /** Release any owned resources (e.g. a managed model process). Optional. */
73
73
  close?(): Promise<void> | void;
74
74
  };
75
+ export type BackendResponseMode = "buffered" | "streaming";
75
76
  export type BackendRequestOptions = {
77
+ /** Original downstream response mode, before any provider forces upstream SSE. */
78
+ responseMode?: BackendResponseMode;
76
79
  modelCallId?: string;
77
80
  reasoningCapabilities?: ModelReasoningCapabilities;
78
81
  /** Request-local, sanitized attribution updates from routing/backends. */
package/dist/backend.js CHANGED
@@ -7,7 +7,14 @@ export function joinPath(baseUrl, path) {
7
7
  return `${base}${suffix}`;
8
8
  }
9
9
  function invalidReasoningControlResponse(message, metadata = false, path) {
10
- return Response.json({ error: { type: "invalid_request_error", code: metadata ? "invalid_reasoning_metadata" : "invalid_reasoning_control", ...(path !== undefined ? { param: path } : {}), message } }, { status: 400 });
10
+ return Response.json({
11
+ error: {
12
+ type: "invalid_request_error",
13
+ code: metadata ? "invalid_reasoning_metadata" : "invalid_reasoning_control",
14
+ ...(path !== undefined ? { param: path } : {}),
15
+ message
16
+ }
17
+ }, { status: 400 });
11
18
  }
12
19
  /** An OpenAI HTTP backend supporting Chat Completions and native Responses. */
13
20
  export class OpenAiBackend {
@@ -32,7 +39,10 @@ export class OpenAiBackend {
32
39
  };
33
40
  }
34
41
  chat(body, signal, options = {}) {
35
- const routed = this.#forceModel !== undefined && typeof body === "object" && body !== null && !Array.isArray(body)
42
+ const routed = this.#forceModel !== undefined &&
43
+ typeof body === "object" &&
44
+ body !== null &&
45
+ !Array.isArray(body)
36
46
  ? { ...body, model: this.#forceModel }
37
47
  : body;
38
48
  const validationError = routeKitRequestValidationErrorOf(routed);
@@ -49,18 +59,22 @@ export class OpenAiBackend {
49
59
  }
50
60
  }, { status: 400 }));
51
61
  }
52
- const canonicalSelection = routed !== null && typeof routed === "object" && !Array.isArray(routed) &&
62
+ const canonicalSelection = routed !== null &&
63
+ typeof routed === "object" &&
64
+ !Array.isArray(routed) &&
53
65
  (routed[REASONING_SELECTION] !== undefined ||
54
66
  routed.x_routekit?.selection !== undefined);
55
- const selectedPayload = canonicalSelection && routed !== null &&
56
- typeof routed === "object" && !Array.isArray(routed)
67
+ const selectedPayload = canonicalSelection && routed !== null && typeof routed === "object" && !Array.isArray(routed)
57
68
  ? {
58
69
  ...routed,
59
70
  ...(selection.mode === "effort" ? { reasoning_effort: selection.effort } : {})
60
71
  }
61
72
  : routed;
62
- if (canonicalSelection && selection.mode !== "effort" && selectedPayload !== null &&
63
- typeof selectedPayload === "object" && !Array.isArray(selectedPayload)) {
73
+ if (canonicalSelection &&
74
+ selection.mode !== "effort" &&
75
+ selectedPayload !== null &&
76
+ typeof selectedPayload === "object" &&
77
+ !Array.isArray(selectedPayload)) {
64
78
  delete selectedPayload.reasoning_effort;
65
79
  }
66
80
  const payload = options.reasoningCapabilities?.wireShape === "openrouter" &&
@@ -148,7 +162,10 @@ export class ModelRoutedBackend {
148
162
  return model !== undefined && this.#routedIds.has(model) ? this.#routed : this.#primary;
149
163
  }
150
164
  listModelIds() {
151
- const ids = [...(this.#primary.listModelIds?.() ?? (this.defaultModel !== undefined ? [this.defaultModel] : []))];
165
+ const ids = [
166
+ ...(this.#primary.listModelIds?.() ??
167
+ (this.defaultModel !== undefined ? [this.defaultModel] : []))
168
+ ];
152
169
  for (const id of this.#routedIds) {
153
170
  if (!ids.includes(id))
154
171
  ids.push(id);
@@ -168,7 +185,9 @@ export class ModelRoutedBackend {
168
185
  return backend.reasoningWireShape?.(delegatedModel);
169
186
  }
170
187
  chat(body, signal, options = {}) {
171
- const model = typeof body === "object" && body !== null && typeof body.model === "string"
188
+ const model = typeof body === "object" &&
189
+ body !== null &&
190
+ typeof body.model === "string"
172
191
  ? body.model
173
192
  : undefined;
174
193
  return this.#backendFor(model).chat(body, signal, options);
@@ -176,8 +195,7 @@ export class ModelRoutedBackend {
176
195
  supportsResponses(model) {
177
196
  const backend = this.#backendFor(model);
178
197
  const delegatedModel = backend.resolveModel?.(model) ?? backend.defaultModel ?? model;
179
- return (backend.responses !== undefined &&
180
- (backend.supportsResponses?.(delegatedModel) ?? true));
198
+ return backend.responses !== undefined && (backend.supportsResponses?.(delegatedModel) ?? true);
181
199
  }
182
200
  responses(body, signal, options = {}) {
183
201
  const model = typeof body === "object" &&
package/dist/index.d.ts CHANGED
@@ -4,13 +4,13 @@ export type { Gateway, GatewayOptions, ProviderRelay, ProviderRelayDialect } fro
4
4
  export { startSwitchingGatewayProxy } from "./switching-proxy.js";
5
5
  export type { SwitchingGatewayProxy } from "./switching-proxy.js";
6
6
  export { joinPath, ModelRoutedBackend, OpenAiBackend } from "./backend.js";
7
- export type { Backend, BackendModelRoute, BackendRequestOptions, RequestAttributionUpdate, ModelRoutedBackendOptions, OpenAiBackendOptions } from "./backend.js";
7
+ export type { Backend, BackendModelRoute, BackendRequestOptions, BackendResponseMode, RequestAttributionUpdate, ModelRoutedBackendOptions, OpenAiBackendOptions } from "./backend.js";
8
8
  export { AnthropicBackend, CodexResponsesBackend, GoogleGenAiBackend } from "./provider-backends.js";
9
9
  export type { ProviderBackendOptions, ProviderTransport } from "./provider-backends.js";
10
10
  export { BedrockProviderSource, fromBedrockConverseOutput, toBedrockConverseInput } from "./bedrock-source.js";
11
11
  export type { BedrockControlClient, BedrockProviderSourceOptions, BedrockRuntime } from "./bedrock-source.js";
12
12
  export { CatalogBackend, DEFAULT_LEADERBOARD_DURABLE_RETENTION_DAYS, DEFAULT_LEADERBOARD_LIVE_LIMIT, DEFAULT_LEADERBOARD_LIVE_TTL_HOURS, isSubscriptionProvider, leaderboardConfigSchema, NoModelAvailableError, modelPolicyAllowsModel, modelPolicyRuleMatches, normalizeRouterConfigAliases, parseRouterConfig, resolveLeaderboardConfig, routerConfigSchema, splitNamespacedModel, UnknownModelError } from "./router.js";
13
- export type { CatalogBackendOptions, CatalogModelInfo, LeaderboardConfig, ModelPolicy, ProviderPolicy, RouterConfig, } from "./router.js";
13
+ export type { CatalogBackendOptions, CatalogModelInfo, LeaderboardConfig, ModelPolicy, ProviderPolicy, RouterConfig } from "./router.js";
14
14
  export { API_PROVIDER_IDS, ApiProviderSource, parseDiscoveredModels, parseReasoningCapabilities, PROVIDER_IDS, SUBSCRIPTION_PROVIDER_IDS } from "./provider-source.js";
15
15
  export type { ApiProviderId, ApiProviderSourceOptions, DiscoveredModel, ProviderId, ProviderSource, ProviderSourceTransport, SubscriptionProviderId } from "./provider-source.js";
16
16
  export { endpointHealthProbe, probeEndpointHealth, providerAuthHeaders } from "./endpoint-health.js";
@@ -5,6 +5,7 @@ import { dirname, join } from "node:path";
5
5
  import { fileURLToPath } from "node:url";
6
6
  import { artifactHash, requestHash, responseHash } from "@velum-labs/routekit-contracts";
7
7
  import { meterCall, parseUsage, parseUsageFromSse } from "./cost.js";
8
+ import { decodeBufferedSse } from "./sse/parse.js";
8
9
  export const MODEL_CALL_ID_HEADER = "x-routekit-model-call-id";
9
10
  export const UNKNOWN_GIT_SHA = "unknown";
10
11
  const GIT_SHA_PATTERN = /^[a-f0-9]{40}$/;
@@ -87,23 +88,56 @@ function providerRequestId(body) {
87
88
  const id = asRecord(parseJson(body))?.id;
88
89
  return typeof id === "string" && id.length > 0 ? id : undefined;
89
90
  }
91
+ function streamedProviderError(body) {
92
+ const text = responseText(body);
93
+ if (!text.includes("response.failed") && !text.includes("response.incomplete"))
94
+ return undefined;
95
+ for (const event of decodeBufferedSse(text)) {
96
+ let payload;
97
+ try {
98
+ payload = asRecord(JSON.parse(event.data));
99
+ }
100
+ catch {
101
+ continue;
102
+ }
103
+ const eventType = event.event ?? payload?.type;
104
+ if (eventType !== "response.failed" &&
105
+ eventType !== "response.incomplete" &&
106
+ eventType !== "error") {
107
+ continue;
108
+ }
109
+ const response = asRecord(payload?.response);
110
+ return (asRecord(payload?.error) ??
111
+ asRecord(response?.error) ??
112
+ asRecord(response?.incomplete_details));
113
+ }
114
+ return undefined;
115
+ }
90
116
  function providerError(result) {
91
- if (result.error === undefined && result.statusCode >= 200 && result.statusCode < 400) {
117
+ const streamError = streamedProviderError(result.responseBody);
118
+ if (result.error === undefined &&
119
+ result.statusCode >= 200 &&
120
+ result.statusCode < 400 &&
121
+ streamError === undefined) {
92
122
  return undefined;
93
123
  }
94
- const responseError = asRecord(asRecord(parseJson(result.responseBody))?.error);
124
+ const responseError = asRecord(asRecord(parseJson(result.responseBody))?.error) ?? streamError;
95
125
  const noModelAvailable = result.statusCode === 503 &&
96
126
  responseError?.type === "unavailable" &&
97
127
  responseError.message === "no model is available; configure a provider";
128
+ const streamErrorIdentity = `${responseError?.type ?? ""} ${responseError?.error_type ?? ""} ${responseError?.code ?? ""}`.toLowerCase();
129
+ const streamRateLimited = /usage[_ ]?limit|rate[_ ]?limit|quota|insufficient_quota/.test(streamErrorIdentity);
98
130
  const kind = noModelAvailable
99
131
  ? "capability_missing"
100
- : result.statusCode === 408
101
- ? "timeout"
102
- : result.statusCode === 429
103
- ? "rate_limited"
104
- : result.statusCode === 400 || result.statusCode === 422
105
- ? "validation_error"
106
- : "provider_error";
132
+ : streamRateLimited
133
+ ? "rate_limited"
134
+ : result.statusCode === 408
135
+ ? "timeout"
136
+ : result.statusCode === 429
137
+ ? "rate_limited"
138
+ : result.statusCode === 400 || result.statusCode === 422
139
+ ? "validation_error"
140
+ : "provider_error";
107
141
  const message = kind === "capability_missing"
108
142
  ? "no model route is configured"
109
143
  : kind === "timeout"
@@ -117,7 +151,10 @@ function providerError(result) {
117
151
  kind,
118
152
  message,
119
153
  retryable: !noModelAvailable &&
120
- (result.statusCode === 408 || result.statusCode === 429 || result.statusCode >= 500)
154
+ (streamRateLimited ||
155
+ result.statusCode === 408 ||
156
+ result.statusCode === 429 ||
157
+ result.statusCode >= 500)
121
158
  };
122
159
  }
123
160
  export function buildModelCallRecord(context, result) {
@@ -1120,6 +1120,27 @@ function codexCompletionResponse(model, payload) {
1120
1120
  ? jsonResponse(chatCompletion(model, message, normalizedOpenAiUsage(payload.usage)))
1121
1121
  : jsonResponse({ error: CODEX_EMPTY_RESPONSE_ERROR }, 502);
1122
1122
  }
1123
+ function codexTerminalError(payload) {
1124
+ const record = typeof payload === "object" && payload !== null
1125
+ ? payload
1126
+ : {};
1127
+ const response = typeof record.response === "object" && record.response !== null
1128
+ ? record.response
1129
+ : undefined;
1130
+ const raw = typeof record.error === "object" && record.error !== null
1131
+ ? record.error
1132
+ : typeof response?.error === "object" && response.error !== null
1133
+ ? response.error
1134
+ : undefined;
1135
+ const details = typeof response?.incomplete_details === "object" && response.incomplete_details !== null
1136
+ ? response.incomplete_details
1137
+ : undefined;
1138
+ return {
1139
+ ...(raw ?? {}),
1140
+ type: raw?.type ?? raw?.error_type ?? details?.reason ?? "upstream_error",
1141
+ message: typeof raw?.message === "string" ? raw.message : "Codex response did not complete"
1142
+ };
1143
+ }
1123
1144
  export class CodexResponsesBackend extends HttpProviderBackend {
1124
1145
  #accountId;
1125
1146
  #forceStream;
@@ -1326,14 +1347,7 @@ export class CodexResponsesBackend extends HttpProviderBackend {
1326
1347
  if (eventType === "response.failed" ||
1327
1348
  eventType === "response.incomplete" ||
1328
1349
  eventType === "error") {
1329
- return [
1330
- {
1331
- error: {
1332
- message: "Codex response did not complete",
1333
- type: "upstream_error"
1334
- }
1335
- }
1336
- ];
1350
+ return [{ error: codexTerminalError(item) }];
1337
1351
  }
1338
1352
  return [];
1339
1353
  });
@@ -1346,6 +1360,7 @@ export class CodexResponsesBackend extends HttpProviderBackend {
1346
1360
  ];
1347
1361
  const completedOutput = new Map();
1348
1362
  let completedResponse;
1363
+ let terminalFailure;
1349
1364
  for (const event of events) {
1350
1365
  let payload;
1351
1366
  try {
@@ -1375,6 +1390,9 @@ export class CodexResponsesBackend extends HttpProviderBackend {
1375
1390
  record.response !== null) {
1376
1391
  completedResponse = record.response;
1377
1392
  }
1393
+ if (eventType === "response.failed" || eventType === "response.incomplete" || eventType === "error") {
1394
+ terminalFailure = codexTerminalError(record);
1395
+ }
1378
1396
  }
1379
1397
  if (completedResponse !== undefined) {
1380
1398
  const terminalOutput = Array.isArray(completedResponse.output)
@@ -1388,6 +1406,8 @@ export class CodexResponsesBackend extends HttpProviderBackend {
1388
1406
  const payload = { ...completedResponse, output: terminalOutput };
1389
1407
  return codexCompletionResponse(model, payload);
1390
1408
  }
1409
+ if (terminalFailure !== undefined)
1410
+ return jsonResponse({ error: terminalFailure }, 502);
1391
1411
  throw new SseParseError("provider SSE stream ended without response.completed");
1392
1412
  }
1393
1413
  const payload = (await response.json());
package/dist/server.d.ts CHANGED
@@ -31,7 +31,7 @@ export type ProviderRelayDialect = "anthropic" | "codex";
31
31
  export type ProviderRelay = {
32
32
  readonly dialect: ProviderRelayDialect;
33
33
  shouldRelay(headers: IncomingMessage["headers"], model: string | undefined, servesLocally: (model: string) => boolean): boolean;
34
- relay(headers: IncomingMessage["headers"], body: AnthropicRequest | ResponsesRequest, signal?: AbortSignal, options?: Pick<BackendRequestOptions, "onAttribution">): Promise<Response>;
34
+ relay(headers: IncomingMessage["headers"], body: AnthropicRequest | ResponsesRequest, signal?: AbortSignal, options?: Pick<BackendRequestOptions, "onAttribution" | "responseMode">): Promise<Response>;
35
35
  models?(headers: IncomingMessage["headers"], search: string, signal?: AbortSignal): Promise<Response>;
36
36
  countTokens?(headers: IncomingMessage["headers"], body: AnthropicRequest, signal?: AbortSignal): Promise<Response>;
37
37
  mergedCatalog?(headers: IncomingMessage["headers"], search: string): Promise<{
package/dist/server.js CHANGED
@@ -382,6 +382,7 @@ export async function startGateway(options) {
382
382
  invoke: (callId, signal, onAttribution) => backend.chat(body, signal, {
383
383
  modelCallId: callId,
384
384
  requestContext,
385
+ responseMode: isStream(body) ? "streaming" : "buffered",
385
386
  onAttribution
386
387
  })
387
388
  });
@@ -438,6 +439,7 @@ export async function startGateway(options) {
438
439
  invoke: (callId, signal, onAttribution) => backend.chat(body, signal, {
439
440
  modelCallId: callId,
440
441
  requestContext,
442
+ responseMode: isStream(body) ? "streaming" : "buffered",
441
443
  onAttribution
442
444
  })
443
445
  });
@@ -456,6 +458,7 @@ export async function startGateway(options) {
456
458
  invoke: (callId, signal, onAttribution) => backend.embeddings(body, signal, {
457
459
  modelCallId: callId,
458
460
  requestContext,
461
+ responseMode: isStream(body) ? "streaming" : "buffered",
459
462
  onAttribution
460
463
  })
461
464
  });
@@ -549,6 +552,7 @@ export async function startGateway(options) {
549
552
  billing_mode: "subscription"
550
553
  },
551
554
  invoke: (_callId, signal, onAttribution) => anthropicRelay.relay(req.headers, relayBody, signal, {
555
+ responseMode: isStream(body) ? "streaming" : "buffered",
552
556
  onAttribution
553
557
  })
554
558
  });
@@ -570,6 +574,7 @@ export async function startGateway(options) {
570
574
  billing_mode: "subscription"
571
575
  },
572
576
  invoke: (_callId, signal, onAttribution) => anthropicRelay.relay(req.headers, body, signal, {
577
+ responseMode: isStream(body) ? "streaming" : "buffered",
573
578
  onAttribution
574
579
  })
575
580
  });
@@ -582,6 +587,7 @@ export async function startGateway(options) {
582
587
  attribution: initialAttribution(backend, resolvedModel, "claude-code"),
583
588
  invoke: (callId, signal, onAttribution) => handleAnthropicMessages(backend, body, callId, signal, {
584
589
  requestContext,
590
+ responseMode: isStream(body) ? "streaming" : "buffered",
585
591
  onAttribution
586
592
  })
587
593
  });
@@ -637,6 +643,7 @@ export async function startGateway(options) {
637
643
  billing_mode: "subscription"
638
644
  },
639
645
  invoke: (_callId, signal, onAttribution) => codexProviderRelay.relay(req.headers, relayBody, signal, {
646
+ responseMode: isStream(body) ? "streaming" : "buffered",
640
647
  onAttribution
641
648
  })
642
649
  });
@@ -660,6 +667,7 @@ export async function startGateway(options) {
660
667
  billing_mode: "client_auth"
661
668
  },
662
669
  invoke: (_callId, signal, onAttribution) => codexRequestRelay.relay(req.headers, body, signal, {
670
+ responseMode: isStream(body) ? "streaming" : "buffered",
663
671
  onAttribution
664
672
  })
665
673
  });
@@ -672,6 +680,7 @@ export async function startGateway(options) {
672
680
  attribution: initialAttribution(backend, requestedModel, codexProviderRelay !== undefined ? "codex" : undefined),
673
681
  invoke: (callId, signal, onAttribution) => handleResponses(backend, canonicalBody, callId, signal, {
674
682
  requestContext,
683
+ responseMode: isStream(body) ? "streaming" : "buffered",
675
684
  onAttribution
676
685
  })
677
686
  });
@@ -174,3 +174,18 @@ test("model-call provenance meters aggregate buffered and Responses SSE usage",
174
174
  assert.deepEqual(streamed.usage, buffered.usage);
175
175
  assert.equal(streamed.metadata?.cost_estimate_usd, 0.0001575);
176
176
  });
177
+ test("HTTP 200 Codex terminal SSE quota failure is rate-limited provenance", () => {
178
+ const body = Buffer.from('event: response.failed\ndata: {"response":{"error":{"type":"usage_limit_reached","code":"weekly","message":"spent"}}}\n\n');
179
+ const record = buildModelCallRecord({
180
+ callId: "call_sse_quota",
181
+ dialect: "openai-responses",
182
+ requestedModel: "codex/gpt-5.5",
183
+ model: "codex/gpt-5.5",
184
+ stream: true,
185
+ requestBody: { model: "codex/gpt-5.5", input: "hi", stream: true },
186
+ startedAt: "2026-07-29T00:00:00.000Z"
187
+ }, { statusCode: 200, durationMs: 5, responseBody: body });
188
+ assert.equal(record.status, "failed");
189
+ assert.equal(record.error?.kind, "rate_limited");
190
+ assert.equal(record.error?.retryable, true);
191
+ });
@@ -1848,3 +1848,28 @@ test("ordinary OpenAI Chat egress strips RouteKit provider-only envelopes", asyn
1848
1848
  globalThis.fetch = original;
1849
1849
  }
1850
1850
  });
1851
+ test("Codex backend preserves structured forced-stream terminal failure", async () => {
1852
+ const backend = new CodexResponsesBackend({
1853
+ baseUrl: "https://codex.test", apiKey: "x", defaultModel: "m", forceStream: true,
1854
+ transport: async () => sse([{ event: "response.failed", data: { response: { error: {
1855
+ type: "usage_limit_reached", code: "weekly", message: "spent", resets_at: 1775000000
1856
+ } } } }])
1857
+ });
1858
+ const response = await backend.chat({ model: "m", messages: [] });
1859
+ assert.equal(response.status, 502);
1860
+ assert.deepEqual(await response.json(), { error: {
1861
+ type: "usage_limit_reached", code: "weekly", message: "spent", resets_at: 1775000000
1862
+ } });
1863
+ });
1864
+ test("Codex streaming backend preserves terminal provider error fields", async () => {
1865
+ const backend = new CodexResponsesBackend({
1866
+ baseUrl: "https://codex.test", apiKey: "x", defaultModel: "m",
1867
+ transport: async () => sse([{ event: "response.failed", data: { response: { error: {
1868
+ type: "usage_limit_reached", code: "weekly", message: "spent", resets_at: 1775000000
1869
+ } } } }])
1870
+ });
1871
+ const text = await (await backend.chat({ model: "m", messages: [], stream: true })).text();
1872
+ assert.match(text, /usage_limit_reached/);
1873
+ assert.match(text, /weekly/);
1874
+ assert.match(text, /1775000000/);
1875
+ });
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@velum-labs/routekit-gateway",
3
3
  "private": false,
4
- "version": "0.16.6",
4
+ "version": "0.16.7",
5
5
  "repository": {
6
6
  "type": "git",
7
7
  "url": "git+https://github.com/velum-labs/routekit.git",
@@ -29,10 +29,10 @@
29
29
  "@aws-sdk/client-bedrock": "3.1095.0",
30
30
  "@aws-sdk/client-bedrock-runtime": "3.1095.0",
31
31
  "zod": "4.4.3",
32
- "@velum-labs/routekit-contracts": "0.16.6",
33
- "@velum-labs/routekit-registry": "0.16.6",
34
- "@velum-labs/routekit-runtime": "0.16.6",
35
- "@velum-labs/routekit-tracing": "0.16.6"
32
+ "@velum-labs/routekit-contracts": "0.16.7",
33
+ "@velum-labs/routekit-registry": "0.16.7",
34
+ "@velum-labs/routekit-runtime": "0.16.7",
35
+ "@velum-labs/routekit-tracing": "0.16.7"
36
36
  },
37
37
  "keywords": [
38
38
  "routekit",