@timo972/cc-router 0.9.0 → 0.10.0-rc.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +88 -2
- package/README.md +2 -0
- package/dist/config/manager.js +15 -4
- package/dist/protocol/openai-response-to-anthropic.js +16 -1
- package/dist/protocol/openai-responses-collect.js +190 -11
- package/dist/protocol/openai-stream-to-anthropic.js +20 -2
- package/dist/protocol/sse.js +10 -1
- package/dist/providers/openai/account-state.js +377 -0
- package/dist/providers/openai/codex-transport.js +8 -0
- package/dist/providers/openai/failure-routing.js +141 -0
- package/dist/providers/openai/token-pool.js +401 -0
- package/dist/providers/openai/token-refresher.js +144 -11
- package/dist/providers/openai/usage.js +160 -0
- package/dist/proxy/account-add.js +10 -7
- package/dist/proxy/account-deletion.js +5 -1
- package/dist/proxy/account-patch.js +64 -0
- package/dist/proxy/account-pool.js +19 -0
- package/dist/proxy/anthropic-routing.js +2 -2
- package/dist/proxy/lease-lifecycle.js +6 -4
- package/dist/proxy/messages-cross-route.js +243 -62
- package/dist/proxy/openai-ingress.js +379 -0
- package/dist/proxy/openai-routing.js +61 -0
- package/dist/proxy/provider-routing.js +8 -4
- package/dist/proxy/responses-server.js +109 -48
- package/dist/proxy/server.js +241 -65
- package/dist/proxy/stats.js +30 -1
- package/dist/proxy/token-pool.js +3 -19
- package/dist/ui/Dashboard.js +93 -17
- package/package.json +1 -1
- package/dist/providers/openai/account-pool.js +0 -11
|
@@ -0,0 +1,379 @@
|
|
|
1
|
+
import { applyCodexRateLimits } from "../providers/openai/account-state.js";
|
|
2
|
+
import { headersToRecord, parseCodexRateLimits } from "../providers/openai/usage.js";
|
|
3
|
+
import { applyCodexFailureRouting } from "../providers/openai/failure-routing.js";
|
|
4
|
+
import { needsOpenAIRefresh } from "../providers/openai/token-refresher.js";
|
|
5
|
+
import { stats, boundModelId, createLocalRoutingErrorLog } from "./stats.js";
|
|
6
|
+
import { logError } from "./logger.js";
|
|
7
|
+
import { EmptyPoolError, NoEligibleAccountError } from "./account-pool.js";
|
|
8
|
+
import { acquireRequestRoute, routeReasonDetails, routeFailureDetails } from "./lease-lifecycle.js";
|
|
9
|
+
/**
|
|
10
|
+
* Mirrors `anthropic-routing.ts`'s `requestTerminated` check. This ingress
|
|
11
|
+
* path never threads the raw `Request` through (only `Response`), so it
|
|
12
|
+
* checks just the response side: `res.destroyed`/`res.writableEnded` are
|
|
13
|
+
* enough to detect a client that disconnected while we were off awaiting a
|
|
14
|
+
* token refresh.
|
|
15
|
+
*/
|
|
16
|
+
function responseTerminated(res) {
|
|
17
|
+
return res.destroyed || res.writableEnded;
|
|
18
|
+
}
|
|
19
|
+
/**
|
|
20
|
+
* Cooldown applied when a *local* token refresh fails. Matches the upstream-401
|
|
21
|
+
* cooldown in `failure-routing.ts`: a refresh that cannot produce a usable token
|
|
22
|
+
* is an auth failure, and without a cooldown the pool would immediately hand the
|
|
23
|
+
* same account back to the next request.
|
|
24
|
+
*/
|
|
25
|
+
const REFRESH_FAILURE_COOLDOWN_MS = 30_000;
|
|
26
|
+
/**
|
|
27
|
+
* Response headers that must never be mirrored to the local client when
|
|
28
|
+
* relaying an upstream response verbatim:
|
|
29
|
+
* - hop-by-hop headers per RFC 7230 §6.1 (content-length, transfer-encoding,
|
|
30
|
+
* connection, keep-alive, te, trailer, upgrade, proxy-authenticate,
|
|
31
|
+
* proxy-authorization) are meaningless (or actively wrong/dangerous) once
|
|
32
|
+
* re-framed by our own HTTP server — e.g. `upgrade` would claim a protocol
|
|
33
|
+
* switch never negotiated with this client, and the two `proxy-*` headers
|
|
34
|
+
* are scoped to the upstream hop's own (unrelated) proxy auth.
|
|
35
|
+
* - content-encoding is dropped because undici's fetch() already
|
|
36
|
+
* transparently decompresses the body while leaving this header intact —
|
|
37
|
+
* forwarding it would tell the client to gunzip bytes that are no longer
|
|
38
|
+
* compressed. Deliberate drop, not RFC hop-by-hop.
|
|
39
|
+
* - set-cookie must never leak the upstream service's session cookies to a
|
|
40
|
+
* local client. Deliberate drop, not RFC hop-by-hop.
|
|
41
|
+
*/
|
|
42
|
+
export const EXCLUDED_UPSTREAM_RELAY_HEADERS = new Set([
|
|
43
|
+
"content-length",
|
|
44
|
+
"transfer-encoding",
|
|
45
|
+
"connection",
|
|
46
|
+
"keep-alive",
|
|
47
|
+
"te",
|
|
48
|
+
"trailer",
|
|
49
|
+
"upgrade",
|
|
50
|
+
"proxy-authenticate",
|
|
51
|
+
"proxy-authorization",
|
|
52
|
+
"content-encoding",
|
|
53
|
+
"set-cookie",
|
|
54
|
+
]);
|
|
55
|
+
/**
|
|
56
|
+
* The `Connection` header can nominate additional header names as hop-by-hop
|
|
57
|
+
* for this specific response (RFC 7230 §6.1), beyond the fixed set above —
|
|
58
|
+
* e.g. `Connection: close, X-Internal-Token` means `X-Internal-Token` is also
|
|
59
|
+
* hop-by-hop here and must not reach the client.
|
|
60
|
+
*/
|
|
61
|
+
function connectionNominatedHeaders(source) {
|
|
62
|
+
const nominated = new Set();
|
|
63
|
+
const connection = source.get("connection");
|
|
64
|
+
if (!connection)
|
|
65
|
+
return nominated;
|
|
66
|
+
for (const token of connection.split(",")) {
|
|
67
|
+
const name = token.trim().toLowerCase();
|
|
68
|
+
if (name)
|
|
69
|
+
nominated.add(name);
|
|
70
|
+
}
|
|
71
|
+
return nominated;
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Single place that decides which upstream response headers are safe to
|
|
75
|
+
* mirror to the local client, so every relay site shares the same policy
|
|
76
|
+
* instead of re-implementing the exclusion set (and the dynamic `Connection`
|
|
77
|
+
* nomination) inline. `apply` is called once per header that passes the
|
|
78
|
+
* filter, in `source`'s own iteration order — callers still own their own
|
|
79
|
+
* per-header special cases (e.g. content-type) by skipping in `apply`.
|
|
80
|
+
*/
|
|
81
|
+
export function mirrorUpstreamHeaders(source, apply) {
|
|
82
|
+
const nominated = connectionNominatedHeaders(source);
|
|
83
|
+
source.forEach((value, key) => {
|
|
84
|
+
const lower = key.toLowerCase();
|
|
85
|
+
if (EXCLUDED_UPSTREAM_RELAY_HEADERS.has(lower))
|
|
86
|
+
return;
|
|
87
|
+
if (nominated.has(lower))
|
|
88
|
+
return;
|
|
89
|
+
apply(key, value);
|
|
90
|
+
});
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* Shared OpenAI/Codex ingress lifecycle: acquire a sticky account lease,
|
|
94
|
+
* refresh its token if needed, forward the request, classify the upstream
|
|
95
|
+
* failure for cooldown/eligibility purposes, relay the response to the
|
|
96
|
+
* client, then record activity/stats keyed on what the client actually
|
|
97
|
+
* received. Every awaited step is guarded so a rejection here can only ever
|
|
98
|
+
* produce a local error response — it must never crash the daemon or leave
|
|
99
|
+
* an unhandled rejection behind.
|
|
100
|
+
*/
|
|
101
|
+
export async function runOpenAIIngress(opts) {
|
|
102
|
+
const { res, sessionKey, path, openAIRouter, openAIPool, prepareOpenAIAccount, forwardOpenAI, forwardBody, recordActivity, now, envelope, relay, onUpstreamAuthFailure, } = opts;
|
|
103
|
+
// The model comes from a client-controlled body and is retained in the
|
|
104
|
+
// activity ring buffer below. Bound it once, here, so every activity entry,
|
|
105
|
+
// routing context and bucket lookup on this path carries an identifier that
|
|
106
|
+
// cannot grow with the request. The body forwarded upstream is untouched —
|
|
107
|
+
// it still carries whatever model the caller asked for.
|
|
108
|
+
const requestedModel = boundModelId(opts.requestedModel);
|
|
109
|
+
// A client that hangs up must take the upstream request with it. Releasing
|
|
110
|
+
// the lease (which the response's own close listener does) only returns the
|
|
111
|
+
// account's *local* capacity — without this the Codex request keeps
|
|
112
|
+
// streaming to a socket nobody is reading, so the pool counts the account
|
|
113
|
+
// idle and routes more work onto an upstream slot that is still occupied.
|
|
114
|
+
//
|
|
115
|
+
// Registered before the lease is acquired so this listener runs before the
|
|
116
|
+
// lifecycle's release, and `once` so it cleans itself up. A normal end also
|
|
117
|
+
// emits `close`, hence the `writableEnded` guard: only a premature close is
|
|
118
|
+
// a disconnect.
|
|
119
|
+
const clientGone = new AbortController();
|
|
120
|
+
res.once("close", () => {
|
|
121
|
+
if (!res.writableEnded)
|
|
122
|
+
clientGone.abort();
|
|
123
|
+
});
|
|
124
|
+
let selected;
|
|
125
|
+
try {
|
|
126
|
+
selected = acquireRequestRoute(sessionKey, res, openAIRouter, { requestedModel });
|
|
127
|
+
}
|
|
128
|
+
catch (error) {
|
|
129
|
+
if (error instanceof EmptyPoolError) {
|
|
130
|
+
stats.totalErrors++;
|
|
131
|
+
res.status(503).json(envelope.wrap("no_accounts", "No OpenAI subscription accounts are configured"));
|
|
132
|
+
return;
|
|
133
|
+
}
|
|
134
|
+
if (error instanceof NoEligibleAccountError) {
|
|
135
|
+
// Local rejections are client-facing failures and must show up in the
|
|
136
|
+
// shared error total, exactly as the Anthropic routing middleware counts
|
|
137
|
+
// its own no-eligible-account rejections.
|
|
138
|
+
stats.totalErrors++;
|
|
139
|
+
recordActivity(createLocalRoutingErrorLog(error.reason, requestedModel));
|
|
140
|
+
envelope.sendNoEligible(error, res, now());
|
|
141
|
+
return;
|
|
142
|
+
}
|
|
143
|
+
// Never let an unexpected routing failure crash the daemon or reject
|
|
144
|
+
// this handler's promise — no account lease was taken, so there is
|
|
145
|
+
// nothing to release.
|
|
146
|
+
stats.totalErrors++;
|
|
147
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
148
|
+
logError("proxy", 500, `unexpected routing failure: ${message}`);
|
|
149
|
+
recordActivity({
|
|
150
|
+
ts: now(),
|
|
151
|
+
accountId: "proxy",
|
|
152
|
+
model: requestedModel,
|
|
153
|
+
type: "error",
|
|
154
|
+
statusCode: 500,
|
|
155
|
+
path,
|
|
156
|
+
details: "proxy_error:acquire",
|
|
157
|
+
});
|
|
158
|
+
res.status(500).json(envelope.wrap("proxy_error", "Unexpected routing error"));
|
|
159
|
+
return;
|
|
160
|
+
}
|
|
161
|
+
const account = selected.route.account;
|
|
162
|
+
const startedAt = now();
|
|
163
|
+
const needed = needsOpenAIRefresh(account);
|
|
164
|
+
let ready;
|
|
165
|
+
try {
|
|
166
|
+
ready = await prepareOpenAIAccount(account);
|
|
167
|
+
}
|
|
168
|
+
catch (error) {
|
|
169
|
+
// A throwing refresh must behave exactly like a `false` return, never
|
|
170
|
+
// crash the request (or the daemon).
|
|
171
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
172
|
+
logError(account.id, 401, `openai token refresh threw: ${message}`);
|
|
173
|
+
ready = false;
|
|
174
|
+
}
|
|
175
|
+
if (!ready) {
|
|
176
|
+
selected.release();
|
|
177
|
+
account.errorCount++;
|
|
178
|
+
stats.totalErrors++;
|
|
179
|
+
// Intentionally does not touch `account.healthy`: a single failed
|
|
180
|
+
// refresh fails only this request. Disabling the account here would
|
|
181
|
+
// hard-block it from every future request until a manual recovery, even
|
|
182
|
+
// though the very next request naturally retries the refresh.
|
|
183
|
+
//
|
|
184
|
+
// It must, however, break session affinity and cool the account down.
|
|
185
|
+
// A sticky binding survives this failure, so without both the session
|
|
186
|
+
// would re-acquire the same broken account on every retry and never fail
|
|
187
|
+
// over — 401ing forever while healthy accounts sit idle.
|
|
188
|
+
if (selected.route.sessionId !== undefined && selected.route.bindingGeneration !== undefined) {
|
|
189
|
+
openAIRouter.invalidate(selected.route.sessionId, account.id, selected.route.bindingGeneration);
|
|
190
|
+
}
|
|
191
|
+
openAIPool.setGlobalCooldownForAccount(account, REFRESH_FAILURE_COOLDOWN_MS, "unavailable");
|
|
192
|
+
recordActivity({
|
|
193
|
+
ts: now(),
|
|
194
|
+
accountId: account.id,
|
|
195
|
+
model: requestedModel,
|
|
196
|
+
type: "error",
|
|
197
|
+
statusCode: 401,
|
|
198
|
+
path,
|
|
199
|
+
details: "openai token refresh failed",
|
|
200
|
+
});
|
|
201
|
+
res.status(401).json(envelope.wrap("authentication_error", "OpenAI subscription token refresh failed"));
|
|
202
|
+
return;
|
|
203
|
+
}
|
|
204
|
+
if (responseTerminated(res)) {
|
|
205
|
+
// The client disconnected while the token refresh was in flight. Forwarding
|
|
206
|
+
// now would burn an upstream request nobody can receive the response to.
|
|
207
|
+
// Just release the lease and stop — no response to send, and it is safe to
|
|
208
|
+
// release again even if the response's own close/finish listener already
|
|
209
|
+
// did so (`attachLeaseLifecycle`'s release() is idempotent).
|
|
210
|
+
selected.release();
|
|
211
|
+
return;
|
|
212
|
+
}
|
|
213
|
+
account.healthy = true;
|
|
214
|
+
if (needed)
|
|
215
|
+
account.lastRefresh = now();
|
|
216
|
+
let upstream;
|
|
217
|
+
try {
|
|
218
|
+
upstream = await forwardOpenAI({
|
|
219
|
+
account,
|
|
220
|
+
body: forwardBody,
|
|
221
|
+
stream: forwardBody.stream === true,
|
|
222
|
+
signal: clientGone.signal,
|
|
223
|
+
});
|
|
224
|
+
}
|
|
225
|
+
catch (error) {
|
|
226
|
+
// A client that hung up mid-forward rejects this call through the abort
|
|
227
|
+
// above. That is a cancellation, not an upstream failure: the account did
|
|
228
|
+
// nothing wrong, so counting it would push a healthy account toward the
|
|
229
|
+
// unhealthy threshold and a cooldown for nothing more than a user pressing
|
|
230
|
+
// Ctrl-C, and there is no client left to receive a 502 or to whom an
|
|
231
|
+
// "upstream_error:network" entry would mean anything. Mirrors the
|
|
232
|
+
// pre-forward disconnect branch above, which also just releases and stops.
|
|
233
|
+
if (clientGone.signal.aborted || responseTerminated(res)) {
|
|
234
|
+
selected.release();
|
|
235
|
+
return;
|
|
236
|
+
}
|
|
237
|
+
// A rejected forward call (network failure) must produce a local 502,
|
|
238
|
+
// never an unhandled rejection. The lease releases via the response's
|
|
239
|
+
// own finish/close lifecycle once this response is sent.
|
|
240
|
+
account.errorCount++;
|
|
241
|
+
stats.totalErrors++;
|
|
242
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
243
|
+
logError(account.id, 502, `openai request failed: ${message}`);
|
|
244
|
+
recordActivity({
|
|
245
|
+
ts: startedAt,
|
|
246
|
+
accountId: account.id,
|
|
247
|
+
model: requestedModel,
|
|
248
|
+
type: "error",
|
|
249
|
+
statusCode: 502,
|
|
250
|
+
path,
|
|
251
|
+
details: "upstream_error:network",
|
|
252
|
+
durationMs: now() - startedAt,
|
|
253
|
+
});
|
|
254
|
+
res.status(502).json(envelope.wrap("upstream_error", `OpenAI request failed: ${message}`));
|
|
255
|
+
return;
|
|
256
|
+
}
|
|
257
|
+
// Cooldown/eligibility react to the raw upstream signal — this must not
|
|
258
|
+
// change based on how the relay later renders the response to the client.
|
|
259
|
+
const upstreamFailed = upstream.status === 401 || upstream.status === 429 || upstream.status >= 500;
|
|
260
|
+
let details = routeReasonDetails(selected.route);
|
|
261
|
+
// Tracks whether `account.errorCount`/`consecutiveErrors` were already
|
|
262
|
+
// incremented for this request by the upstream-classification branch below,
|
|
263
|
+
// so a relay-synthesized failure (e.g. a byte-transparent stream that
|
|
264
|
+
// observed an upstream `response.failed`/`error` event on an otherwise-200
|
|
265
|
+
// response) can still increment them once further down without double
|
|
266
|
+
// counting an upstream 401/429/5xx that already did.
|
|
267
|
+
let accountFailureCounted = false;
|
|
268
|
+
try {
|
|
269
|
+
// Header/rate-limit parsing and cooldown bookkeeping run on live upstream
|
|
270
|
+
// data between the two request-level try/catches above — a throw here
|
|
271
|
+
// (e.g. an unreadable header) must degrade to "skip this bookkeeping",
|
|
272
|
+
// never crash the daemon or leave the relay below un-reached.
|
|
273
|
+
const headerRecord = headersToRecord(upstream.headers);
|
|
274
|
+
applyCodexRateLimits(account, parseCodexRateLimits(headerRecord, now()), now());
|
|
275
|
+
if (upstreamFailed) {
|
|
276
|
+
account.errorCount++;
|
|
277
|
+
account.consecutiveErrors++;
|
|
278
|
+
accountFailureCounted = true;
|
|
279
|
+
const applied = applyCodexFailureRouting(upstream.status, headerRecord, selected.route, requestedModel, openAIRouter, openAIPool, now);
|
|
280
|
+
details = routeFailureDetails(selected.route, upstream.status === 401 ? "token-invalid"
|
|
281
|
+
: upstream.status === 429 ? "rate-limited"
|
|
282
|
+
// Only 503/529 are treated as upstream overload for cooldown
|
|
283
|
+
// purposes; labelling an isolated 500/502/504 "service-overloaded"
|
|
284
|
+
// would contradict the routing decision actually taken.
|
|
285
|
+
: upstream.status === 503 || upstream.status === 529 ? "service-overloaded"
|
|
286
|
+
: "upstream-error", applied.limitingScope);
|
|
287
|
+
if (upstream.status === 401)
|
|
288
|
+
onUpstreamAuthFailure?.(account);
|
|
289
|
+
}
|
|
290
|
+
}
|
|
291
|
+
catch (error) {
|
|
292
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
293
|
+
logError(account.id, upstream.status, `openai response classification failed: ${message}`);
|
|
294
|
+
}
|
|
295
|
+
const entry = {
|
|
296
|
+
ts: startedAt,
|
|
297
|
+
accountId: account.id,
|
|
298
|
+
model: requestedModel,
|
|
299
|
+
type: "route",
|
|
300
|
+
path,
|
|
301
|
+
details,
|
|
302
|
+
};
|
|
303
|
+
let finalStatus = upstream.status;
|
|
304
|
+
let relayFailed = false;
|
|
305
|
+
const relayReport = { upstreamReportedFailure: false };
|
|
306
|
+
try {
|
|
307
|
+
const result = await relay(upstream, res, entry, relayReport);
|
|
308
|
+
finalStatus = result.statusCode;
|
|
309
|
+
}
|
|
310
|
+
catch (error) {
|
|
311
|
+
// Never let a relay failure become an unhandled rejection. Only send a
|
|
312
|
+
// local response if no upstream bytes have reached the client yet —
|
|
313
|
+
// otherwise the client already has a partial response and the best we
|
|
314
|
+
// can do is tear the connection down.
|
|
315
|
+
relayFailed = true;
|
|
316
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
317
|
+
logError(account.id, 502, `openai response relay failed: ${message}`);
|
|
318
|
+
// The recorded status is what this request *became*, which is a failure
|
|
319
|
+
// whether or not another HTTP response can still be sent. Leaving it at
|
|
320
|
+
// the upstream's 200 in the headers-already-sent case produced an
|
|
321
|
+
// activity entry typed "error" carrying statusCode 200 — a diagnostic
|
|
322
|
+
// that contradicts itself, and one that reads as a success in any view
|
|
323
|
+
// that keys off the status.
|
|
324
|
+
finalStatus = 502;
|
|
325
|
+
if (!res.headersSent) {
|
|
326
|
+
res.status(502).json(envelope.wrap("upstream_error", `OpenAI response relay failed: ${message}`));
|
|
327
|
+
}
|
|
328
|
+
else {
|
|
329
|
+
if (!res.writableEnded && !res.destroyed)
|
|
330
|
+
res.destroy();
|
|
331
|
+
}
|
|
332
|
+
}
|
|
333
|
+
// A client that hung up during the relay produces every symptom of a
|
|
334
|
+
// failure without there being one: the aborted body rejects the reader (so
|
|
335
|
+
// `relayFailed`), and a stream cut short never reaches its terminal event
|
|
336
|
+
// (so the observer synthesizes a 502). Neither is the account's doing, and
|
|
337
|
+
// charging them would let routine Ctrl-C walk a healthy account to the
|
|
338
|
+
// unhealthy threshold. Only the abort signal can say this — `res` reads as
|
|
339
|
+
// "terminated" after every normal response too.
|
|
340
|
+
//
|
|
341
|
+
// Upstream's own verdict still stands: a 429 is a 429 whether or not the
|
|
342
|
+
// client stayed to read it, and neither is an explicit `response.failed` on
|
|
343
|
+
// a 200 stream — the client can truncate a stream, but it cannot make
|
|
344
|
+
// upstream announce a failure. Only the truncation is the disconnect's to
|
|
345
|
+
// explain away.
|
|
346
|
+
const clientCancelled = clientGone.signal.aborted && !relayReport.upstreamReportedFailure;
|
|
347
|
+
// Activity/stats must reflect what the client actually received, not just
|
|
348
|
+
// the raw upstream signal: the non-streaming collector can synthesize a
|
|
349
|
+
// local 502 from an upstream 200 whose SSE stream ended in
|
|
350
|
+
// `response.failed` (or malformed/incomplete), and a relay failure is
|
|
351
|
+
// always a client-facing failure regardless of the upstream status.
|
|
352
|
+
const failedFinal = upstreamFailed || (!clientCancelled && (relayFailed || finalStatus >= 400));
|
|
353
|
+
if (clientCancelled && !upstreamFailed) {
|
|
354
|
+
// Record what the client had actually received when it left, not the 502
|
|
355
|
+
// its own disconnect manufactured.
|
|
356
|
+
finalStatus = upstream.status;
|
|
357
|
+
entry.details = details ? `${details} client-cancelled` : "client-cancelled";
|
|
358
|
+
}
|
|
359
|
+
if (failedFinal) {
|
|
360
|
+
stats.totalErrors++;
|
|
361
|
+
// Upstream classification above only counts 401/429/5xx against the
|
|
362
|
+
// account. A relay-synthesized failure on an otherwise-successful
|
|
363
|
+
// upstream status (e.g. a streamed `response.failed` event, or a relay
|
|
364
|
+
// exception after upstream returned 200) is just as real a failure for
|
|
365
|
+
// this account and must not be dropped on the floor.
|
|
366
|
+
if (!accountFailureCounted) {
|
|
367
|
+
account.errorCount++;
|
|
368
|
+
account.consecutiveErrors++;
|
|
369
|
+
}
|
|
370
|
+
}
|
|
371
|
+
else {
|
|
372
|
+
account.consecutiveErrors = 0;
|
|
373
|
+
stats.totalRequests++;
|
|
374
|
+
}
|
|
375
|
+
entry.type = failedFinal ? "error" : "route";
|
|
376
|
+
entry.statusCode = finalStatus;
|
|
377
|
+
entry.durationMs = now() - startedAt;
|
|
378
|
+
recordActivity(entry);
|
|
379
|
+
}
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
import { extractClaudeSessionId } from "./anthropic-routing.js";
|
|
2
|
+
import { normalizeSessionId } from "./session-router.js";
|
|
3
|
+
const CODEX_SESSION_HEADER = "session_id";
|
|
4
|
+
/** Extract exactly one native HTTP header field without joined duplicates. */
|
|
5
|
+
function extractSingleHeader(request, name) {
|
|
6
|
+
const distinct = request.headersDistinct;
|
|
7
|
+
if (distinct !== undefined) {
|
|
8
|
+
const values = distinct[name];
|
|
9
|
+
if (!values || values.length !== 1)
|
|
10
|
+
return undefined;
|
|
11
|
+
return normalizeSessionId(values[0]);
|
|
12
|
+
}
|
|
13
|
+
const values = [];
|
|
14
|
+
for (let index = 0; index < request.rawHeaders.length; index += 2) {
|
|
15
|
+
if (request.rawHeaders[index]?.toLowerCase() !== name)
|
|
16
|
+
continue;
|
|
17
|
+
values.push(request.rawHeaders[index + 1] ?? "");
|
|
18
|
+
}
|
|
19
|
+
if (values.length !== 1)
|
|
20
|
+
return undefined;
|
|
21
|
+
return normalizeSessionId(values[0]);
|
|
22
|
+
}
|
|
23
|
+
/**
|
|
24
|
+
* Resolve the OpenAI affinity key in priority order: Codex session_id header,
|
|
25
|
+
* Claude Code session header, then the request body's prompt_cache_key
|
|
26
|
+
* (Codex thread id). Returns undefined for unscoped requests.
|
|
27
|
+
*/
|
|
28
|
+
export function extractCodexSessionKey(request, body) {
|
|
29
|
+
const codexSession = extractSingleHeader(request, CODEX_SESSION_HEADER);
|
|
30
|
+
if (codexSession !== undefined)
|
|
31
|
+
return codexSession;
|
|
32
|
+
const claudeSession = extractClaudeSessionId(request);
|
|
33
|
+
if (claudeSession !== undefined)
|
|
34
|
+
return claudeSession;
|
|
35
|
+
const promptCacheKey = body !== null && typeof body === "object"
|
|
36
|
+
? body.prompt_cache_key
|
|
37
|
+
: undefined;
|
|
38
|
+
return normalizeSessionId(promptCacheKey);
|
|
39
|
+
}
|
|
40
|
+
/** Local OpenAI/Responses-shaped rejection — zero upstream requests were made. */
|
|
41
|
+
export function sendOpenAINoEligibleResponse(error, response, nowMs) {
|
|
42
|
+
if (error.reason === "rate_limited") {
|
|
43
|
+
if (error.retryAtMs !== undefined) {
|
|
44
|
+
const retryAfterSeconds = Math.max(0, Math.ceil((error.retryAtMs - nowMs) / 1_000));
|
|
45
|
+
response.setHeader("Retry-After", String(retryAfterSeconds));
|
|
46
|
+
}
|
|
47
|
+
response.status(429).json({
|
|
48
|
+
error: {
|
|
49
|
+
type: "rate_limit_exceeded",
|
|
50
|
+
message: "All configured OpenAI accounts are currently rate limited",
|
|
51
|
+
},
|
|
52
|
+
});
|
|
53
|
+
return;
|
|
54
|
+
}
|
|
55
|
+
response.status(503).json({
|
|
56
|
+
error: {
|
|
57
|
+
type: "service_unavailable",
|
|
58
|
+
message: "All configured OpenAI accounts are currently unavailable",
|
|
59
|
+
},
|
|
60
|
+
});
|
|
61
|
+
}
|
|
@@ -1,11 +1,15 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Persist a provider toggle before discarding
|
|
3
|
-
* persistence throws, the caller can roll back runtime
|
|
4
|
-
* losing bindings that still point at valid accounts.
|
|
2
|
+
* Persist a provider toggle before discarding that provider's session
|
|
3
|
+
* affinity. If persistence throws, the caller can roll back runtime
|
|
4
|
+
* enablement without losing bindings that still point at valid accounts.
|
|
5
|
+
*
|
|
6
|
+
* Invalidation is provider-agnostic: the caller supplies the account ids and
|
|
7
|
+
* the invalidator belonging to `provider`, so both the Anthropic and the
|
|
8
|
+
* OpenAI router drop bindings when their provider is disabled.
|
|
5
9
|
*/
|
|
6
10
|
export function persistProviderEnabledState(options) {
|
|
7
11
|
const result = options.persist();
|
|
8
|
-
if (
|
|
12
|
+
if (!options.enabled) {
|
|
9
13
|
for (const accountId of options.accountIds) {
|
|
10
14
|
options.invalidateAccount(accountId);
|
|
11
15
|
}
|