@timo972/cc-router 0.9.0 → 0.10.0-rc.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,379 @@
1
+ import { applyCodexRateLimits } from "../providers/openai/account-state.js";
2
+ import { headersToRecord, parseCodexRateLimits } from "../providers/openai/usage.js";
3
+ import { applyCodexFailureRouting } from "../providers/openai/failure-routing.js";
4
+ import { needsOpenAIRefresh } from "../providers/openai/token-refresher.js";
5
+ import { stats, boundModelId, createLocalRoutingErrorLog } from "./stats.js";
6
+ import { logError } from "./logger.js";
7
+ import { EmptyPoolError, NoEligibleAccountError } from "./account-pool.js";
8
+ import { acquireRequestRoute, routeReasonDetails, routeFailureDetails } from "./lease-lifecycle.js";
9
+ /**
10
+ * Mirrors `anthropic-routing.ts`'s `requestTerminated` check. This ingress
11
+ * path never threads the raw `Request` through (only `Response`), so it
12
+ * checks just the response side: `res.destroyed`/`res.writableEnded` are
13
+ * enough to detect a client that disconnected while we were off awaiting a
14
+ * token refresh.
15
+ */
16
+ function responseTerminated(res) {
17
+ return res.destroyed || res.writableEnded;
18
+ }
19
+ /**
20
+ * Cooldown applied when a *local* token refresh fails. Matches the upstream-401
21
+ * cooldown in `failure-routing.ts`: a refresh that cannot produce a usable token
22
+ * is an auth failure, and without a cooldown the pool would immediately hand the
23
+ * same account back to the next request.
24
+ */
25
+ const REFRESH_FAILURE_COOLDOWN_MS = 30_000;
26
+ /**
27
+ * Response headers that must never be mirrored to the local client when
28
+ * relaying an upstream response verbatim:
29
+ * - hop-by-hop headers per RFC 7230 §6.1 (content-length, transfer-encoding,
30
+ * connection, keep-alive, te, trailer, upgrade, proxy-authenticate,
31
+ * proxy-authorization) are meaningless (or actively wrong/dangerous) once
32
+ * re-framed by our own HTTP server — e.g. `upgrade` would claim a protocol
33
+ * switch never negotiated with this client, and the two `proxy-*` headers
34
+ * are scoped to the upstream hop's own (unrelated) proxy auth.
35
+ * - content-encoding is dropped because undici's fetch() already
36
+ * transparently decompresses the body while leaving this header intact —
37
+ * forwarding it would tell the client to gunzip bytes that are no longer
38
+ * compressed. Deliberate drop, not RFC hop-by-hop.
39
+ * - set-cookie must never leak the upstream service's session cookies to a
40
+ * local client. Deliberate drop, not RFC hop-by-hop.
41
+ */
42
+ export const EXCLUDED_UPSTREAM_RELAY_HEADERS = new Set([
43
+ "content-length",
44
+ "transfer-encoding",
45
+ "connection",
46
+ "keep-alive",
47
+ "te",
48
+ "trailer",
49
+ "upgrade",
50
+ "proxy-authenticate",
51
+ "proxy-authorization",
52
+ "content-encoding",
53
+ "set-cookie",
54
+ ]);
55
+ /**
56
+ * The `Connection` header can nominate additional header names as hop-by-hop
57
+ * for this specific response (RFC 7230 §6.1), beyond the fixed set above —
58
+ * e.g. `Connection: close, X-Internal-Token` means `X-Internal-Token` is also
59
+ * hop-by-hop here and must not reach the client.
60
+ */
61
+ function connectionNominatedHeaders(source) {
62
+ const nominated = new Set();
63
+ const connection = source.get("connection");
64
+ if (!connection)
65
+ return nominated;
66
+ for (const token of connection.split(",")) {
67
+ const name = token.trim().toLowerCase();
68
+ if (name)
69
+ nominated.add(name);
70
+ }
71
+ return nominated;
72
+ }
73
+ /**
74
+ * Single place that decides which upstream response headers are safe to
75
+ * mirror to the local client, so every relay site shares the same policy
76
+ * instead of re-implementing the exclusion set (and the dynamic `Connection`
77
+ * nomination) inline. `apply` is called once per header that passes the
78
+ * filter, in `source`'s own iteration order — callers still own their own
79
+ * per-header special cases (e.g. content-type) by skipping in `apply`.
80
+ */
81
+ export function mirrorUpstreamHeaders(source, apply) {
82
+ const nominated = connectionNominatedHeaders(source);
83
+ source.forEach((value, key) => {
84
+ const lower = key.toLowerCase();
85
+ if (EXCLUDED_UPSTREAM_RELAY_HEADERS.has(lower))
86
+ return;
87
+ if (nominated.has(lower))
88
+ return;
89
+ apply(key, value);
90
+ });
91
+ }
92
+ /**
93
+ * Shared OpenAI/Codex ingress lifecycle: acquire a sticky account lease,
94
+ * refresh its token if needed, forward the request, classify the upstream
95
+ * failure for cooldown/eligibility purposes, relay the response to the
96
+ * client, then record activity/stats keyed on what the client actually
97
+ * received. Every awaited step is guarded so a rejection here can only ever
98
+ * produce a local error response — it must never crash the daemon or leave
99
+ * an unhandled rejection behind.
100
+ */
101
+ export async function runOpenAIIngress(opts) {
102
+ const { res, sessionKey, path, openAIRouter, openAIPool, prepareOpenAIAccount, forwardOpenAI, forwardBody, recordActivity, now, envelope, relay, onUpstreamAuthFailure, } = opts;
103
+ // The model comes from a client-controlled body and is retained in the
104
+ // activity ring buffer below. Bound it once, here, so every activity entry,
105
+ // routing context and bucket lookup on this path carries an identifier that
106
+ // cannot grow with the request. The body forwarded upstream is untouched —
107
+ // it still carries whatever model the caller asked for.
108
+ const requestedModel = boundModelId(opts.requestedModel);
109
+ // A client that hangs up must take the upstream request with it. Releasing
110
+ // the lease (which the response's own close listener does) only returns the
111
+ // account's *local* capacity — without this the Codex request keeps
112
+ // streaming to a socket nobody is reading, so the pool counts the account
113
+ // idle and routes more work onto an upstream slot that is still occupied.
114
+ //
115
+ // Registered before the lease is acquired so this listener runs before the
116
+ // lifecycle's release, and `once` so it cleans itself up. A normal end also
117
+ // emits `close`, hence the `writableEnded` guard: only a premature close is
118
+ // a disconnect.
119
+ const clientGone = new AbortController();
120
+ res.once("close", () => {
121
+ if (!res.writableEnded)
122
+ clientGone.abort();
123
+ });
124
+ let selected;
125
+ try {
126
+ selected = acquireRequestRoute(sessionKey, res, openAIRouter, { requestedModel });
127
+ }
128
+ catch (error) {
129
+ if (error instanceof EmptyPoolError) {
130
+ stats.totalErrors++;
131
+ res.status(503).json(envelope.wrap("no_accounts", "No OpenAI subscription accounts are configured"));
132
+ return;
133
+ }
134
+ if (error instanceof NoEligibleAccountError) {
135
+ // Local rejections are client-facing failures and must show up in the
136
+ // shared error total, exactly as the Anthropic routing middleware counts
137
+ // its own no-eligible-account rejections.
138
+ stats.totalErrors++;
139
+ recordActivity(createLocalRoutingErrorLog(error.reason, requestedModel));
140
+ envelope.sendNoEligible(error, res, now());
141
+ return;
142
+ }
143
+ // Never let an unexpected routing failure crash the daemon or reject
144
+ // this handler's promise — no account lease was taken, so there is
145
+ // nothing to release.
146
+ stats.totalErrors++;
147
+ const message = error instanceof Error ? error.message : String(error);
148
+ logError("proxy", 500, `unexpected routing failure: ${message}`);
149
+ recordActivity({
150
+ ts: now(),
151
+ accountId: "proxy",
152
+ model: requestedModel,
153
+ type: "error",
154
+ statusCode: 500,
155
+ path,
156
+ details: "proxy_error:acquire",
157
+ });
158
+ res.status(500).json(envelope.wrap("proxy_error", "Unexpected routing error"));
159
+ return;
160
+ }
161
+ const account = selected.route.account;
162
+ const startedAt = now();
163
+ const needed = needsOpenAIRefresh(account);
164
+ let ready;
165
+ try {
166
+ ready = await prepareOpenAIAccount(account);
167
+ }
168
+ catch (error) {
169
+ // A throwing refresh must behave exactly like a `false` return, never
170
+ // crash the request (or the daemon).
171
+ const message = error instanceof Error ? error.message : String(error);
172
+ logError(account.id, 401, `openai token refresh threw: ${message}`);
173
+ ready = false;
174
+ }
175
+ if (!ready) {
176
+ selected.release();
177
+ account.errorCount++;
178
+ stats.totalErrors++;
179
+ // Intentionally does not touch `account.healthy`: a single failed
180
+ // refresh fails only this request. Disabling the account here would
181
+ // hard-block it from every future request until a manual recovery, even
182
+ // though the very next request naturally retries the refresh.
183
+ //
184
+ // It must, however, break session affinity and cool the account down.
185
+ // A sticky binding survives this failure, so without both the session
186
+ // would re-acquire the same broken account on every retry and never fail
187
+ // over — 401ing forever while healthy accounts sit idle.
188
+ if (selected.route.sessionId !== undefined && selected.route.bindingGeneration !== undefined) {
189
+ openAIRouter.invalidate(selected.route.sessionId, account.id, selected.route.bindingGeneration);
190
+ }
191
+ openAIPool.setGlobalCooldownForAccount(account, REFRESH_FAILURE_COOLDOWN_MS, "unavailable");
192
+ recordActivity({
193
+ ts: now(),
194
+ accountId: account.id,
195
+ model: requestedModel,
196
+ type: "error",
197
+ statusCode: 401,
198
+ path,
199
+ details: "openai token refresh failed",
200
+ });
201
+ res.status(401).json(envelope.wrap("authentication_error", "OpenAI subscription token refresh failed"));
202
+ return;
203
+ }
204
+ if (responseTerminated(res)) {
205
+ // The client disconnected while the token refresh was in flight. Forwarding
206
+ // now would burn an upstream request nobody can receive the response to.
207
+ // Just release the lease and stop — no response to send, and it is safe to
208
+ // release again even if the response's own close/finish listener already
209
+ // did so (`attachLeaseLifecycle`'s release() is idempotent).
210
+ selected.release();
211
+ return;
212
+ }
213
+ account.healthy = true;
214
+ if (needed)
215
+ account.lastRefresh = now();
216
+ let upstream;
217
+ try {
218
+ upstream = await forwardOpenAI({
219
+ account,
220
+ body: forwardBody,
221
+ stream: forwardBody.stream === true,
222
+ signal: clientGone.signal,
223
+ });
224
+ }
225
+ catch (error) {
226
+ // A client that hung up mid-forward rejects this call through the abort
227
+ // above. That is a cancellation, not an upstream failure: the account did
228
+ // nothing wrong, so counting it would push a healthy account toward the
229
+ // unhealthy threshold and a cooldown for nothing more than a user pressing
230
+ // Ctrl-C, and there is no client left to receive a 502 or to whom an
231
+ // "upstream_error:network" entry would mean anything. Mirrors the
232
+ // pre-forward disconnect branch above, which also just releases and stops.
233
+ if (clientGone.signal.aborted || responseTerminated(res)) {
234
+ selected.release();
235
+ return;
236
+ }
237
+ // A rejected forward call (network failure) must produce a local 502,
238
+ // never an unhandled rejection. The lease releases via the response's
239
+ // own finish/close lifecycle once this response is sent.
240
+ account.errorCount++;
241
+ stats.totalErrors++;
242
+ const message = error instanceof Error ? error.message : String(error);
243
+ logError(account.id, 502, `openai request failed: ${message}`);
244
+ recordActivity({
245
+ ts: startedAt,
246
+ accountId: account.id,
247
+ model: requestedModel,
248
+ type: "error",
249
+ statusCode: 502,
250
+ path,
251
+ details: "upstream_error:network",
252
+ durationMs: now() - startedAt,
253
+ });
254
+ res.status(502).json(envelope.wrap("upstream_error", `OpenAI request failed: ${message}`));
255
+ return;
256
+ }
257
+ // Cooldown/eligibility react to the raw upstream signal — this must not
258
+ // change based on how the relay later renders the response to the client.
259
+ const upstreamFailed = upstream.status === 401 || upstream.status === 429 || upstream.status >= 500;
260
+ let details = routeReasonDetails(selected.route);
261
+ // Tracks whether `account.errorCount`/`consecutiveErrors` were already
262
+ // incremented for this request by the upstream-classification branch below,
263
+ // so a relay-synthesized failure (e.g. a byte-transparent stream that
264
+ // observed an upstream `response.failed`/`error` event on an otherwise-200
265
+ // response) can still increment them once further down without double
266
+ // counting an upstream 401/429/5xx that already did.
267
+ let accountFailureCounted = false;
268
+ try {
269
+ // Header/rate-limit parsing and cooldown bookkeeping run on live upstream
270
+ // data between the two request-level try/catches above — a throw here
271
+ // (e.g. an unreadable header) must degrade to "skip this bookkeeping",
272
+ // never crash the daemon or leave the relay below un-reached.
273
+ const headerRecord = headersToRecord(upstream.headers);
274
+ applyCodexRateLimits(account, parseCodexRateLimits(headerRecord, now()), now());
275
+ if (upstreamFailed) {
276
+ account.errorCount++;
277
+ account.consecutiveErrors++;
278
+ accountFailureCounted = true;
279
+ const applied = applyCodexFailureRouting(upstream.status, headerRecord, selected.route, requestedModel, openAIRouter, openAIPool, now);
280
+ details = routeFailureDetails(selected.route, upstream.status === 401 ? "token-invalid"
281
+ : upstream.status === 429 ? "rate-limited"
282
+ // Only 503/529 are treated as upstream overload for cooldown
283
+ // purposes; labelling an isolated 500/502/504 "service-overloaded"
284
+ // would contradict the routing decision actually taken.
285
+ : upstream.status === 503 || upstream.status === 529 ? "service-overloaded"
286
+ : "upstream-error", applied.limitingScope);
287
+ if (upstream.status === 401)
288
+ onUpstreamAuthFailure?.(account);
289
+ }
290
+ }
291
+ catch (error) {
292
+ const message = error instanceof Error ? error.message : String(error);
293
+ logError(account.id, upstream.status, `openai response classification failed: ${message}`);
294
+ }
295
+ const entry = {
296
+ ts: startedAt,
297
+ accountId: account.id,
298
+ model: requestedModel,
299
+ type: "route",
300
+ path,
301
+ details,
302
+ };
303
+ let finalStatus = upstream.status;
304
+ let relayFailed = false;
305
+ const relayReport = { upstreamReportedFailure: false };
306
+ try {
307
+ const result = await relay(upstream, res, entry, relayReport);
308
+ finalStatus = result.statusCode;
309
+ }
310
+ catch (error) {
311
+ // Never let a relay failure become an unhandled rejection. Only send a
312
+ // local response if no upstream bytes have reached the client yet —
313
+ // otherwise the client already has a partial response and the best we
314
+ // can do is tear the connection down.
315
+ relayFailed = true;
316
+ const message = error instanceof Error ? error.message : String(error);
317
+ logError(account.id, 502, `openai response relay failed: ${message}`);
318
+ // The recorded status is what this request *became*, which is a failure
319
+ // whether or not another HTTP response can still be sent. Leaving it at
320
+ // the upstream's 200 in the headers-already-sent case produced an
321
+ // activity entry typed "error" carrying statusCode 200 — a diagnostic
322
+ // that contradicts itself, and one that reads as a success in any view
323
+ // that keys off the status.
324
+ finalStatus = 502;
325
+ if (!res.headersSent) {
326
+ res.status(502).json(envelope.wrap("upstream_error", `OpenAI response relay failed: ${message}`));
327
+ }
328
+ else {
329
+ if (!res.writableEnded && !res.destroyed)
330
+ res.destroy();
331
+ }
332
+ }
333
+ // A client that hung up during the relay produces every symptom of a
334
+ // failure without there being one: the aborted body rejects the reader (so
335
+ // `relayFailed`), and a stream cut short never reaches its terminal event
336
+ // (so the observer synthesizes a 502). Neither is the account's doing, and
337
+ // charging them would let routine Ctrl-C walk a healthy account to the
338
+ // unhealthy threshold. Only the abort signal can say this — `res` reads as
339
+ // "terminated" after every normal response too.
340
+ //
341
+ // Upstream's own verdict still stands: a 429 is a 429 whether or not the
342
+ // client stayed to read it, and neither is an explicit `response.failed` on
343
+ // a 200 stream — the client can truncate a stream, but it cannot make
344
+ // upstream announce a failure. Only the truncation is the disconnect's to
345
+ // explain away.
346
+ const clientCancelled = clientGone.signal.aborted && !relayReport.upstreamReportedFailure;
347
+ // Activity/stats must reflect what the client actually received, not just
348
+ // the raw upstream signal: the non-streaming collector can synthesize a
349
+ // local 502 from an upstream 200 whose SSE stream ended in
350
+ // `response.failed` (or malformed/incomplete), and a relay failure is
351
+ // always a client-facing failure regardless of the upstream status.
352
+ const failedFinal = upstreamFailed || (!clientCancelled && (relayFailed || finalStatus >= 400));
353
+ if (clientCancelled && !upstreamFailed) {
354
+ // Record what the client had actually received when it left, not the 502
355
+ // its own disconnect manufactured.
356
+ finalStatus = upstream.status;
357
+ entry.details = details ? `${details} client-cancelled` : "client-cancelled";
358
+ }
359
+ if (failedFinal) {
360
+ stats.totalErrors++;
361
+ // Upstream classification above only counts 401/429/5xx against the
362
+ // account. A relay-synthesized failure on an otherwise-successful
363
+ // upstream status (e.g. a streamed `response.failed` event, or a relay
364
+ // exception after upstream returned 200) is just as real a failure for
365
+ // this account and must not be dropped on the floor.
366
+ if (!accountFailureCounted) {
367
+ account.errorCount++;
368
+ account.consecutiveErrors++;
369
+ }
370
+ }
371
+ else {
372
+ account.consecutiveErrors = 0;
373
+ stats.totalRequests++;
374
+ }
375
+ entry.type = failedFinal ? "error" : "route";
376
+ entry.statusCode = finalStatus;
377
+ entry.durationMs = now() - startedAt;
378
+ recordActivity(entry);
379
+ }
@@ -0,0 +1,61 @@
1
+ import { extractClaudeSessionId } from "./anthropic-routing.js";
2
+ import { normalizeSessionId } from "./session-router.js";
3
+ const CODEX_SESSION_HEADER = "session_id";
4
+ /** Extract exactly one native HTTP header field without joined duplicates. */
5
+ function extractSingleHeader(request, name) {
6
+ const distinct = request.headersDistinct;
7
+ if (distinct !== undefined) {
8
+ const values = distinct[name];
9
+ if (!values || values.length !== 1)
10
+ return undefined;
11
+ return normalizeSessionId(values[0]);
12
+ }
13
+ const values = [];
14
+ for (let index = 0; index < request.rawHeaders.length; index += 2) {
15
+ if (request.rawHeaders[index]?.toLowerCase() !== name)
16
+ continue;
17
+ values.push(request.rawHeaders[index + 1] ?? "");
18
+ }
19
+ if (values.length !== 1)
20
+ return undefined;
21
+ return normalizeSessionId(values[0]);
22
+ }
23
+ /**
24
+ * Resolve the OpenAI affinity key in priority order: Codex session_id header,
25
+ * Claude Code session header, then the request body's prompt_cache_key
26
+ * (Codex thread id). Returns undefined for unscoped requests.
27
+ */
28
+ export function extractCodexSessionKey(request, body) {
29
+ const codexSession = extractSingleHeader(request, CODEX_SESSION_HEADER);
30
+ if (codexSession !== undefined)
31
+ return codexSession;
32
+ const claudeSession = extractClaudeSessionId(request);
33
+ if (claudeSession !== undefined)
34
+ return claudeSession;
35
+ const promptCacheKey = body !== null && typeof body === "object"
36
+ ? body.prompt_cache_key
37
+ : undefined;
38
+ return normalizeSessionId(promptCacheKey);
39
+ }
40
+ /** Local OpenAI/Responses-shaped rejection — zero upstream requests were made. */
41
+ export function sendOpenAINoEligibleResponse(error, response, nowMs) {
42
+ if (error.reason === "rate_limited") {
43
+ if (error.retryAtMs !== undefined) {
44
+ const retryAfterSeconds = Math.max(0, Math.ceil((error.retryAtMs - nowMs) / 1_000));
45
+ response.setHeader("Retry-After", String(retryAfterSeconds));
46
+ }
47
+ response.status(429).json({
48
+ error: {
49
+ type: "rate_limit_exceeded",
50
+ message: "All configured OpenAI accounts are currently rate limited",
51
+ },
52
+ });
53
+ return;
54
+ }
55
+ response.status(503).json({
56
+ error: {
57
+ type: "service_unavailable",
58
+ message: "All configured OpenAI accounts are currently unavailable",
59
+ },
60
+ });
61
+ }
@@ -1,11 +1,15 @@
1
1
  /**
2
- * Persist a provider toggle before discarding any Anthropic affinity. If
3
- * persistence throws, the caller can roll back runtime enablement without
4
- * losing bindings that still point at valid accounts.
2
+ * Persist a provider toggle before discarding that provider's session
3
+ * affinity. If persistence throws, the caller can roll back runtime
4
+ * enablement without losing bindings that still point at valid accounts.
5
+ *
6
+ * Invalidation is provider-agnostic: the caller supplies the account ids and
7
+ * the invalidator belonging to `provider`, so both the Anthropic and the
8
+ * OpenAI router drop bindings when their provider is disabled.
5
9
  */
6
10
  export function persistProviderEnabledState(options) {
7
11
  const result = options.persist();
8
- if (options.provider === "anthropic_subscription" && !options.enabled) {
12
+ if (!options.enabled) {
9
13
  for (const accountId of options.accountIds) {
10
14
  options.invalidateAccount(accountId);
11
15
  }