@timo972/cc-router 0.10.1 → 0.11.0-rc.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +106 -0
- package/README.md +4 -3
- package/dist/config/manager.js +9 -0
- package/dist/providers/anthropic/rate-limit-headers.js +44 -0
- package/dist/providers/anthropic/usage.js +24 -5
- package/dist/proxy/anthropic-messages-route.js +456 -0
- package/dist/proxy/anthropic-response-capture.js +40 -0
- package/dist/proxy/event-sequence.js +18 -0
- package/dist/proxy/lease-lifecycle.js +20 -13
- package/dist/proxy/messages-cross-route.js +7 -0
- package/dist/proxy/openai-ingress.js +207 -108
- package/dist/proxy/responses-server.js +7 -0
- package/dist/proxy/server.js +81 -109
- package/dist/proxy/stats.js +19 -0
- package/dist/proxy/token-pool.js +302 -37
- package/dist/proxy/upstream-retry.js +87 -0
- package/package.json +1 -1
|
@@ -6,6 +6,7 @@ import { stats, boundModelId, createLocalRoutingErrorLog } from "./stats.js";
|
|
|
6
6
|
import { logError } from "./logger.js";
|
|
7
7
|
import { EmptyPoolError, NoEligibleAccountError } from "./account-pool.js";
|
|
8
8
|
import { acquireRequestRoute, routeReasonDetails, routeFailureDetails } from "./lease-lifecycle.js";
|
|
9
|
+
import { MAX_UPSTREAM_ATTEMPTS, RETRY_REFRESH_TIMEOUT_MS, SAME_ACCOUNT_RETRY_DELAY_MS, boundedWait, isRetryableUpstreamStatus, retryDelay, } from "./upstream-retry.js";
|
|
9
10
|
/**
|
|
10
11
|
* Mirrors `anthropic-routing.ts`'s `requestTerminated` check. This ingress
|
|
11
12
|
* path never threads the raw `Request` through (only `Response`), so it
|
|
@@ -158,46 +159,64 @@ export async function runOpenAIIngress(opts) {
|
|
|
158
159
|
res.status(500).json(envelope.wrap("proxy_error", "Unexpected routing error"));
|
|
159
160
|
return;
|
|
160
161
|
}
|
|
161
|
-
const account = selected.route.account;
|
|
162
162
|
const startedAt = now();
|
|
163
|
-
const
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
stats.totalErrors++;
|
|
179
|
-
// Intentionally does not touch `account.healthy`: a single failed
|
|
180
|
-
// refresh fails only this request. Disabling the account here would
|
|
181
|
-
// hard-block it from every future request until a manual recovery, even
|
|
182
|
-
// though the very next request naturally retries the refresh.
|
|
183
|
-
//
|
|
184
|
-
// It must, however, break session affinity and cool the account down.
|
|
185
|
-
// A sticky binding survives this failure, so without both the session
|
|
186
|
-
// would re-acquire the same broken account on every retry and never fail
|
|
187
|
-
// over — 401ing forever while healthy accounts sit idle.
|
|
188
|
-
if (selected.route.sessionId !== undefined && selected.route.bindingGeneration !== undefined) {
|
|
189
|
-
openAIRouter.invalidate(selected.route.sessionId, account.id, selected.route.bindingGeneration);
|
|
163
|
+
const maxAttempts = Math.max(1, opts.maxAttempts ?? MAX_UPSTREAM_ATTEMPTS);
|
|
164
|
+
const sameAccountDelayMs = opts.sameAccountRetryDelayMs ?? SAME_ACCOUNT_RETRY_DELAY_MS;
|
|
165
|
+
const retryRefreshTimeoutMs = opts.retryRefreshTimeoutMs ?? RETRY_REFRESH_TIMEOUT_MS;
|
|
166
|
+
/**
|
|
167
|
+
* Refresh a routed account's token if needed. On failure this applies the
|
|
168
|
+
* shared refresh-failure bookkeeping and returns false; what the caller
|
|
169
|
+
* sends instead is its own decision — the first attempt answers a local
|
|
170
|
+
* 401, a retry attempt relays the upstream failure it already holds.
|
|
171
|
+
*/
|
|
172
|
+
const prepareRoute = async (routed) => {
|
|
173
|
+
const account = routed.route.account;
|
|
174
|
+
const needed = needsOpenAIRefresh(account);
|
|
175
|
+
let ready;
|
|
176
|
+
try {
|
|
177
|
+
ready = await prepareOpenAIAccount(account);
|
|
190
178
|
}
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
179
|
+
catch (error) {
|
|
180
|
+
// A throwing refresh must behave exactly like a `false` return, never
|
|
181
|
+
// crash the request (or the daemon).
|
|
182
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
183
|
+
logError(account.id, 401, `openai token refresh threw: ${message}`);
|
|
184
|
+
ready = false;
|
|
185
|
+
}
|
|
186
|
+
if (!ready) {
|
|
187
|
+
routed.release();
|
|
188
|
+
account.errorCount++;
|
|
189
|
+
stats.totalErrors++;
|
|
190
|
+
// Intentionally does not touch `account.healthy`: a single failed
|
|
191
|
+
// refresh fails only this request. Disabling the account here would
|
|
192
|
+
// hard-block it from every future request until a manual recovery, even
|
|
193
|
+
// though the very next request naturally retries the refresh.
|
|
194
|
+
//
|
|
195
|
+
// It must, however, break session affinity and cool the account down.
|
|
196
|
+
// A sticky binding survives this failure, so without both the session
|
|
197
|
+
// would re-acquire the same broken account on every retry and never fail
|
|
198
|
+
// over — 401ing forever while healthy accounts sit idle.
|
|
199
|
+
if (routed.route.sessionId !== undefined && routed.route.bindingGeneration !== undefined) {
|
|
200
|
+
openAIRouter.invalidate(routed.route.sessionId, account.id, routed.route.bindingGeneration);
|
|
201
|
+
}
|
|
202
|
+
openAIPool.setGlobalCooldownForAccount(account, REFRESH_FAILURE_COOLDOWN_MS, "unavailable");
|
|
203
|
+
recordActivity({
|
|
204
|
+
ts: now(),
|
|
205
|
+
accountId: account.id,
|
|
206
|
+
model: requestedModel,
|
|
207
|
+
type: "error",
|
|
208
|
+
statusCode: 401,
|
|
209
|
+
path,
|
|
210
|
+
details: "openai token refresh failed",
|
|
211
|
+
});
|
|
212
|
+
return false;
|
|
213
|
+
}
|
|
214
|
+
account.healthy = true;
|
|
215
|
+
if (needed)
|
|
216
|
+
account.lastRefresh = now();
|
|
217
|
+
return true;
|
|
218
|
+
};
|
|
219
|
+
if (!(await prepareRoute(selected))) {
|
|
201
220
|
res.status(401).json(envelope.wrap("authentication_error", "OpenAI subscription token refresh failed"));
|
|
202
221
|
return;
|
|
203
222
|
}
|
|
@@ -210,88 +229,168 @@ export async function runOpenAIIngress(opts) {
|
|
|
210
229
|
selected.release();
|
|
211
230
|
return;
|
|
212
231
|
}
|
|
213
|
-
account.healthy = true;
|
|
214
|
-
if (needed)
|
|
215
|
-
account.lastRefresh = now();
|
|
216
232
|
let upstream;
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
233
|
+
let upstreamFailed;
|
|
234
|
+
let details;
|
|
235
|
+
let accountFailureCounted;
|
|
236
|
+
for (let attempt = 1;; attempt++) {
|
|
237
|
+
const account = selected.route.account;
|
|
238
|
+
const attemptStartedAt = now();
|
|
239
|
+
try {
|
|
240
|
+
upstream = await forwardOpenAI({
|
|
241
|
+
account,
|
|
242
|
+
body: forwardBody,
|
|
243
|
+
stream: forwardBody.stream === true,
|
|
244
|
+
signal: clientGone.signal,
|
|
245
|
+
});
|
|
246
|
+
}
|
|
247
|
+
catch (error) {
|
|
248
|
+
// A client that hung up mid-forward rejects this call through the abort
|
|
249
|
+
// above. That is a cancellation, not an upstream failure: the account did
|
|
250
|
+
// nothing wrong, so counting it would push a healthy account toward the
|
|
251
|
+
// unhealthy threshold and a cooldown for nothing more than a user pressing
|
|
252
|
+
// Ctrl-C, and there is no client left to receive a 502 or to whom an
|
|
253
|
+
// "upstream_error:network" entry would mean anything. Mirrors the
|
|
254
|
+
// pre-forward disconnect branch above, which also just releases and stops.
|
|
255
|
+
if (clientGone.signal.aborted || responseTerminated(res)) {
|
|
256
|
+
selected.release();
|
|
257
|
+
return;
|
|
258
|
+
}
|
|
259
|
+
// A rejected forward call (network failure) must produce a local 502,
|
|
260
|
+
// never an unhandled rejection. The lease releases via the response's
|
|
261
|
+
// own finish/close lifecycle once this response is sent.
|
|
262
|
+
account.errorCount++;
|
|
263
|
+
stats.totalErrors++;
|
|
264
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
265
|
+
logError(account.id, 502, `openai request failed: ${message}`);
|
|
266
|
+
recordActivity({
|
|
267
|
+
ts: startedAt,
|
|
268
|
+
accountId: account.id,
|
|
269
|
+
model: requestedModel,
|
|
270
|
+
type: "error",
|
|
271
|
+
statusCode: 502,
|
|
272
|
+
path,
|
|
273
|
+
details: "upstream_error:network",
|
|
274
|
+
durationMs: now() - startedAt,
|
|
275
|
+
});
|
|
276
|
+
res.status(502).json(envelope.wrap("upstream_error", `OpenAI request failed: ${message}`));
|
|
235
277
|
return;
|
|
236
278
|
}
|
|
237
|
-
//
|
|
238
|
-
//
|
|
239
|
-
|
|
240
|
-
|
|
279
|
+
// Cooldown/eligibility react to the raw upstream signal — this must not
|
|
280
|
+
// change based on how the relay later renders the response to the client.
|
|
281
|
+
upstreamFailed = upstream.status === 401 || upstream.status === 429 || upstream.status >= 500;
|
|
282
|
+
details = routeReasonDetails(selected.route);
|
|
283
|
+
// Tracks whether `account.errorCount`/`consecutiveErrors` were already
|
|
284
|
+
// incremented for this request by the upstream-classification branch below,
|
|
285
|
+
// so a relay-synthesized failure (e.g. a byte-transparent stream that
|
|
286
|
+
// observed an upstream `response.failed`/`error` event on an otherwise-200
|
|
287
|
+
// response) can still increment them once further down without double
|
|
288
|
+
// counting an upstream 401/429/5xx that already did.
|
|
289
|
+
accountFailureCounted = false;
|
|
290
|
+
try {
|
|
291
|
+
// Header/rate-limit parsing and cooldown bookkeeping run on live upstream
|
|
292
|
+
// data between the two request-level try/catches above — a throw here
|
|
293
|
+
// (e.g. an unreadable header) must degrade to "skip this bookkeeping",
|
|
294
|
+
// never crash the daemon or leave the relay below un-reached.
|
|
295
|
+
const headerRecord = headersToRecord(upstream.headers);
|
|
296
|
+
applyCodexRateLimits(account, parseCodexRateLimits(headerRecord, now()), now());
|
|
297
|
+
if (upstreamFailed) {
|
|
298
|
+
account.errorCount++;
|
|
299
|
+
account.consecutiveErrors++;
|
|
300
|
+
accountFailureCounted = true;
|
|
301
|
+
const applied = applyCodexFailureRouting(upstream.status, headerRecord, selected.route, requestedModel, openAIRouter, openAIPool, now);
|
|
302
|
+
details = routeFailureDetails(selected.route, upstream.status === 401 ? "token-invalid"
|
|
303
|
+
: upstream.status === 429 ? "rate-limited"
|
|
304
|
+
// Only 503/529 are treated as upstream overload for cooldown
|
|
305
|
+
// purposes; labelling an isolated 500/502/504 "service-overloaded"
|
|
306
|
+
// would contradict the routing decision actually taken.
|
|
307
|
+
: upstream.status === 503 || upstream.status === 529 ? "service-overloaded"
|
|
308
|
+
: "upstream-error", applied.limitingScope);
|
|
309
|
+
if (upstream.status === 401)
|
|
310
|
+
onUpstreamAuthFailure?.(account);
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
catch (error) {
|
|
314
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
315
|
+
logError(account.id, upstream.status, `openai response classification failed: ${message}`);
|
|
316
|
+
}
|
|
317
|
+
// Retryable statuses (429 || >= 500) are a strict subset of
|
|
318
|
+
// `upstreamFailed`, so this predicate alone decides the loop.
|
|
319
|
+
if (!isRetryableUpstreamStatus(upstream.status)
|
|
320
|
+
|| attempt >= maxAttempts || clientGone.signal.aborted) {
|
|
321
|
+
break;
|
|
322
|
+
}
|
|
323
|
+
// Router-side failover/retry. The failure bookkeeping above already ran
|
|
324
|
+
// (cooldown, affinity break, error counters), and not a single response
|
|
325
|
+
// byte has been relayed, so the request can move to whichever account the
|
|
326
|
+
// pool would hand a brand-new request: a different one after a 429 or
|
|
327
|
+
// 503/529 cooldown, the same one after a plain 5xx. Everything is decided
|
|
328
|
+
// BEFORE the held failure response is abandoned — any dead end below
|
|
329
|
+
// still relays the original upstream failure unchanged.
|
|
330
|
+
let next;
|
|
331
|
+
try {
|
|
332
|
+
next = acquireRequestRoute(sessionKey, res, openAIRouter, { requestedModel });
|
|
333
|
+
}
|
|
334
|
+
catch (error) {
|
|
335
|
+
// Nothing eligible to fail over to — pass the failure through. Only
|
|
336
|
+
// routing-level rejections are expected here; anything else is a bug
|
|
337
|
+
// worth a log line, though pass-through stays the safe outcome.
|
|
338
|
+
if (!(error instanceof NoEligibleAccountError) && !(error instanceof EmptyPoolError)) {
|
|
339
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
340
|
+
logError("proxy", 500, `unexpected routing failure during retry: ${message}`);
|
|
341
|
+
}
|
|
342
|
+
break;
|
|
343
|
+
}
|
|
344
|
+
if (upstream.status === 429 && next.route.account.id === account.id) {
|
|
345
|
+
// Re-sending a 429 to the account that produced it would only
|
|
346
|
+
// reproduce the rate limit. The cooldown normally guarantees a
|
|
347
|
+
// different account here; if it ever does not, pass through instead.
|
|
348
|
+
next.release();
|
|
349
|
+
break;
|
|
350
|
+
}
|
|
351
|
+
// Same bound as the Anthropic route: the held failure is ready to relay
|
|
352
|
+
// and the refresh fetch has no deadline of its own, so an unbounded wait
|
|
353
|
+
// here could withhold it for minutes. The refresh is not cancelled — a
|
|
354
|
+
// late outcome still runs prepareRoute's own bookkeeping in the
|
|
355
|
+
// background and readies (or cools down) the account for later requests.
|
|
356
|
+
const prepared = await boundedWait(prepareRoute(next), retryRefreshTimeoutMs, "still-pending", clientGone.signal);
|
|
357
|
+
if (prepared === "still-pending") {
|
|
358
|
+
if (!clientGone.signal.aborted) {
|
|
359
|
+
logError(next.route.account.id, 0, `failover token refresh still pending after ${retryRefreshTimeoutMs}ms — relaying held upstream failure`);
|
|
360
|
+
}
|
|
361
|
+
next.release();
|
|
362
|
+
break;
|
|
363
|
+
}
|
|
364
|
+
if (!prepared)
|
|
365
|
+
break;
|
|
366
|
+
// Committed: record the failed attempt and abandon its response.
|
|
241
367
|
stats.totalErrors++;
|
|
242
|
-
const message = error instanceof Error ? error.message : String(error);
|
|
243
|
-
logError(account.id, 502, `openai request failed: ${message}`);
|
|
244
368
|
recordActivity({
|
|
245
|
-
ts:
|
|
369
|
+
ts: attemptStartedAt,
|
|
246
370
|
accountId: account.id,
|
|
247
371
|
model: requestedModel,
|
|
248
372
|
type: "error",
|
|
249
|
-
statusCode:
|
|
373
|
+
statusCode: upstream.status,
|
|
250
374
|
path,
|
|
251
|
-
|
|
252
|
-
|
|
375
|
+
...(opts.method !== undefined ? { method: opts.method } : {}),
|
|
376
|
+
...(opts.source !== undefined ? { source: opts.source } : {}),
|
|
377
|
+
details: `${details}:will-retry`,
|
|
378
|
+
durationMs: now() - attemptStartedAt,
|
|
253
379
|
});
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
// response) can still increment them once further down without double
|
|
266
|
-
// counting an upstream 401/429/5xx that already did.
|
|
267
|
-
let accountFailureCounted = false;
|
|
268
|
-
try {
|
|
269
|
-
// Header/rate-limit parsing and cooldown bookkeeping run on live upstream
|
|
270
|
-
// data between the two request-level try/catches above — a throw here
|
|
271
|
-
// (e.g. an unreadable header) must degrade to "skip this bookkeeping",
|
|
272
|
-
// never crash the daemon or leave the relay below un-reached.
|
|
273
|
-
const headerRecord = headersToRecord(upstream.headers);
|
|
274
|
-
applyCodexRateLimits(account, parseCodexRateLimits(headerRecord, now()), now());
|
|
275
|
-
if (upstreamFailed) {
|
|
276
|
-
account.errorCount++;
|
|
277
|
-
account.consecutiveErrors++;
|
|
278
|
-
accountFailureCounted = true;
|
|
279
|
-
const applied = applyCodexFailureRouting(upstream.status, headerRecord, selected.route, requestedModel, openAIRouter, openAIPool, now);
|
|
280
|
-
details = routeFailureDetails(selected.route, upstream.status === 401 ? "token-invalid"
|
|
281
|
-
: upstream.status === 429 ? "rate-limited"
|
|
282
|
-
// Only 503/529 are treated as upstream overload for cooldown
|
|
283
|
-
// purposes; labelling an isolated 500/502/504 "service-overloaded"
|
|
284
|
-
// would contradict the routing decision actually taken.
|
|
285
|
-
: upstream.status === 503 || upstream.status === 529 ? "service-overloaded"
|
|
286
|
-
: "upstream-error", applied.limitingScope);
|
|
287
|
-
if (upstream.status === 401)
|
|
288
|
-
onUpstreamAuthFailure?.(account);
|
|
380
|
+
void upstream.body?.cancel().catch(() => { });
|
|
381
|
+
selected.release();
|
|
382
|
+
const sameAccount = next.route.account.id === account.id;
|
|
383
|
+
selected = next;
|
|
384
|
+
// An immediate same-account replay would hit whatever transient condition
|
|
385
|
+
// produced the 5xx still in progress; a failover needs no pause.
|
|
386
|
+
if (sameAccount)
|
|
387
|
+
await retryDelay(sameAccountDelayMs, clientGone.signal);
|
|
388
|
+
if (clientGone.signal.aborted || responseTerminated(res)) {
|
|
389
|
+
selected.release();
|
|
390
|
+
return;
|
|
289
391
|
}
|
|
290
392
|
}
|
|
291
|
-
|
|
292
|
-
const message = error instanceof Error ? error.message : String(error);
|
|
293
|
-
logError(account.id, upstream.status, `openai response classification failed: ${message}`);
|
|
294
|
-
}
|
|
393
|
+
const account = selected.route.account;
|
|
295
394
|
const entry = {
|
|
296
395
|
ts: startedAt,
|
|
297
396
|
accountId: account.id,
|
|
@@ -136,6 +136,13 @@ export function mountResponsesRoutes(app, opts) {
|
|
|
136
136
|
now,
|
|
137
137
|
envelope: RESPONSES_ENVELOPE,
|
|
138
138
|
onUpstreamAuthFailure: opts.onUpstreamAuthFailure,
|
|
139
|
+
...(opts.maxAttempts !== undefined ? { maxAttempts: opts.maxAttempts } : {}),
|
|
140
|
+
...(opts.sameAccountRetryDelayMs !== undefined
|
|
141
|
+
? { sameAccountRetryDelayMs: opts.sameAccountRetryDelayMs }
|
|
142
|
+
: {}),
|
|
143
|
+
...(opts.retryRefreshTimeoutMs !== undefined
|
|
144
|
+
? { retryRefreshTimeoutMs: opts.retryRefreshTimeoutMs }
|
|
145
|
+
: {}),
|
|
139
146
|
relay: async (upstream, res, entry, report) => {
|
|
140
147
|
if (body.stream === true) {
|
|
141
148
|
const observer = createCodexUsageObserver();
|
package/dist/proxy/server.js
CHANGED
|
@@ -4,12 +4,14 @@ import { ServerResponse } from "http";
|
|
|
4
4
|
import { timingSafeEqual } from "crypto";
|
|
5
5
|
import { TokenPool } from "./token-pool.js";
|
|
6
6
|
import { needsRefresh, refreshAccountIfCurrent, saveAccounts, startRefreshLoop } from "./token-refresher.js";
|
|
7
|
-
import { loadAccounts, loadOpenAIAccounts, saveOpenAIAccountsToPath, accountsFileExists, readAccountsFromPath, readConfig, writeConfig, getProxyRequestTimeoutMs, migrateLegacyAccountProviders, setProviderAccountsEnabled } from "../config/manager.js";
|
|
7
|
+
import { loadAccounts, loadOpenAIAccounts, saveOpenAIAccountsToPath, accountsFileExists, readAccountsFromPath, readConfig, writeConfig, getAutoFailoverEnabled, getProxyRequestTimeoutMs, migrateLegacyAccountProviders, setProviderAccountsEnabled } from "../config/manager.js";
|
|
8
8
|
import { checkForUpdate, performUpdate, restartSelf, printUpdateBanner, getCurrentVersion } from "../utils/self-update.js";
|
|
9
9
|
import { trackEvent, startHeartbeat } from "../utils/telemetry.js";
|
|
10
10
|
import { loadTelemetryState } from "../config/telemetry.js";
|
|
11
11
|
import { logRoute, logError, logStartup } from "./logger.js";
|
|
12
12
|
import { createLocalRoutingErrorLog, stats } from "./stats.js";
|
|
13
|
+
import { applyRateLimitHeaders } from "../providers/anthropic/rate-limit-headers.js";
|
|
14
|
+
import { mountAnthropicMessagesRoute, withOAuthBeta } from "./anthropic-messages-route.js";
|
|
13
15
|
import { PROXY_PORT, LITELLM_URL, ACCOUNTS_PATH } from "../config/paths.js";
|
|
14
16
|
import { writePid, removePid, managesPidFile } from "../daemon/pid.js";
|
|
15
17
|
import { applyOpenAIAccountPatch, validateAccountPatchBody } from "./account-patch.js";
|
|
@@ -26,14 +28,13 @@ import { SessionRouter } from "./session-router.js";
|
|
|
26
28
|
import { createAnthropicProxy } from "./anthropic-proxy.js";
|
|
27
29
|
import { AnthropicUsageRefresher } from "../providers/anthropic/usage-refresher.js";
|
|
28
30
|
import { OpenAIUsageRefresher } from "../providers/openai/usage-fetch.js";
|
|
29
|
-
import {
|
|
31
|
+
import { attachAnthropicResponseCapture } from "./anthropic-response-capture.js";
|
|
30
32
|
import { canUseExtraUsage } from "../providers/anthropic/usage.js";
|
|
31
33
|
import { applyUpstreamFailureRoutingDetailed, reconcileAmbiguousRateLimitCooldown, routeFailureDetails, routeReasonDetails, } from "./lease-lifecycle.js";
|
|
32
34
|
import { persistProviderEnabledState } from "./provider-routing.js";
|
|
33
35
|
import { accountDeletionStatusCode, deleteAnthropicAccountTransaction, deleteOpenAIAccountTransaction, } from "./account-deletion.js";
|
|
34
36
|
import { addOpenAIAccountTransaction } from "./account-add.js";
|
|
35
37
|
import { createAnthropicRefreshMiddleware, createAnthropicRoutingMiddleware, } from "./anthropic-routing.js";
|
|
36
|
-
import { createStreamLifecycleTracker } from "./stream-lifecycle.js";
|
|
37
38
|
const zeroRoutingMetrics = () => ({
|
|
38
39
|
inFlightRequests: 0,
|
|
39
40
|
activeSessions: 0,
|
|
@@ -139,6 +140,8 @@ function publicUsageSnapshot(usage) {
|
|
|
139
140
|
fetchStatus: usage.fetchStatus,
|
|
140
141
|
};
|
|
141
142
|
}
|
|
143
|
+
// An unreported utilization surfaces as 0 here, which is what the dashboard
|
|
144
|
+
// has always shown for it; only release decisions need the distinction.
|
|
142
145
|
function publicWindow(window) {
|
|
143
146
|
return { utilization: publicUtilization(window.utilization), resetAt: publicTimestamp(window.resetAt) };
|
|
144
147
|
}
|
|
@@ -277,57 +280,9 @@ function providerStatus(accounts) {
|
|
|
277
280
|
enabled: accounts.filter(a => a.enabled !== false).length,
|
|
278
281
|
};
|
|
279
282
|
}
|
|
280
|
-
//
|
|
281
|
-
//
|
|
282
|
-
|
|
283
|
-
function applyInputUsage(entry, usage) {
|
|
284
|
-
entry.cacheReadTokens = usage["cache_read_input_tokens"] ?? 0;
|
|
285
|
-
entry.cacheCreationTokens = usage["cache_creation_input_tokens"] ?? 0;
|
|
286
|
-
entry.inputTokens = usage["input_tokens"] ?? 0;
|
|
287
|
-
stats.totalCacheReadTokens += entry.cacheReadTokens;
|
|
288
|
-
stats.totalCacheCreationTokens += entry.cacheCreationTokens;
|
|
289
|
-
stats.totalInputTokens += entry.inputTokens;
|
|
290
|
-
}
|
|
291
|
-
function applyOutputUsage(entry, usage) {
|
|
292
|
-
entry.outputTokens = usage["output_tokens"] ?? 0;
|
|
293
|
-
stats.totalOutputTokens += entry.outputTokens;
|
|
294
|
-
}
|
|
295
|
-
// ─── Rate limit header extraction ──────────────────────────────────────────
|
|
296
|
-
function inferPlan(requestsLimit) {
|
|
297
|
-
if (requestsLimit <= 0)
|
|
298
|
-
return "";
|
|
299
|
-
if (requestsLimit <= 100)
|
|
300
|
-
return "Pro";
|
|
301
|
-
if (requestsLimit <= 500)
|
|
302
|
-
return "Max 5x";
|
|
303
|
-
return "Max 20x";
|
|
304
|
-
}
|
|
305
|
-
function extractRateLimits(headers) {
|
|
306
|
-
const h = (name) => String(headers[name] ?? "");
|
|
307
|
-
const status = h("anthropic-ratelimit-unified-status");
|
|
308
|
-
if (!status)
|
|
309
|
-
return null; // No unified headers in this response
|
|
310
|
-
const requestsLimit = parseInt(h("anthropic-ratelimit-requests-limit"), 10) || 0;
|
|
311
|
-
return {
|
|
312
|
-
status: status === "rate_limited" ? "rate_limited" : "allowed",
|
|
313
|
-
fiveHourUtil: parseFloat(h("anthropic-ratelimit-unified-5h-utilization")) || 0,
|
|
314
|
-
fiveHourReset: parseInt(h("anthropic-ratelimit-unified-5h-reset"), 10) || 0,
|
|
315
|
-
sevenDayUtil: parseFloat(h("anthropic-ratelimit-unified-7d-utilization")) || 0,
|
|
316
|
-
sevenDayReset: parseInt(h("anthropic-ratelimit-unified-7d-reset"), 10) || 0,
|
|
317
|
-
claim: h("anthropic-ratelimit-unified-representative-claim"),
|
|
318
|
-
plan: inferPlan(requestsLimit),
|
|
319
|
-
requestsLimit,
|
|
320
|
-
lastUpdated: Date.now(),
|
|
321
|
-
};
|
|
322
|
-
}
|
|
323
|
-
/** Apply upstream rate-limit headers without discarding the usage snapshot. */
|
|
324
|
-
export function applyRateLimitHeaders(account, headers) {
|
|
325
|
-
const rateLimits = extractRateLimits(headers);
|
|
326
|
-
if (!rateLimits)
|
|
327
|
-
return false;
|
|
328
|
-
account.rateLimits = { ...account.rateLimits, ...rateLimits };
|
|
329
|
-
return true;
|
|
330
|
-
}
|
|
283
|
+
// Re-exported so existing importers keep working; the implementation moved to
|
|
284
|
+
// providers/anthropic so both Anthropic transports share it.
|
|
285
|
+
export { applyRateLimitHeaders } from "../providers/anthropic/rate-limit-headers.js";
|
|
331
286
|
/**
|
|
332
287
|
* Build the single function through which this server writes OpenAI accounts.
|
|
333
288
|
*
|
|
@@ -442,6 +397,13 @@ export async function startServer(opts = {}) {
|
|
|
442
397
|
openAIUsageRefresher.start();
|
|
443
398
|
const app = express();
|
|
444
399
|
const proxyRequestTimeoutMs = getProxyRequestTimeoutMs();
|
|
400
|
+
// Router-side 429 failover / 5xx retry is on by default; `"autoFailover":
|
|
401
|
+
// false` in config.json opts out for anyone who cannot work with the
|
|
402
|
+
// trade-off (a committed retry abandons the original failure response).
|
|
403
|
+
// A single-attempt budget IS the off switch: both transports then relay
|
|
404
|
+
// every upstream failure unchanged, exactly as before the feature existed.
|
|
405
|
+
const autoFailover = getAutoFailoverEnabled();
|
|
406
|
+
const upstreamAttempts = autoFailover ? {} : { maxAttempts: 1 };
|
|
445
407
|
// ─── Proxy auth middleware ─────────────────────────────────────────────────
|
|
446
408
|
// If a proxySecret is configured, all requests must present it as EITHER
|
|
447
409
|
// "Authorization: Bearer <secret>" (Claude Code CLI, HTTP clients)
|
|
@@ -923,6 +885,7 @@ export async function startServer(opts = {}) {
|
|
|
923
885
|
prepareOpenAIAccount: (account) => prepareOpenAIAccountForRequest(account, openAIAccounts, persistOpenAIAccounts),
|
|
924
886
|
modelRouting,
|
|
925
887
|
onUpstreamAuthFailure: onOpenAIUpstreamAuthFailure,
|
|
888
|
+
...upstreamAttempts,
|
|
926
889
|
});
|
|
927
890
|
mountMessagesCrossProviderRoute(app, {
|
|
928
891
|
openAIRouter,
|
|
@@ -930,6 +893,58 @@ export async function startServer(opts = {}) {
|
|
|
930
893
|
prepareOpenAIAccount: (account) => prepareOpenAIAccountForRequest(account, openAIAccounts, persistOpenAIAccounts),
|
|
931
894
|
modelRouting,
|
|
932
895
|
onUpstreamAuthFailure: onOpenAIUpstreamAuthFailure,
|
|
896
|
+
...upstreamAttempts,
|
|
897
|
+
});
|
|
898
|
+
// Shared between the retrying /v1/messages route and the generic /v1 chain
|
|
899
|
+
// so a locally rejected request is reported identically on both.
|
|
900
|
+
const onAnthropicEmptyPool = (err, _req, res) => {
|
|
901
|
+
stats.totalErrors++;
|
|
902
|
+
logError("proxy", 503, err.message);
|
|
903
|
+
res.status(503).json({
|
|
904
|
+
type: "error",
|
|
905
|
+
error: { type: "no_accounts", message: err.message },
|
|
906
|
+
});
|
|
907
|
+
};
|
|
908
|
+
const onAnthropicNoEligibleAccount = (err, req) => {
|
|
909
|
+
stats.totalErrors++;
|
|
910
|
+
const entry = createLocalRoutingErrorLog(err.reason, req._ccRouteContext?.modelFamily);
|
|
911
|
+
stats.addLog(entry);
|
|
912
|
+
logError(entry.accountId, entry.statusCode ?? 0, entry.details ?? "no-eligible");
|
|
913
|
+
};
|
|
914
|
+
const onAnthropicRefreshFailure = (account) => {
|
|
915
|
+
stats.totalErrors++;
|
|
916
|
+
logError(account.id, 401, "Token refresh failed");
|
|
917
|
+
};
|
|
918
|
+
// Claude-bound POST /v1/messages goes through its own transport with
|
|
919
|
+
// router-side 429 failover and 5xx retry; every other /v1 endpoint stays on
|
|
920
|
+
// the generic byte-transparent proxy below.
|
|
921
|
+
mountAnthropicMessagesRoute(app, {
|
|
922
|
+
target,
|
|
923
|
+
timeoutMs: proxyRequestTimeoutMs,
|
|
924
|
+
pool,
|
|
925
|
+
sessionRouter,
|
|
926
|
+
...upstreamAttempts,
|
|
927
|
+
needsRefresh,
|
|
928
|
+
refresh: account => refreshAccountIfCurrent(account, pool),
|
|
929
|
+
onRefreshFailure: onAnthropicRefreshFailure,
|
|
930
|
+
onEmptyPool: onAnthropicEmptyPool,
|
|
931
|
+
onNoEligibleAccount: onAnthropicNoEligibleAccount,
|
|
932
|
+
// A relayed 401 means the token is stale — refresh in the background so
|
|
933
|
+
// the next request succeeds without making this client wait on it.
|
|
934
|
+
onUpstream401: account => {
|
|
935
|
+
void refreshAccountIfCurrent(account, pool).catch(console.error);
|
|
936
|
+
},
|
|
937
|
+
// Refresh in the background to narrow only ambiguity-owned global state
|
|
938
|
+
// when fresh usage proves a requested-model exhaustion.
|
|
939
|
+
onRateLimited: (route, ambiguousCooldownToken) => {
|
|
940
|
+
queueMicrotask(() => {
|
|
941
|
+
void usageRefresher.refreshAfterCurrent(route.account).then(result => {
|
|
942
|
+
if (result.ok) {
|
|
943
|
+
reconcileAmbiguousRateLimitCooldown(route, pool, ambiguousCooldownToken);
|
|
944
|
+
}
|
|
945
|
+
});
|
|
946
|
+
});
|
|
947
|
+
},
|
|
933
948
|
});
|
|
934
949
|
// ─── Proxy middleware ──────────────────────────────────────────────────────
|
|
935
950
|
// IMPORTANT: selfHandleResponse must be false (default) for SSE streaming to
|
|
@@ -952,15 +967,9 @@ export async function startServer(opts = {}) {
|
|
|
952
967
|
// CRITICAL: api.anthropic.com requires the "oauth-2025-04-20" beta flag to
|
|
953
968
|
// accept OAuth tokens (sk-ant-oat01-*). Without it the request is rejected
|
|
954
969
|
// with "OAuth authentication is currently not supported."
|
|
955
|
-
// APPEND — do NOT replace — so existing betas (tools, computer-use, etc.)
|
|
956
|
-
|
|
957
|
-
|
|
958
|
-
? String(existingBeta).split(",").map(b => b.trim()).filter(Boolean)
|
|
959
|
-
: [];
|
|
960
|
-
if (!betas.includes("oauth-2025-04-20")) {
|
|
961
|
-
betas.push("oauth-2025-04-20");
|
|
962
|
-
proxyReq.setHeader("anthropic-beta", betas.join(","));
|
|
963
|
-
}
|
|
970
|
+
// APPEND — do NOT replace — so existing betas (tools, computer-use, etc.)
|
|
971
|
+
// are preserved. Shared with the retrying /v1/messages transport.
|
|
972
|
+
proxyReq.setHeader("anthropic-beta", withOAuthBeta(proxyReq.getHeader("anthropic-beta")));
|
|
964
973
|
// All other headers are forwarded automatically by http-proxy-middleware:
|
|
965
974
|
// anthropic-version — required by Anthropic API
|
|
966
975
|
// X-Claude-Code-Session-Id — session aggregation header sent by Claude Code
|
|
@@ -1061,34 +1070,9 @@ export async function startServer(opts = {}) {
|
|
|
1061
1070
|
const entry = pendingLog;
|
|
1062
1071
|
stats.addLog(entry);
|
|
1063
1072
|
// ── Capture token usage from Anthropic response body ─────────────────
|
|
1064
|
-
//
|
|
1065
|
-
//
|
|
1066
|
-
|
|
1067
|
-
// Non-streaming JSON carries all fields in a single usage object.
|
|
1068
|
-
// The proxy is byte-transparent and the client's accept-encoding makes
|
|
1069
|
-
// upstream compress, so the capture decompresses its own copy of the
|
|
1070
|
-
// stream (see usage-capture.ts) — previously compressed responses were
|
|
1071
|
-
// skipped, which in practice was EVERY response: no cache rate or
|
|
1072
|
-
// token counts ever appeared on Anthropic activity rows.
|
|
1073
|
-
const contentType = String(proxyRes.headers["content-type"] ?? "");
|
|
1074
|
-
const encoding = String(proxyRes.headers["content-encoding"] ?? "");
|
|
1075
|
-
const isCompressed = /gzip|br|deflate/.test(encoding);
|
|
1076
|
-
const streamTracker = createStreamLifecycleTracker(req._startTime ?? Date.now(), !isCompressed && contentType.includes("text/event-stream"));
|
|
1077
|
-
entry.streamLifecycle = streamTracker.state;
|
|
1078
|
-
streamTracker.attach(proxyRes, response);
|
|
1079
|
-
proxyRes.on("data", (chunk) => streamTracker.observeChunk(chunk));
|
|
1080
|
-
const usageCapture = createAnthropicUsageCapture({
|
|
1081
|
-
contentType,
|
|
1082
|
-
contentEncoding: encoding,
|
|
1083
|
-
// Mutates the already-logged entry in place; the dashboard picks the
|
|
1084
|
-
// values up on its next poll.
|
|
1085
|
-
onInputUsage: (usage) => applyInputUsage(entry, usage),
|
|
1086
|
-
onOutputUsage: (usage) => applyOutputUsage(entry, usage),
|
|
1087
|
-
});
|
|
1088
|
-
if (usageCapture) {
|
|
1089
|
-
proxyRes.on("data", (chunk) => usageCapture.write(chunk));
|
|
1090
|
-
proxyRes.on("end", () => usageCapture.end());
|
|
1091
|
-
}
|
|
1073
|
+
// Passive stream-lifecycle + token-usage taps, shared with the
|
|
1074
|
+
// retrying /v1/messages transport (see anthropic-response-capture.ts).
|
|
1075
|
+
attachAnthropicResponseCapture(proxyRes, response, entry, req._startTime ?? Date.now());
|
|
1092
1076
|
},
|
|
1093
1077
|
error: (err, _req, res) => {
|
|
1094
1078
|
const request = _req;
|
|
@@ -1125,27 +1109,12 @@ export async function startServer(opts = {}) {
|
|
|
1125
1109
|
// and breaks SSE streaming passthrough.
|
|
1126
1110
|
app.use("/v1", createAnthropicRoutingMiddleware({
|
|
1127
1111
|
sessionRouter,
|
|
1128
|
-
onEmptyPool:
|
|
1129
|
-
|
|
1130
|
-
logError("proxy", 503, err.message);
|
|
1131
|
-
res.status(503).json({
|
|
1132
|
-
type: "error",
|
|
1133
|
-
error: { type: "no_accounts", message: err.message },
|
|
1134
|
-
});
|
|
1135
|
-
},
|
|
1136
|
-
onNoEligibleAccount: (err, req) => {
|
|
1137
|
-
stats.totalErrors++;
|
|
1138
|
-
const entry = createLocalRoutingErrorLog(err.reason, req._ccRouteContext?.modelFamily);
|
|
1139
|
-
stats.addLog(entry);
|
|
1140
|
-
logError(entry.accountId, entry.statusCode ?? 0, entry.details ?? "no-eligible");
|
|
1141
|
-
},
|
|
1112
|
+
onEmptyPool: onAnthropicEmptyPool,
|
|
1113
|
+
onNoEligibleAccount: onAnthropicNoEligibleAccount,
|
|
1142
1114
|
}), createAnthropicRefreshMiddleware({
|
|
1143
1115
|
needsRefresh,
|
|
1144
1116
|
refresh: account => refreshAccountIfCurrent(account, pool),
|
|
1145
|
-
onRefreshFailure:
|
|
1146
|
-
stats.totalErrors++;
|
|
1147
|
-
logError(account.id, 401, "Token refresh failed");
|
|
1148
|
-
},
|
|
1117
|
+
onRefreshFailure: onAnthropicRefreshFailure,
|
|
1149
1118
|
}), (req, _res, next) => {
|
|
1150
1119
|
const route = req._ccRoute;
|
|
1151
1120
|
const account = route.account;
|
|
@@ -1258,6 +1227,9 @@ export async function startServer(opts = {}) {
|
|
|
1258
1227
|
console.log(autoUpdate
|
|
1259
1228
|
? chalk.gray(" Auto-update: enabled (patch/minor)")
|
|
1260
1229
|
: chalk.gray(" Auto-update: off (notify-only) — run 'cc-router update' to install"));
|
|
1230
|
+
console.log(autoFailover
|
|
1231
|
+
? chalk.gray(" Auto-failover: on — 429/5xx retried across accounts before the first relayed byte")
|
|
1232
|
+
: chalk.gray(" Auto-failover: off — upstream failures pass through; clients own retries"));
|
|
1261
1233
|
// Anonymous telemetry — fire-and-forget, never blocks proxy startup.
|
|
1262
1234
|
try {
|
|
1263
1235
|
const telemetryState = loadTelemetryState();
|