@rikcodes/teamclaude 1.1.20-rik.10 → 1.1.20-rik.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/server.js +71 -4
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@rikcodes/teamclaude",
|
|
3
|
-
"version": "1.1.20-rik.
|
|
3
|
+
"version": "1.1.20-rik.11",
|
|
4
4
|
"description": "Multi-account proxy for Claude Code and Codex: pools Claude Max, ChatGPT/Codex, API-key and third-party backend accounts, and rotates on quota",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "src/index.js",
|
package/src/server.js
CHANGED
|
@@ -61,6 +61,29 @@ const INLINE_RETRY_AFTER_MAX_SECONDS = 15;
|
|
|
61
61
|
// the account so concurrent requests wait, then retries the same account.
|
|
62
62
|
const RATE_LIMIT_ABSORB_MAX_SECONDS =
|
|
63
63
|
Number(process.env.TEAMCLAUDE_RATE_LIMIT_ABSORB_MAX_SECONDS) || 60;
|
|
64
|
+
// How long to wait before the one retry of a headerless 429 — a 429 carrying no
|
|
65
|
+
// retry-after and no anthropic-ratelimit-* headers at all.
|
|
66
|
+
//
|
|
67
|
+
// Measured over a 32-minute window: these arrive in 0.6-0.8s, about once every
|
|
68
|
+
// 8 minutes on Fable traffic and never on any other model, and they follow the
|
|
69
|
+
// request onto whichever account the failover hop moves it to. That hop re-asks
|
|
70
|
+
// roughly 0.7s after the first refusal and is refused again, which is direct
|
|
71
|
+
// evidence that a wait shorter than that buys nothing but a third identical
|
|
72
|
+
// refusal. The ceiling is what the client does instead: Claude Code shows
|
|
73
|
+
// "will retry in 2m 38s" and then usually succeeds, so any wait measured in
|
|
74
|
+
// seconds trades a visible stall for an invisible one. 2s clears the interval
|
|
75
|
+
// already known to fail while keeping the worst case — the retry is refused too
|
|
76
|
+
// and the client gets its 429 anyway, just later — at about 4s.
|
|
77
|
+
//
|
|
78
|
+
// One delay, not a ladder: the limit's window is unknown, and a second guess at
|
|
79
|
+
// it would cost the client the wait without evidence that it helps. Override
|
|
80
|
+
// with TEAMCLAUDE_HEADERLESS_429_RETRY_DELAY_MS.
|
|
81
|
+
const DEFAULT_HEADERLESS_429_RETRY_DELAY_MS = 2000;
|
|
82
|
+
|
|
83
|
+
function resolveHeaderless429RetryDelayMs() {
|
|
84
|
+
const env = Number(process.env.TEAMCLAUDE_HEADERLESS_429_RETRY_DELAY_MS);
|
|
85
|
+
return env > 0 ? env : DEFAULT_HEADERLESS_429_RETRY_DELAY_MS;
|
|
86
|
+
}
|
|
64
87
|
const OAUTH_ENTITLEMENT_ERROR_CODE = 'oauth_not_allowed_for_organization';
|
|
65
88
|
const ERROR_BODY_INSPECTION_LIMIT = 64 * 1024;
|
|
66
89
|
// How long an idle keep-alive connection is held open.
|
|
@@ -2419,7 +2442,45 @@ export async function forwardRequest(req, res, body, accountManager, upstream, r
|
|
|
2419
2442
|
} else if (ctx.rateLimitHopped && requestScoped) {
|
|
2420
2443
|
// Second headerless 429, on a different account: it followed the
|
|
2421
2444
|
// request. Nothing here is about either account.
|
|
2422
|
-
|
|
2445
|
+
//
|
|
2446
|
+
// That does not make it permanent. Measured: these land about once every
|
|
2447
|
+
// 8 minutes on Fable traffic, from four different starting accounts, and
|
|
2448
|
+
// the client's own retry usually succeeds — so most are a transient the
|
|
2449
|
+
// fleet cannot route around, not a model id upstream refuses. Returning
|
|
2450
|
+
// it straight away is what left Claude Code sitting on "will retry in
|
|
2451
|
+
// 2m 38s", with nothing in the session transcript to explain the pause.
|
|
2452
|
+
//
|
|
2453
|
+
// So take one short retry first, on THIS account. Not on a third one:
|
|
2454
|
+
// the limit is scoped to neither account, so another hop would pay a
|
|
2455
|
+
// cold prompt cache to learn what the first hop already established —
|
|
2456
|
+
// the same argument that bounds the hop budget above. ctx.hopTo keeps
|
|
2457
|
+
// the attempt on the account it landed on and off the fleet cursor
|
|
2458
|
+
// (#286), and `route` keeps its egress, since the wait is the only
|
|
2459
|
+
// variable being tested. A fresh IP is a different hypothesis and the sx
|
|
2460
|
+
// retry below still owns it: when it is armed it goes first, for free,
|
|
2461
|
+
// and this retry takes the attempt after it.
|
|
2462
|
+
//
|
|
2463
|
+
// Only for a caller that actually waits. A Codex one does not: it gives
|
|
2464
|
+
// the response head 60s, then retries the whole request itself (see
|
|
2465
|
+
// holdsConnection, and the incident recorded in
|
|
2466
|
+
// test/codex-no-inline-hold.test.js). Holding it here would stack our
|
|
2467
|
+
// wait underneath its own, which is the trade that file exists to refuse.
|
|
2468
|
+
const retryDelayMs = resolveHeaderless429RetryDelayMs();
|
|
2469
|
+
const callerWaits = holdsConnection(ctx.provider);
|
|
2470
|
+
if (!ctx.headerless429Retried && !switchingToSx && retryCount < maxRetries
|
|
2471
|
+
&& !res.headersSent && !clientGone(res) && !ctx.signal?.aborted && callerWaits) {
|
|
2472
|
+
// Once per request: a retry that is refused too has made the point.
|
|
2473
|
+
ctx.headerless429Retried = true;
|
|
2474
|
+
console.log(`[TeamClaude] 429 followed the request onto "${account.name}" with no rate-limit headers — retrying it once on the same account in ${retryDelayMs}ms`
|
|
2475
|
+
+ (refusal ? ` (${safeLine(refusal)})` : ''));
|
|
2476
|
+
await waitForRetry(retryDelayMs, ctx.signal);
|
|
2477
|
+
if (clientGone(res)) { ctx.abandoned = true; return; }
|
|
2478
|
+
ctx.hopTo = account.index;
|
|
2479
|
+
return forwardRequest(req, res, body, accountManager, upstream, retryCount + 1, hooks, reqId, ctx, logDir, sx, route);
|
|
2480
|
+
}
|
|
2481
|
+
console.log(`[TeamClaude] 429 followed the request onto "${account.name}" with no rate-limit headers — `
|
|
2482
|
+
+ (callerWaits ? 'it is about the request, not the accounts' : `${ctx.provider} caller does not wait`)
|
|
2483
|
+
+ '; returning it to the client'
|
|
2423
2484
|
+ (refusal ? ` (${safeLine(refusal)})` : ''));
|
|
2424
2485
|
} else if (ctx.rateLimitHopped) {
|
|
2425
2486
|
// Second 429 this request, on a different account. Say so once: the
|
|
@@ -2445,10 +2506,16 @@ export async function forwardRequest(req, res, body, accountManager, upstream, r
|
|
|
2445
2506
|
// third time is the other half of #288. With no sibling to hop to, one
|
|
2446
2507
|
// short retry covers a momentary blip, and then it is the client's turn.
|
|
2447
2508
|
if (requestScoped) {
|
|
2448
|
-
|
|
2509
|
+
// Same rule as the post-hop retry and the inline absorb below: a wait is
|
|
2510
|
+
// only invisible to a caller that waits longer than we do.
|
|
2511
|
+
if (!ctx.rateLimitHopped && !ctx.requestScopedRetried && retryCount < maxRetries
|
|
2512
|
+
&& holdsConnection(ctx.provider)) {
|
|
2449
2513
|
ctx.requestScopedRetried = true;
|
|
2450
|
-
|
|
2451
|
-
|
|
2514
|
+
// The same number as the post-hop retry above: one phenomenon, one
|
|
2515
|
+
// delay, one env var to move both.
|
|
2516
|
+
const retryDelayMs = resolveHeaderless429RetryDelayMs();
|
|
2517
|
+
console.log(`[TeamClaude] 429 with no rate-limit headers on "${account.name}" — retrying once in ${retryDelayMs}ms${refusal ? ` (${safeLine(refusal)})` : ''}`);
|
|
2518
|
+
await waitForRetry(retryDelayMs, ctx.signal);
|
|
2452
2519
|
if (clientGone(res)) { ctx.abandoned = true; return; }
|
|
2453
2520
|
return forwardRequest(req, res, body, accountManager, upstream, retryCount + 1, hooks, reqId, ctx, logDir, sx, nextUseSx);
|
|
2454
2521
|
}
|