maxpool 1.0.7 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -3
- package/package.json +1 -1
- package/src/account-manager.js +86 -14
- package/src/config.js +12 -5
- package/src/server.js +74 -13
package/README.md
CHANGED
|
@@ -299,8 +299,11 @@ TEAMCLAUDE_CONFIG=./my-config.json maxpool server
|
|
|
299
299
|
"maxWaitMs": 86400000,
|
|
300
300
|
"autoMaxWaitMs": null,
|
|
301
301
|
"capacityMaxWaitMs": 900000,
|
|
302
|
+
"weeklyMaxWaitMs": 86400000,
|
|
303
|
+
"nonStreamMaxWaitMs": 300000,
|
|
304
|
+
"maxConcurrentQueued": 64,
|
|
305
|
+
"maxQueuedBytes": 1073741824,
|
|
302
306
|
"maxQueuedBodyBytes": 268435456,
|
|
303
|
-
"weeklyMaxWaitMs": 0,
|
|
304
307
|
"pollMs": 1000
|
|
305
308
|
},
|
|
306
309
|
"shutdown": {
|
|
@@ -338,7 +341,10 @@ TEAMCLAUDE_CONFIG=./my-config.json maxpool server
|
|
|
338
341
|
| `queue.autoMaxWaitMs` | Optional shorter auto-queue cap. Set to `null` or omit it to use `queue.maxWaitMs`; set a number for interactive sessions where you prefer fast errors |
|
|
339
342
|
| `queue.capacityMaxWaitMs` | Separate cap for repeated upstream 5xx/overload failures; defaults to 15m so broken providers do not park requests for 24h |
|
|
340
343
|
| `queue.maxQueuedBodyBytes` | Maximum request body Maxpool will hold in memory while waiting for capacity before the request has been sent upstream; defaults to 256 MiB |
|
|
341
|
-
| `queue.weeklyMaxWaitMs` |
|
|
344
|
+
| `queue.weeklyMaxWaitMs` | How long to hold a request when every account is at its weekly (7d) cap. Defaults to 24h — but the early-exit gates on each account's REAL reset time, so it only waits when a reset genuinely lands inside the window and errors honestly otherwise. (Was `0` = fail-fast, which killed sessions the instant all accounts hit their weekly cap.) |
|
|
345
|
+
| `queue.nonStreamMaxWaitMs` | Max hold for non-streaming requests; defaults to 5m. They have no SSE keepalive, so a longer hold would die on the client timeout anyway |
|
|
346
|
+
| `queue.maxConcurrentQueued` | Backpressure: max requests held waiting at once; defaults to 64. Beyond it, new waiters get a clear "queue full" error instead of growing the heap |
|
|
347
|
+
| `queue.maxQueuedBytes` | Backpressure: max aggregate buffered request-body bytes across all held requests; defaults to 1 GiB |
|
|
342
348
|
| `queue.pollMs` | How often queued requests check for a recovered account/provider |
|
|
343
349
|
| `queue.heartbeatMs` | SSE heartbeat interval for queued streaming requests; defaults to 10s so Claude Code keeps the queued connection alive |
|
|
344
350
|
| `shutdown.drainTimeoutMs` | Maximum time quit/Ctrl-C waits for active requests before exiting |
|
|
@@ -368,7 +374,7 @@ The weekly usage bar shows raw upstream utilization and reset timing. Reset-awar
|
|
|
368
374
|
11. In the `all` profile only, if all Claude accounts are unavailable, provider fallbacks are tried by priority: GLM before Kimi
|
|
369
375
|
12. If all eligible accounts/providers are temporarily unavailable for a temporary reason (5h/session limit, provider cooldown, short 429), the proxy queues the request and retries when one recovers
|
|
370
376
|
13. Repeated upstream 5xx/overload failures use the shorter `capacityMaxWaitMs` cap, not the long quota wait
|
|
371
|
-
14. Weekly exhaustion and
|
|
377
|
+
14. Weekly (7d) exhaustion is held and retried like a 5h cap, up to `weeklyMaxWaitMs`, but only when an account's real reset lands inside the window; if the soonest reset is beyond it (or unknown), it returns 429 promptly with an honest message naming the soonest reset rather than spinning. Non-retryable 4xx errors still fail fast
|
|
372
378
|
15. Temporary OAuth refresh failures cool the account down and queue/fail over; invalid refresh credentials disable only that account and require login
|
|
373
379
|
16. In an interactive terminal, the server runs under a foreground supervisor so confirmed Restart (`r`) can drain and restart without detaching the replacement TUI
|
|
374
380
|
17. When Restart is confirmed, new upstream admission pauses immediately. Existing upstream requests finish, queued requests cannot deadlock restart, and their sockets close during relaunch so Claude Code reconnects automatically
|
package/package.json
CHANGED
package/src/account-manager.js
CHANGED
|
@@ -34,14 +34,20 @@ const DEFAULT_SCHEDULER = {
|
|
|
34
34
|
weeklyCriticalThreshold: 0.95,
|
|
35
35
|
weeklyExhaustedThreshold: 0.985,
|
|
36
36
|
weeklyBurnDebtWeight: 0.6,
|
|
37
|
-
// Routing-cost tuning (lower cost = preferred).
|
|
38
|
-
//
|
|
39
|
-
//
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
37
|
+
// Routing-cost tuning (lower cost = preferred). The goal is to AVOID
|
|
38
|
+
// short-term (rate/concurrency) throttling by spreading load across healthy
|
|
39
|
+
// accounts. So in-flight concurrency is the DOMINANT term, with a steep
|
|
40
|
+
// per-account soft cap; burn-pace is only a soft de-preference (never a
|
|
41
|
+
// bench); quota "use-it-or-lose-it" is intentionally a minor signal here.
|
|
42
|
+
concurrencyWeight: 2, // multiplies in-flight load (activeWeight+reqWeight) — dominant
|
|
43
|
+
perAccountConcurrencyTarget: 3, // D: soft per-account in-flight target; past it, capPenalty bites
|
|
44
|
+
capPenaltyWeight: 10, // steep penalty per unit of in-flight depth past D (throttle safety floor)
|
|
45
|
+
paceCostWeight: 1.5, // soft de-preference of accounts burning ahead of pace (was the ×6 term)
|
|
46
|
+
scarcityWeight: 6, // legacy; superseded by paceCostWeight (kept so old configs don't error)
|
|
47
|
+
spreadShareWeight: 3, // multiplies an account's share of recent fleet load (0..1)
|
|
48
|
+
recoveryRampWeight: 4, // decaying penalty applied to a just-recovered account
|
|
49
|
+
recoveryRampMs: 5 * 60_000, // how long the post-recovery ramp lasts
|
|
50
|
+
spreadWindowMs: 15 * 60_000, // rolling window used to measure recent per-account load
|
|
45
51
|
};
|
|
46
52
|
const LOAD_EVENT_MAX_AGE_MS = 60 * 60 * 1000;
|
|
47
53
|
const WEEK_MS = 7 * 24 * 60 * 60 * 1000;
|
|
@@ -173,6 +179,7 @@ export class AccountManager {
|
|
|
173
179
|
waiting: [],
|
|
174
180
|
lastAdmissionAt: 0,
|
|
175
181
|
rampUntil: 0,
|
|
182
|
+
bytes: 0, // aggregate buffered body bytes across all held requests
|
|
176
183
|
};
|
|
177
184
|
this.admissionPaused = false;
|
|
178
185
|
}
|
|
@@ -197,6 +204,7 @@ export class AccountManager {
|
|
|
197
204
|
let soonestTemporary = Infinity;
|
|
198
205
|
let temporaryCause = null;
|
|
199
206
|
let soonestWeekly = Infinity;
|
|
207
|
+
let weeklyUnknownReset = 0; // weekly-exhausted accounts whose reset time we don't know yet
|
|
200
208
|
let matchingRoutes = 0;
|
|
201
209
|
const reasons = {};
|
|
202
210
|
|
|
@@ -244,6 +252,11 @@ export class AccountManager {
|
|
|
244
252
|
} else if (retry.cause === 'weekly_exhausted' && retry.retryAt) {
|
|
245
253
|
const ms = retry.retryAt - Date.now();
|
|
246
254
|
if (ms < soonestWeekly) soonestWeekly = ms;
|
|
255
|
+
} else if (retry.cause === 'weekly_exhausted' && !retry.retryAt) {
|
|
256
|
+
// Weekly-capped but we haven't learned the reset time (cold start /
|
|
257
|
+
// probe failure). We cannot estimate a wait — flag it so the caller
|
|
258
|
+
// emits an honest "reset time unknown" error instead of waiting forever.
|
|
259
|
+
weeklyUnknownReset++;
|
|
247
260
|
}
|
|
248
261
|
}
|
|
249
262
|
|
|
@@ -267,6 +280,16 @@ export class AccountManager {
|
|
|
267
280
|
};
|
|
268
281
|
}
|
|
269
282
|
|
|
283
|
+
if (weeklyUnknownReset > 0) {
|
|
284
|
+
return {
|
|
285
|
+
available: false,
|
|
286
|
+
retryAfterMs: Infinity,
|
|
287
|
+
cause: 'weekly_reset_unknown',
|
|
288
|
+
reasons,
|
|
289
|
+
matchingRoutes,
|
|
290
|
+
};
|
|
291
|
+
}
|
|
292
|
+
|
|
270
293
|
return {
|
|
271
294
|
available: false,
|
|
272
295
|
retryAfterMs: Infinity,
|
|
@@ -437,7 +460,10 @@ export class AccountManager {
|
|
|
437
460
|
if (this.getGlobalInFlight() >= this.scheduler.safetyMaxGlobalActive) return false;
|
|
438
461
|
if (account.status === 'exhausted' || account.status === 'error') return false;
|
|
439
462
|
if (this._isSessionQuotaUnavailable(account)) return false;
|
|
440
|
-
|
|
463
|
+
// Gate on RAW weekly usage, not pace-adjusted: an account with real
|
|
464
|
+
// headroom (e.g. 69% used, resets in days) must stay in the healthy-spread
|
|
465
|
+
// pool even if it's burning fast. Pace is a soft SCORE cost, never a bench.
|
|
466
|
+
const weeklyState = this._weeklyRawState(account);
|
|
441
467
|
if (weeklyState === 'exhausted') return false;
|
|
442
468
|
if (weeklyState === 'critical' && !options.allowWeeklyCritical) return false;
|
|
443
469
|
if (weeklyState === 'reserve' && !options.allowWeeklyReserve) return false;
|
|
@@ -574,13 +600,41 @@ export class AccountManager {
|
|
|
574
600
|
});
|
|
575
601
|
}
|
|
576
602
|
|
|
577
|
-
|
|
603
|
+
// Drop wedged head tickets (deadline passed or explicitly marked dead) so a
|
|
604
|
+
// single orphaned waiter cannot block every other request behind it. Cheap;
|
|
605
|
+
// safe to call before every head check.
|
|
606
|
+
_reapStaleQueueHead() {
|
|
607
|
+
const q = this.queueState;
|
|
608
|
+
const now = Date.now();
|
|
609
|
+
let guard = 0;
|
|
610
|
+
while (q.waiting.length && guard++ < 10_000) {
|
|
611
|
+
const head = q.waiting[0];
|
|
612
|
+
const stale = head.dead === true || (head.deadlineAt && now > head.deadlineAt);
|
|
613
|
+
if (!stale) break;
|
|
614
|
+
q.waiting.shift();
|
|
615
|
+
q.bytes = Math.max(0, q.bytes - (head.bytes || 0));
|
|
616
|
+
}
|
|
617
|
+
}
|
|
618
|
+
|
|
619
|
+
// Register a waiter. Returns the ticket, or null if a backpressure limit
|
|
620
|
+
// (maxConcurrentQueued / maxQueuedBytes) would be exceeded — the caller then
|
|
621
|
+
// rejects the request with a "queue full" error instead of holding it.
|
|
622
|
+
registerQueuedRequest(requestInfo = {}, opts = {}) {
|
|
578
623
|
if (requestInfo.queueTicket) return requestInfo.queueTicket;
|
|
624
|
+
this._reapStaleQueueHead();
|
|
625
|
+
const bytes = Math.max(0, Number(opts.bytes) || 0);
|
|
626
|
+
const { maxConcurrentQueued, maxQueuedBytes } = opts;
|
|
627
|
+
if (maxConcurrentQueued != null && this.queueState.waiting.length >= maxConcurrentQueued) return null;
|
|
628
|
+
if (maxQueuedBytes != null && this.queueState.waiting.length > 0
|
|
629
|
+
&& this.queueState.bytes + bytes > maxQueuedBytes) return null;
|
|
579
630
|
const ticket = {
|
|
580
631
|
id: this.queueState.nextId++,
|
|
581
632
|
queuedAt: Date.now(),
|
|
633
|
+
bytes,
|
|
634
|
+
deadlineAt: opts.deadlineAt || null,
|
|
582
635
|
};
|
|
583
636
|
this.queueState.waiting.push(ticket);
|
|
637
|
+
this.queueState.bytes += bytes;
|
|
584
638
|
requestInfo.queueTicket = ticket;
|
|
585
639
|
return ticket;
|
|
586
640
|
}
|
|
@@ -588,10 +642,12 @@ export class AccountManager {
|
|
|
588
642
|
canAdmitQueuedRequest(requestInfo = {}) {
|
|
589
643
|
const ticket = requestInfo.queueTicket;
|
|
590
644
|
if (!ticket) return true;
|
|
645
|
+
this._reapStaleQueueHead();
|
|
591
646
|
if (this.queueState.waiting[0]?.id !== ticket.id) return false;
|
|
592
647
|
const now = Date.now();
|
|
593
648
|
if (now < this.queueState.rampUntil && now - this.queueState.lastAdmissionAt < 250) return false;
|
|
594
649
|
this.queueState.waiting.shift();
|
|
650
|
+
this.queueState.bytes = Math.max(0, this.queueState.bytes - (ticket.bytes || 0));
|
|
595
651
|
this.queueState.lastAdmissionAt = now;
|
|
596
652
|
requestInfo.queueTicket = null;
|
|
597
653
|
requestInfo.queueAdmitted = true;
|
|
@@ -602,7 +658,10 @@ export class AccountManager {
|
|
|
602
658
|
const ticket = requestInfo.queueTicket;
|
|
603
659
|
if (!ticket) return;
|
|
604
660
|
const index = this.queueState.waiting.findIndex(entry => entry.id === ticket.id);
|
|
605
|
-
if (index >= 0)
|
|
661
|
+
if (index >= 0) {
|
|
662
|
+
this.queueState.waiting.splice(index, 1);
|
|
663
|
+
this.queueState.bytes = Math.max(0, this.queueState.bytes - (ticket.bytes || 0));
|
|
664
|
+
}
|
|
606
665
|
requestInfo.queueTicket = null;
|
|
607
666
|
}
|
|
608
667
|
|
|
@@ -1075,8 +1134,21 @@ export class AccountManager {
|
|
|
1075
1134
|
_scoreAccount(account, requestInfo = {}, ctx = null) {
|
|
1076
1135
|
const now = ctx?.now ?? Date.now();
|
|
1077
1136
|
const reqWeight = Math.max(1, requestInfo.weight || 1);
|
|
1078
|
-
const
|
|
1079
|
-
|
|
1137
|
+
const inflight = account.activeWeight + reqWeight;
|
|
1138
|
+
|
|
1139
|
+
// DOMINANT term: in-flight concurrency. Short-term throttling is driven by
|
|
1140
|
+
// how many requests pile on one account, so least-loaded-first spread is
|
|
1141
|
+
// the primary objective.
|
|
1142
|
+
const concurrency = inflight * this.scheduler.concurrencyWeight;
|
|
1143
|
+
|
|
1144
|
+
// Steep soft cap past depth D — the throttle safety floor. No single
|
|
1145
|
+
// account absorbs a deep concurrent burst no matter how "cheap" it looks.
|
|
1146
|
+
const capPenalty = this.scheduler.capPenaltyWeight
|
|
1147
|
+
* Math.max(0, inflight - this.scheduler.perAccountConcurrencyTarget);
|
|
1148
|
+
|
|
1149
|
+
// Burn-pace COST only (demoted from the old dominant scarcity×6 term): a
|
|
1150
|
+
// soft de-preference of accounts burning ahead of an even pace. Never a bench.
|
|
1151
|
+
const paceCost = this._accountScarcity(account, now) * this.scheduler.paceCostWeight;
|
|
1080
1152
|
|
|
1081
1153
|
const fleetRecentWeight = ctx?.fleetRecentWeight ?? 0;
|
|
1082
1154
|
const recentWeight = this._loadSummary(account, this.scheduler.spreadWindowMs, now).weight;
|
|
@@ -1089,7 +1161,7 @@ export class AccountManager {
|
|
|
1089
1161
|
// probed and learned (matches the legacy unknown-quota exploration nudge).
|
|
1090
1162
|
const explorationBonus = account.quota.unified7dReset == null ? -0.5 : 0;
|
|
1091
1163
|
|
|
1092
|
-
return concurrency +
|
|
1164
|
+
return concurrency + capPenalty + paceCost + spread + ramp + failurePenalty + explorationBonus;
|
|
1093
1165
|
}
|
|
1094
1166
|
|
|
1095
1167
|
/**
|
package/src/config.js
CHANGED
|
@@ -81,13 +81,20 @@ export function createDefaultConfig() {
|
|
|
81
81
|
shutdown: {
|
|
82
82
|
drainTimeoutMs: 15_000,
|
|
83
83
|
},
|
|
84
|
+
// When every account is rate-limited, hold the request and retry until one
|
|
85
|
+
// frees up, instead of erroring and killing the session. Only error if
|
|
86
|
+
// nothing recovers within the window below (the early-exit gates on each
|
|
87
|
+
// account's REAL reset time, so a generous bound never spins pointlessly).
|
|
84
88
|
queue: {
|
|
85
89
|
enabled: true,
|
|
86
|
-
maxWaitMs: 24 * 60 * 60 * 1000,
|
|
87
|
-
autoMaxWaitMs: null,
|
|
88
|
-
capacityMaxWaitMs: 15 * 60 * 1000,
|
|
89
|
-
|
|
90
|
-
|
|
90
|
+
maxWaitMs: 24 * 60 * 60 * 1000, // hard ceiling for any hold
|
|
91
|
+
autoMaxWaitMs: null, // 5h/session-cap hold (null = maxWaitMs)
|
|
92
|
+
capacityMaxWaitMs: 15 * 60 * 1000, // upstream 529/overload — stays short, never governed by the others
|
|
93
|
+
weeklyMaxWaitMs: 24 * 60 * 60 * 1000, // weekly (7d) cap hold; was 0 (fail-fast) — that killed sessions on weekly cap
|
|
94
|
+
nonStreamMaxWaitMs: 5 * 60 * 1000, // non-streaming requests have no keepalive; cap their wait
|
|
95
|
+
maxConcurrentQueued: 64, // backpressure: max requests held at once
|
|
96
|
+
maxQueuedBytes: 1024 * 1024 * 1024, // backpressure: max aggregate buffered body bytes (1 GiB)
|
|
97
|
+
maxQueuedBodyBytes: 256 * 1024 * 1024, // per-request cap on a queueable body
|
|
91
98
|
pollMs: 1000,
|
|
92
99
|
heartbeatMs: 10_000,
|
|
93
100
|
},
|
package/src/server.js
CHANGED
|
@@ -19,7 +19,19 @@ const DEFAULT_QUEUE = {
|
|
|
19
19
|
maxWaitMs: 24 * 60 * 60 * 1000,
|
|
20
20
|
autoMaxWaitMs: null,
|
|
21
21
|
capacityMaxWaitMs: 15 * 60 * 1000,
|
|
22
|
-
|
|
22
|
+
// Weekly (7d) cap hold. A generous bound is SAFE because the early-exit
|
|
23
|
+
// gates on the REAL reset time (unified7dReset - now): it only waits when a
|
|
24
|
+
// reset genuinely lands inside the window, and errors honestly otherwise.
|
|
25
|
+
// 0 here was the bug — it fail-fast-killed sessions the instant every
|
|
26
|
+
// account hit its weekly cap, instead of waiting for the soonest reset.
|
|
27
|
+
weeklyMaxWaitMs: 24 * 60 * 60 * 1000,
|
|
28
|
+
// Non-streaming requests have no SSE heartbeat to keep them alive, so a long
|
|
29
|
+
// hold would die on the client timeout anyway. Cap their wait conservatively.
|
|
30
|
+
nonStreamMaxWaitMs: 5 * 60 * 1000,
|
|
31
|
+
// Backpressure: holds used to be 0ms, now they can be hours. Bound the queue
|
|
32
|
+
// so 22 retrying agents can't grow the heap without limit.
|
|
33
|
+
maxConcurrentQueued: 64,
|
|
34
|
+
maxQueuedBytes: 1024 * 1024 * 1024, // 1 GiB aggregate across all held bodies
|
|
23
35
|
pollMs: 1000,
|
|
24
36
|
heartbeatMs: 10_000,
|
|
25
37
|
};
|
|
@@ -825,6 +837,14 @@ function hasEligibleRoute(accountManager, requestInfo = {}, excludedIndexes = ne
|
|
|
825
837
|
return accountManager.hasAvailableRoute?.(requestInfo, excludedIndexes) || false;
|
|
826
838
|
}
|
|
827
839
|
|
|
840
|
+
function formatRetryDuration(seconds) {
|
|
841
|
+
const s = Math.max(0, Math.round(Number(seconds) || 0));
|
|
842
|
+
if (s >= 86400) return `${Math.round(s / 86400)}d`;
|
|
843
|
+
if (s >= 3600) return `${Math.round(s / 3600)}h`;
|
|
844
|
+
if (s >= 60) return `${Math.round(s / 60)}m`;
|
|
845
|
+
return `${s}s`;
|
|
846
|
+
}
|
|
847
|
+
|
|
828
848
|
function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRecoverSoon = true) {
|
|
829
849
|
const thinking = requestInfo.requiresAnthropicThinkingIntegrity
|
|
830
850
|
|| accountManager._requiresAnthropicThinkingIntegrity?.(requestInfo);
|
|
@@ -834,7 +854,10 @@ function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRe
|
|
|
834
854
|
// account is at its own 5h/weekly limit. A short "retry in Ns" would be a lie;
|
|
835
855
|
// tell the user the real fix.
|
|
836
856
|
if (!willRecoverSoon) {
|
|
837
|
-
const
|
|
857
|
+
const eta = Number.isFinite(retryAfter) && retryAfter > 0
|
|
858
|
+
? ` Soonest reset in ~${formatRetryDuration(retryAfter)}, beyond the hold window.`
|
|
859
|
+
: '';
|
|
860
|
+
const base = `No Claude account can take this request — all ${n} are at their 5h or weekly limit.${eta} Add another Claude account or wait for a quota reset.`;
|
|
838
861
|
return thinking
|
|
839
862
|
? `${base} GLM/Kimi fallback is unavailable because this session contains Anthropic signed thinking blocks; start a fresh non-thinking session to use them.`
|
|
840
863
|
: base;
|
|
@@ -1062,29 +1085,64 @@ async function queueAndRetry(
|
|
|
1062
1085
|
? autoMaxWaitMs
|
|
1063
1086
|
: Math.max(0, Number(queueConfig.capacityMaxWaitMs) || 0);
|
|
1064
1087
|
const weeklyMaxWaitMs = Math.max(0, Number(queueConfig.weeklyMaxWaitMs) || 0);
|
|
1088
|
+
const nonStreamMaxWaitMs = queueConfig.nonStreamMaxWaitMs == null
|
|
1089
|
+
? 5 * 60_000
|
|
1090
|
+
: Math.max(0, Number(queueConfig.nonStreamMaxWaitMs) || 0);
|
|
1065
1091
|
const retryPlan = accountManager.nextRetryForRequest?.(requestInfo, new Set()) || {
|
|
1066
1092
|
retryAfterMs: Infinity,
|
|
1067
1093
|
cause: 'unavailable',
|
|
1068
1094
|
};
|
|
1069
|
-
|
|
1070
|
-
|
|
1071
|
-
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
|
|
1095
|
+
|
|
1096
|
+
// Honest, cause-/thinking-aware message used for every give-up path below.
|
|
1097
|
+
const honestMessage = unavailableMessage(
|
|
1098
|
+
accountManager, requestInfo,
|
|
1099
|
+
Math.ceil((Number.isFinite(retryPlan.retryAfterMs) ? retryPlan.retryAfterMs : 0) / 1000),
|
|
1100
|
+
false,
|
|
1101
|
+
);
|
|
1102
|
+
|
|
1103
|
+
// Weekly-capped but the reset time is unknown (cold start / probe failure):
|
|
1104
|
+
// we can't estimate a wait, so don't pretend to — error honestly now.
|
|
1105
|
+
if (retryPlan.cause === 'weekly_reset_unknown') {
|
|
1106
|
+
return finishQueuedStreamIfNeeded(res, requestInfo, honestMessage);
|
|
1107
|
+
}
|
|
1108
|
+
|
|
1109
|
+
// Pick the wait window. Capacity (upstream 529/overload) MUST stay on its own
|
|
1110
|
+
// short cap even when it coincides with weekly exhaustion — never let a
|
|
1111
|
+
// transient overload inherit the long weekly bound.
|
|
1112
|
+
let queueWindowMs = cause === 'capacity'
|
|
1113
|
+
? Math.min(maxWaitMs, capacityMaxWaitMs)
|
|
1114
|
+
: retryPlan.cause === 'weekly_exhausted'
|
|
1115
|
+
? Math.min(maxWaitMs, weeklyMaxWaitMs)
|
|
1116
|
+
: Math.min(maxWaitMs, autoMaxWaitMs);
|
|
1117
|
+
// Non-streaming requests have no SSE heartbeat, so a long hold would die on
|
|
1118
|
+
// the client timeout. Cap them so we never promise a wait we can't deliver.
|
|
1119
|
+
if (!requestInfo.stream) queueWindowMs = Math.min(queueWindowMs, nonStreamMaxWaitMs);
|
|
1120
|
+
|
|
1121
|
+
if (queueWindowMs <= 0) return finishQueuedStreamIfNeeded(res, requestInfo, honestMessage);
|
|
1075
1122
|
|
|
1076
1123
|
const retryAfterMs = retryPlan.retryAfterMs;
|
|
1077
1124
|
if (!Number.isFinite(retryAfterMs) || retryAfterMs > queueWindowMs) {
|
|
1078
|
-
return finishQueuedStreamIfNeeded(res, requestInfo,
|
|
1125
|
+
return finishQueuedStreamIfNeeded(res, requestInfo, honestMessage);
|
|
1079
1126
|
}
|
|
1080
1127
|
|
|
1081
1128
|
requestInfo.queueStartedAt ||= Date.now();
|
|
1082
|
-
accountManager.registerQueuedRequest?.(requestInfo
|
|
1129
|
+
const ticket = accountManager.registerQueuedRequest?.(requestInfo, {
|
|
1130
|
+
bytes: body?.length || 0,
|
|
1131
|
+
deadlineAt: requestInfo.queueStartedAt + queueWindowMs,
|
|
1132
|
+
maxConcurrentQueued: queueConfig.maxConcurrentQueued,
|
|
1133
|
+
maxQueuedBytes: queueConfig.maxQueuedBytes,
|
|
1134
|
+
});
|
|
1135
|
+
if (ticket === null) {
|
|
1136
|
+
// Backpressure: too many requests already waiting / too many bytes buffered.
|
|
1137
|
+
// Reject honestly instead of growing the heap unbounded.
|
|
1138
|
+
return finishQueuedStreamIfNeeded(res, requestInfo,
|
|
1139
|
+
'Maxpool queue is full — too many requests are already waiting for capacity. Try again shortly.');
|
|
1140
|
+
}
|
|
1083
1141
|
const elapsed = Date.now() - requestInfo.queueStartedAt;
|
|
1084
1142
|
const remaining = queueWindowMs - elapsed;
|
|
1085
1143
|
if (remaining <= 0) {
|
|
1086
1144
|
accountManager.removeQueuedRequest?.(requestInfo);
|
|
1087
|
-
return finishQueuedStreamIfNeeded(res, requestInfo,
|
|
1145
|
+
return finishQueuedStreamIfNeeded(res, requestInfo, honestMessage);
|
|
1088
1146
|
}
|
|
1089
1147
|
|
|
1090
1148
|
ctx.account = '(queued)';
|
|
@@ -1096,7 +1154,7 @@ async function queueAndRetry(
|
|
|
1096
1154
|
if (!available) {
|
|
1097
1155
|
if (res.destroyed || req.destroyed) return true;
|
|
1098
1156
|
accountManager.removeQueuedRequest?.(requestInfo);
|
|
1099
|
-
return finishQueuedStreamIfNeeded(res, requestInfo,
|
|
1157
|
+
return finishQueuedStreamIfNeeded(res, requestInfo, honestMessage);
|
|
1100
1158
|
}
|
|
1101
1159
|
|
|
1102
1160
|
return forwardRequest(
|
|
@@ -1163,7 +1221,10 @@ async function waitForAvailableRoute(req, res, accountManager, requestInfo, queu
|
|
|
1163
1221
|
) return true;
|
|
1164
1222
|
|
|
1165
1223
|
const remaining = maxWaitMs - (Date.now() - startedAt);
|
|
1166
|
-
|
|
1224
|
+
// Jitter the poll so a synchronized weekly-reset event doesn't re-align
|
|
1225
|
+
// every waiter's poll into the same instant (thundering scan).
|
|
1226
|
+
const jittered = pollMs * (0.8 + Math.random() * 0.4);
|
|
1227
|
+
await sleep(Math.min(jittered, remaining));
|
|
1167
1228
|
}
|
|
1168
1229
|
|
|
1169
1230
|
return accountManager.hasAvailableRoute(requestInfo, new Set())
|