@rikcodes/teamclaude 1.1.20-rik.16 → 1.1.20-rik.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -2
- package/package.json +1 -1
- package/src/account-manager.js +40 -12
- package/src/codex-quota.js +61 -0
- package/src/codex-reset-credits.js +148 -20
- package/src/conversation.js +230 -0
- package/src/dashboard.js +49 -15
- package/src/index.js +8 -5
- package/src/prober.js +8 -1
- package/src/responses-usage.js +29 -11
- package/src/server.js +115 -49
- package/src/session-tracker.js +85 -17
- package/src/warmer.js +10 -2
package/README.md
CHANGED
|
@@ -65,6 +65,7 @@ Already logged into Claude Code? `teamclaude import` takes its credentials inste
|
|
|
65
65
|
- Pools OpenAI Codex subscriptions alongside Claude accounts (experimental): the Codex CLI is routed through the same proxy, by config or transparently through the MITM proxy, and rotates on its own quota.
|
|
66
66
|
- Takes any Anthropic-compatible API (DeepSeek, GLM) as a low-priority fallback for when the Claude accounts are done.
|
|
67
67
|
- Serves OpenAI models next to Claude ones — a supervised local sidecar translates `gpt-*` requests onto a ChatGPT subscription, with real model names in `/model` and GPT subagents dispatchable from a Claude parent (`sidecars` + `customModels`, this fork).
|
|
68
|
+
- Ships those subagents as a Claude Code plugin: one named agent per model **and effort level**, which the Agent tool's `model` parameter cannot express (`/plugin install rikclaude-agents@rikclaude`, this fork).
|
|
68
69
|
- No dependencies. Node built-ins only.
|
|
69
70
|
|
|
70
71
|
## Everyday commands
|
|
@@ -152,10 +153,10 @@ At launch, `teamclaude run` — and the `claude` alias, which passes through `ru
|
|
|
152
153
|
| Row field | Where it ends up |
|
|
153
154
|
| --- | --- |
|
|
154
155
|
| `model`, `label`, `description` | A `/model` picker row under the **real** model id (`--settings`), so `/model gpt-5.6-sol` works picked or typed |
|
|
155
|
-
| `model` |
|
|
156
|
+
| `model` | Nothing by default. Set `"customModelAgents": true` and each row also becomes a plain subagent named after the model (`--agents`), so "dispatch a `gpt-5.6-terra` subagent" works from a Claude parent. Off because those agents pin no effort and outrank every agent file on disk — prefer the effort-pinned [plugin agents](docs/agents.md) |
|
|
156
157
|
| `contextTokens` | `CLAUDE_CODE_MAX_CONTEXT_TOKENS`, set to the largest value across rows, so Claude Code compacts at the real window instead of assuming 200k |
|
|
157
158
|
|
|
158
|
-
For tools that spawn `claude` themselves, `teamclaude env` can set only environment variables. It carries the window and `ANTHROPIC_CUSTOM_MODEL_OPTION` for the **first** row.
|
|
159
|
+
For tools that spawn `claude` themselves, `teamclaude env` can set only environment variables. It carries the window and `ANTHROPIC_CUSTOM_MODEL_OPTION` for the **first** row. Agents do not depend on either mode: they are read from disk, whether they come from the [`rikclaude-agents` plugin](docs/agents.md) or from a `~/.claude/agents/<name>.md` you write with `model: gpt-5.6-terra` in its frontmatter.
|
|
159
160
|
|
|
160
161
|
Each request is routed by the model name in its body, so one session can freely mix models: use `claude --model gpt-5.6-sol` for a whole session, `/model gpt-5.6-sol` during a session, or a Claude parent that dispatches a GPT subagent.
|
|
161
162
|
|
|
@@ -240,6 +241,7 @@ This feature is on by default. Each account row shows which window binds first:
|
|
|
240
241
|
| [Routing](docs/routing.md) | Rotation, the two kinds of 429, storm control, model routes, session spreading, pinning, prompt cache |
|
|
241
242
|
| [Quota](docs/quota.md) | Quota probe, keep-warm, holding on exhaustion |
|
|
242
243
|
| [OpenAI models](docs/openai.md) | Codex sidecar setup, custom model registration, GPT subagents, limitations |
|
|
244
|
+
| [Subagents](docs/agents.md) | The `rikclaude-agents` plugin: effort-pinned agents per model, installing it, seeding it for a team |
|
|
243
245
|
| [Configuration](docs/configuration.md) | Config format, every field, environment variables, network tuning |
|
|
244
246
|
| [Proxy modes](docs/proxy-modes.md) | MITM forward proxy, sx.org residential egress |
|
|
245
247
|
| [Remote host](docs/remote.md) | Running the fleet on an always-on box: reaching it, moving the accounts, service supervision, pointing clients at it |
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@rikcodes/teamclaude",
|
|
3
|
-
"version": "1.1.20-rik.
|
|
3
|
+
"version": "1.1.20-rik.17",
|
|
4
4
|
"description": "Multi-account proxy for Claude Code and Codex: pools Claude Max, ChatGPT/Codex, API-key and third-party backend accounts, and rotates on quota",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "src/index.js",
|
package/src/account-manager.js
CHANGED
|
@@ -383,6 +383,7 @@ export class AccountManager {
|
|
|
383
383
|
// `distributeSessions` gates the behavioural change: keep each session on its
|
|
384
384
|
// account for cache reuse, but spread NEW sessions across equal-priority
|
|
385
385
|
// accounts by load instead of funnelling them all onto the current one.
|
|
386
|
+
/** @type {SessionTracker} */
|
|
386
387
|
this.sessionTracker = sessionTracker || new SessionTracker();
|
|
387
388
|
// 'off' | 'even' | 'adaptive'. `distributeSessions` stays a boolean beside
|
|
388
389
|
// it ("is distribution on at all") because the status readout, the TUI
|
|
@@ -770,6 +771,12 @@ export class AccountManager {
|
|
|
770
771
|
* bucket, so the account must be eligible for both models. When no account
|
|
771
772
|
* satisfies both, selection degrades to executor-only routing so the main
|
|
772
773
|
* request keeps flowing (upstream then fails just the advisor call).
|
|
774
|
+
*
|
|
775
|
+
* `sessionId` is a PIN KEY throughout this class and the session tracker: a
|
|
776
|
+
* client session narrowed to the conversation within it, since one client
|
|
777
|
+
* session fans out across many (see conversation.js). It is spelled
|
|
778
|
+
* `sessionId` because it is one for every request that names no conversation,
|
|
779
|
+
* and because the affinity it drives is the same affinity it always was.
|
|
773
780
|
*/
|
|
774
781
|
getActiveAccount(exclude = null, model = null, advisorModel = null, sessionId = null, provider = DEFAULT_PROVIDER, decision = null) {
|
|
775
782
|
// Selection reads this.currentIndex as "where the fleet is". With more than
|
|
@@ -896,11 +903,15 @@ export class AccountManager {
|
|
|
896
903
|
// The empty set marks this call as a request's, since a poll hands none.
|
|
897
904
|
// Allocated fresh: a set handed out once is one a later reader could add to.
|
|
898
905
|
this.refreshExpiredQuotas(model, this.expiryRouting.enabled ? (exclude ?? new Set()) : exclude, advisorModel);
|
|
899
|
-
// Session-affinity distribution (opt-in): keep a
|
|
900
|
-
// account for cache reuse, and route a new
|
|
901
|
-
//
|
|
902
|
-
//
|
|
903
|
-
//
|
|
906
|
+
// Session-affinity distribution (opt-in): keep a conversation on its pinned
|
|
907
|
+
// account for cache reuse, and route a new one to the least-loaded account.
|
|
908
|
+
// The unit is the conversation rather than the client session because one
|
|
909
|
+
// session fans out — a Claude Code session and every subagent it launches
|
|
910
|
+
// share its id, and holding them together pinned a whole fan-out to one
|
|
911
|
+
// account for a cache none of them shared (see conversation.js). Only when
|
|
912
|
+
// enabled, only for a request that named one, and only outside a manual
|
|
913
|
+
// route pin (which must still win). Falls through to the normal walk if
|
|
914
|
+
// nothing eligible is found (e.g. the whole tier is exhausted).
|
|
904
915
|
if (sessionId && !this._pinnedAccountForModel(model, advisorModel)) {
|
|
905
916
|
if (this.distributeSessions) {
|
|
906
917
|
const acc = this._selectForSession(sessionId, exclude, model, advisorModel);
|
|
@@ -1054,13 +1065,18 @@ export class AccountManager {
|
|
|
1054
1065
|
// (_isAvailable, below) rather than a second key.
|
|
1055
1066
|
const bucket = this._weeklyBucketFor(model);
|
|
1056
1067
|
const pinIdx = this.sessionTracker.pinnedAccount(sessionId, bucket);
|
|
1057
|
-
// The bucket's own pin first. Failing that, any account
|
|
1058
|
-
// sits on for another family: one
|
|
1059
|
-
// account cannot serve the request (the README's "pins it
|
|
1060
|
-
// this,
|
|
1061
|
-
// _pickLeastLoaded, which counts the
|
|
1062
|
-
// the new family onto a sibling — splitting every mixed-model
|
|
1063
|
-
// across two accounts by construction, not only on a real
|
|
1068
|
+
// The bucket's own pin first. Failing that, any account this conversation
|
|
1069
|
+
// already sits on for another family: one conversation stays on one account
|
|
1070
|
+
// unless that account cannot serve the request (the README's "pins it
|
|
1071
|
+
// there"). Without this, its first request of a second family would go to
|
|
1072
|
+
// _pickLeastLoaded, which counts the conversation's own pin as load and
|
|
1073
|
+
// pushes the new family onto a sibling — splitting every mixed-model
|
|
1074
|
+
// conversation across two accounts by construction, not only on a real
|
|
1075
|
+
// diversion.
|
|
1076
|
+
//
|
|
1077
|
+
// Scoped to the one conversation, not to everything sharing its session id:
|
|
1078
|
+
// two agents of one fan-out are not each other's affinity, and treating
|
|
1079
|
+
// them as one is what funnelled them onto a single account.
|
|
1064
1080
|
const candidates = [];
|
|
1065
1081
|
if (pinIdx != null) candidates.push(pinIdx);
|
|
1066
1082
|
for (const idx of this.sessionTracker.pinnedAccounts(sessionId)) {
|
|
@@ -1354,6 +1370,18 @@ export class AccountManager {
|
|
|
1354
1370
|
this.sessionTracker.recordOutcome(sessionId, usable);
|
|
1355
1371
|
}
|
|
1356
1372
|
|
|
1373
|
+
/**
|
|
1374
|
+
* The same, for an exit that can name only the client session — every live
|
|
1375
|
+
* conversation of it takes the outcome (see SessionTracker).
|
|
1376
|
+
*
|
|
1377
|
+
* @param {string|null} sessionId
|
|
1378
|
+
* @param {boolean|null} usable
|
|
1379
|
+
*/
|
|
1380
|
+
recordOutcomeForSession(sessionId, usable) {
|
|
1381
|
+
if (!sessionId || usable === null) return;
|
|
1382
|
+
this.sessionTracker.recordOutcomeForSession(sessionId, usable);
|
|
1383
|
+
}
|
|
1384
|
+
|
|
1357
1385
|
/**
|
|
1358
1386
|
* Per-account diagnostics for adaptive distribution: the profile plan tier,
|
|
1359
1387
|
* relative score weight, and actual next target.
|
package/src/codex-quota.js
CHANGED
|
@@ -110,6 +110,18 @@ function classify(windows) {
|
|
|
110
110
|
return out;
|
|
111
111
|
}
|
|
112
112
|
|
|
113
|
+
/**
|
|
114
|
+
* A parsed reading: the fields `account.quota` already uses, each present only
|
|
115
|
+
* when the headers stated it.
|
|
116
|
+
*
|
|
117
|
+
* @typedef {object} CodexQuota
|
|
118
|
+
* @property {number} [unified5h] Session-window utilization, 0-1.
|
|
119
|
+
* @property {number} [unified5hReset] When that window resets, ms epoch.
|
|
120
|
+
* @property {number} [unified7d] Weekly-window utilization, 0-1.
|
|
121
|
+
* @property {number} [unified7dReset] When that window resets, ms epoch.
|
|
122
|
+
* @property {{slug: string, name: string, utilization: number, resetAt: number|null}[]} [modelBuckets] Model-scoped weekly buckets.
|
|
123
|
+
*/
|
|
124
|
+
|
|
113
125
|
/**
|
|
114
126
|
* Parse Codex rate-limit headers into the fields `account.quota` already uses.
|
|
115
127
|
*
|
|
@@ -117,9 +129,12 @@ function classify(windows) {
|
|
|
117
129
|
* an existing quota without blanking readings this response did not mention.
|
|
118
130
|
* An empty object means "this response carried no quota", which is normal:
|
|
119
131
|
* the catalog fetch (`/models`) has none.
|
|
132
|
+
*
|
|
133
|
+
* @returns {CodexQuota}
|
|
120
134
|
*/
|
|
121
135
|
export function parseCodexQuota(headers) {
|
|
122
136
|
const families = collectFamilies(headers);
|
|
137
|
+
/** @type {CodexQuota} */
|
|
123
138
|
const quota = {};
|
|
124
139
|
|
|
125
140
|
const account = classify(families.get('')?.windows || {});
|
|
@@ -172,6 +187,52 @@ export function parseCodexQuota(headers) {
|
|
|
172
187
|
return quota;
|
|
173
188
|
}
|
|
174
189
|
|
|
190
|
+
/**
|
|
191
|
+
* Is a window at or past its limit?
|
|
192
|
+
*
|
|
193
|
+
* Readings are 0-1 fractions here, and the comparison is `>=` because upstream
|
|
194
|
+
* keeps counting once a window is past its limit: 104% arrives as 1.04. A
|
|
195
|
+
* window the headers did not state is not spent — `classify` has already
|
|
196
|
+
* dropped the unparseable and the zeroed.
|
|
197
|
+
*
|
|
198
|
+
* @param {number} [utilization]
|
|
199
|
+
*/
|
|
200
|
+
const isSpent = (utilization) => utilization != null && utilization >= 1;
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* Which windows these headers report as spent — at or past their limit. Empty
|
|
204
|
+
* when none are, including when the response carried no Codex quota at all.
|
|
205
|
+
*
|
|
206
|
+
* Anthropic names a spent bucket outright, as `…-status: rejected`. This API
|
|
207
|
+
* publishes no status at all: it reports how much of each window is gone, and a
|
|
208
|
+
* window at its limit is the same fact in the other spelling. That distinction
|
|
209
|
+
* is what a 429 handler needs, because a spent window is durable — the account
|
|
210
|
+
* cannot serve again until the window resets, so waiting out `retry-after` and
|
|
211
|
+
* asking the SAME account again is futile, however momentary the 429 looked.
|
|
212
|
+
*
|
|
213
|
+
* Every family is read, not just the account-wide one. A subscription states
|
|
214
|
+
* its only 5-hour window inside a NAMED family (see parseCodexQuota), so the
|
|
215
|
+
* account-wide percentages alone would never show a spent session window; and a
|
|
216
|
+
* model-scoped weekly bucket is spent on its own terms, whatever the
|
|
217
|
+
* account-wide reading says.
|
|
218
|
+
*
|
|
219
|
+
* The labels name what is spent, for the log line that follows. A caller that
|
|
220
|
+
* only wants the verdict tests the length.
|
|
221
|
+
*
|
|
222
|
+
* @param {Record<string, string>} headers Rate-limit headers from the response.
|
|
223
|
+
* @returns {string[]} One label per spent window, e.g. `['weekly']`.
|
|
224
|
+
*/
|
|
225
|
+
export function codexSpentWindows(headers) {
|
|
226
|
+
const quota = parseCodexQuota(headers);
|
|
227
|
+
const spent = [];
|
|
228
|
+
if (isSpent(quota.unified5h)) spent.push('5h');
|
|
229
|
+
if (isSpent(quota.unified7d)) spent.push('weekly');
|
|
230
|
+
for (const bucket of quota.modelBuckets ?? []) {
|
|
231
|
+
if (isSpent(bucket.utilization)) spent.push(`${bucket.name} weekly`);
|
|
232
|
+
}
|
|
233
|
+
return spent;
|
|
234
|
+
}
|
|
235
|
+
|
|
175
236
|
/** The subscription plan upstream reports, for status output. Null when absent. */
|
|
176
237
|
export function parseCodexPlanType(headers) {
|
|
177
238
|
const value = headers?.['x-codex-plan-type'];
|
|
@@ -50,17 +50,26 @@ export const CREDIT_EXPIRY_WINDOW_MS = 3 * 24 * 60 * 60 * 1000;
|
|
|
50
50
|
// touches the detail endpoint before the policy has even had a chance to say no.
|
|
51
51
|
const DETAIL_TTL_MS = 6 * 60 * 60 * 1000;
|
|
52
52
|
|
|
53
|
-
// After an attempt that spent nothing (upstream declined, or the request
|
|
54
|
-
//
|
|
53
|
+
// After an attempt that spent nothing (upstream declined, or the request never
|
|
54
|
+
// got that far), how long before this account may try again. An attempt whose
|
|
55
|
+
// verdict we never read holds the whole fleet for the same span — see _consume.
|
|
55
56
|
const RETRY_COOLDOWN_MS = 30 * 60 * 1000;
|
|
56
57
|
|
|
57
|
-
// After an attempt that DID spend a credit
|
|
58
|
-
//
|
|
59
|
-
//
|
|
60
|
-
//
|
|
61
|
-
// not.
|
|
58
|
+
// After an attempt that DID spend a credit — on the account that spent it and
|
|
59
|
+
// on the fleet alike. Longer, and deliberately so: if the weekly window still
|
|
60
|
+
// reads exhausted afterwards — a reset that only covered the session window, a
|
|
61
|
+
// usage read that failed or had not caught up — the next trigger must not reach
|
|
62
|
+
// for a second credit to fix what the first one apparently did not. And the
|
|
63
|
+
// account that redeemed is not the only one that must not: the refusal behind
|
|
64
|
+
// it walks the whole pool, and a sibling holding a credit is just as able to
|
|
65
|
+
// spend one.
|
|
62
66
|
const SUCCESS_COOLDOWN_MS = 6 * 60 * 60 * 1000;
|
|
63
67
|
|
|
68
|
+
// What a refresh that outlived the attempt's budget resolves with. A sentinel
|
|
69
|
+
// rather than a value, because every real outcome of a refresh — including a
|
|
70
|
+
// failure — is a value the race could otherwise be confused for.
|
|
71
|
+
const BUDGET_LAPSED = Symbol('redeem budget lapsed');
|
|
72
|
+
|
|
64
73
|
/**
|
|
65
74
|
* How long the whole redemption may take — the token refresh, the detail read
|
|
66
75
|
* and the consume TOGETHER, not each.
|
|
@@ -203,8 +212,9 @@ export async function consumeResetCredit(account, { creditId = null, redeemReque
|
|
|
203
212
|
* The 429 path cannot answer this from the flat `x-codex-*-used-percent`
|
|
204
213
|
* headers: those are positions, not durations, so a spent 5-hour window looks
|
|
205
214
|
* identical there. The parsed quota keeps the two apart (see codex-quota.js),
|
|
206
|
-
* and the 5-hour one is explicitly NOT a trigger — it
|
|
207
|
-
*
|
|
215
|
+
* and the 5-hour one is explicitly NOT a trigger — it comes back on its own
|
|
216
|
+
* within hours, while the weekly one is what leaves an account walled for days,
|
|
217
|
+
* and a "Full reset" is far too scarce to burn on the short window.
|
|
208
218
|
*
|
|
209
219
|
* @param {Record<string, any>|null|undefined} account
|
|
210
220
|
* @param {number} [now]
|
|
@@ -386,6 +396,14 @@ export function shouldRedeemReset({ account, autoRedeemResets = false, pool = []
|
|
|
386
396
|
* refusals arrives here asking the same question. Concurrent callers therefore
|
|
387
397
|
* join one attempt rather than starting their own, per account and per pool.
|
|
388
398
|
*
|
|
399
|
+
* Joining only answers the triggers that arrive together, though; they also
|
|
400
|
+
* arrive one after another. What bounds those is a fleet-wide hold, armed the
|
|
401
|
+
* moment an attempt has spent a credit — or may have — and read before the next
|
|
402
|
+
* attempt reads anything at all. It is deliberately not the per-account
|
|
403
|
+
* cooldown, and not the per-account join either: a pool-dry attempt walks the
|
|
404
|
+
* whole pool, so an account that has never touched the endpoint is just as able
|
|
405
|
+
* to spend the second credit as the one that did.
|
|
406
|
+
*
|
|
389
407
|
* That waiting client is also what sets the time budget. The budget is ONE
|
|
390
408
|
* deadline for the whole attempt (REDEEM_BUDGET_MS above), not a
|
|
391
409
|
* timeout per call: what the waiting client can spare is a property of the
|
|
@@ -446,6 +464,34 @@ export class ResetCreditRedeemer {
|
|
|
446
464
|
// out, and only one of them should spend anything about it.
|
|
447
465
|
/** @type {Promise<RedeemResult>|null} */
|
|
448
466
|
this.poolInFlight = null;
|
|
467
|
+
// The hold a per-account cooldown cannot express. Two outcomes are facts
|
|
468
|
+
// about the FLEET rather than about the account they happened on:
|
|
469
|
+
//
|
|
470
|
+
// - a redemption that worked. What otherwise stops the next trigger
|
|
471
|
+
// spending a sibling's credit is this account reading available again —
|
|
472
|
+
// and that is a re-read, which can fail or find upstream not yet caught
|
|
473
|
+
// up with its own reset. The hold must not depend on it having landed.
|
|
474
|
+
// - a consume whose verdict we never read. This endpoint states a refusal
|
|
475
|
+
// as a 200 with a `code`, so an error is never upstream saying no — only
|
|
476
|
+
// "we do not know", over a POST that may well have spent the credit.
|
|
477
|
+
//
|
|
478
|
+
// Armed in _consume, read at the top of both entry points before anything
|
|
479
|
+
// is fetched, and it stops a pool walk mid-pool.
|
|
480
|
+
this.fleetCooldownUntil = 0;
|
|
481
|
+
}
|
|
482
|
+
|
|
483
|
+
/**
|
|
484
|
+
* Has an attempt just spent a credit, or possibly spent one, on behalf of the
|
|
485
|
+
* whole fleet? Read before any account-level state, because the answer is
|
|
486
|
+
* about none of them in particular.
|
|
487
|
+
*
|
|
488
|
+
* @returns {RedeemResult|null}
|
|
489
|
+
*/
|
|
490
|
+
_fleetHold() {
|
|
491
|
+
if (this.now() < this.fleetCooldownUntil) {
|
|
492
|
+
return { redeemed: false, reason: 'the pool is cooling down after a recent redemption attempt' };
|
|
493
|
+
}
|
|
494
|
+
return null;
|
|
449
495
|
}
|
|
450
496
|
|
|
451
497
|
/**
|
|
@@ -525,6 +571,11 @@ export class ResetCreditRedeemer {
|
|
|
525
571
|
* @returns {Promise<RedeemResult>}
|
|
526
572
|
*/
|
|
527
573
|
async _poolAttempt(accounts) {
|
|
574
|
+
const held = this._fleetHold();
|
|
575
|
+
// Before a single row is read: an attempt that has just spent a credit, or
|
|
576
|
+
// may have, speaks for the whole pool rather than for the account it
|
|
577
|
+
// touched.
|
|
578
|
+
if (held) return held;
|
|
528
579
|
const now = this.now();
|
|
529
580
|
// ONE budget for the whole refusal, shared across every candidate: the
|
|
530
581
|
// client is waiting on the refusal, not on an account, so a pool of three
|
|
@@ -551,6 +602,11 @@ export class ResetCreditRedeemer {
|
|
|
551
602
|
state => this._decide(account, state, rows.get(account) || [], now, deadline));
|
|
552
603
|
if (result.redeemed) return result;
|
|
553
604
|
reason = result.reason;
|
|
605
|
+
// Two things end the walk rather than move it on to the next account: a
|
|
606
|
+
// fleet hold, armed by an attempt that may have spent something — walking
|
|
607
|
+
// on would answer a credit we cannot account for with a second one — and
|
|
608
|
+
// a budget the waiting client has nothing left of to give.
|
|
609
|
+
if (this._fleetHold() || this._timeLeft(deadline) <= 0) break;
|
|
554
610
|
}
|
|
555
611
|
return { redeemed: false, reason };
|
|
556
612
|
}
|
|
@@ -632,6 +688,11 @@ export class ResetCreditRedeemer {
|
|
|
632
688
|
* @returns {Promise<RedeemResult>}
|
|
633
689
|
*/
|
|
634
690
|
async _attempt(account, state) {
|
|
691
|
+
// The same hold the pool walk reads, and for the same reason: a credit just
|
|
692
|
+
// spent — or possibly spent — elsewhere in the fleet is not this account's
|
|
693
|
+
// business to answer with another one.
|
|
694
|
+
const held = this._fleetHold();
|
|
695
|
+
if (held) return held;
|
|
635
696
|
const now = this.now();
|
|
636
697
|
// Everything past the free checks can touch the network, and all of it
|
|
637
698
|
// shares this one deadline: what the waiting client can spare belongs to the
|
|
@@ -656,6 +717,49 @@ export class ResetCreditRedeemer {
|
|
|
656
717
|
return deadline - this.now();
|
|
657
718
|
}
|
|
658
719
|
|
|
720
|
+
/**
|
|
721
|
+
* Refresh the account's token, and never wait longer for it than the attempt
|
|
722
|
+
* has left.
|
|
723
|
+
*
|
|
724
|
+
* `ensureTokenFresh` takes no timeout of its own and joins whatever refresh is
|
|
725
|
+
* already running, so awaiting it plainly is unbounded — and reading the clock
|
|
726
|
+
* afterwards can only report a deadline already missed, with every concurrent
|
|
727
|
+
* trigger held for exactly as long. The only way to bound it is to race it.
|
|
728
|
+
*
|
|
729
|
+
* A refresh that lands late is not wrong, merely too late to act on: it
|
|
730
|
+
* carries on in the background, and whatever asks next gets the token it
|
|
731
|
+
* fetched. What must not happen is this attempt proceeding on it — an
|
|
732
|
+
* irreversible call made for a client that has already stopped waiting is the
|
|
733
|
+
* worst of both outcomes.
|
|
734
|
+
*
|
|
735
|
+
* @param {Record<string, any>} account
|
|
736
|
+
* @param {number} deadline
|
|
737
|
+
* @returns {Promise<{ok: boolean, error?: string}>}
|
|
738
|
+
*/
|
|
739
|
+
async _freshToken(account, deadline) {
|
|
740
|
+
const left = this._timeLeft(deadline);
|
|
741
|
+
if (left <= 0) return { ok: false, error: 'ran out of the redeem budget before refreshing the token' };
|
|
742
|
+
/** @type {any} */
|
|
743
|
+
let timer = null;
|
|
744
|
+
/** @type {Promise<any>} */
|
|
745
|
+
const lapsed = new Promise(resolve => { timer = setTimeout(() => resolve(BUDGET_LAPSED), left); });
|
|
746
|
+
try {
|
|
747
|
+
// Wrapped so a synchronous throw arrives as a rejection like any other,
|
|
748
|
+
// and settled both ways so losing the race cannot leave one unhandled.
|
|
749
|
+
const refreshed = (async () => this.am.ensureTokenFresh(account.index))()
|
|
750
|
+
.then(() => null, (/** @type {any} */ err) => err ?? new Error('token refresh failed'));
|
|
751
|
+
const outcome = await Promise.race([refreshed, lapsed]);
|
|
752
|
+
if (outcome === BUDGET_LAPSED) return { ok: false, error: 'ran out of the redeem budget refreshing the token' };
|
|
753
|
+
// A refresh that failed is an answer, and the answer is no: the credential
|
|
754
|
+
// in hand is the one upstream has already stopped accepting, and the next
|
|
755
|
+
// call would spend the budget proving it.
|
|
756
|
+
if (outcome) return { ok: false, error: `token refresh failed (${safeLine(outcome.message || String(outcome), 80)})` };
|
|
757
|
+
return { ok: true };
|
|
758
|
+
} finally {
|
|
759
|
+
clearTimeout(timer);
|
|
760
|
+
}
|
|
761
|
+
}
|
|
762
|
+
|
|
659
763
|
/**
|
|
660
764
|
* The detail rows, from cache while they are fresh.
|
|
661
765
|
*
|
|
@@ -667,10 +771,11 @@ export class ResetCreditRedeemer {
|
|
|
667
771
|
*/
|
|
668
772
|
async _credits(account, state, now, deadline) {
|
|
669
773
|
if (state.credits && now - state.creditsAt < this.detailTtlMs) return { credits: state.credits };
|
|
670
|
-
// The refresh is inside the budget too
|
|
671
|
-
//
|
|
672
|
-
//
|
|
673
|
-
await this.
|
|
774
|
+
// The refresh is inside the budget too, and it is the one step here with no
|
|
775
|
+
// timeout of its own, so it is raced against the deadline rather than
|
|
776
|
+
// trusted to come back inside it.
|
|
777
|
+
const fresh = await this._freshToken(account, deadline);
|
|
778
|
+
if (!fresh.ok) return { error: fresh.error };
|
|
674
779
|
const timeoutMs = this._timeLeft(deadline);
|
|
675
780
|
if (timeoutMs <= 0) return { error: 'ran out of the redeem budget before reading the credit rows' };
|
|
676
781
|
const result = await this.detailsFn(account, { timeoutMs });
|
|
@@ -691,15 +796,21 @@ export class ResetCreditRedeemer {
|
|
|
691
796
|
*/
|
|
692
797
|
async _consume(account, state, verdict, now, deadline) {
|
|
693
798
|
const name = safeLine(account.name, 64);
|
|
694
|
-
|
|
695
|
-
|
|
799
|
+
// Raced against the deadline, not merely measured after the fact: the
|
|
800
|
+
// budget exists to bound what the waiting client is held for, and a check
|
|
801
|
+
// that runs once the refresh has returned can only report a promise already
|
|
802
|
+
// broken. See _freshToken.
|
|
803
|
+
const fresh = await this._freshToken(account, deadline);
|
|
804
|
+
const timeoutMs = fresh.ok ? this._timeLeft(deadline) : 0;
|
|
696
805
|
if (timeoutMs <= 0) {
|
|
697
806
|
// Declined like any other attempt that spent nothing: a cooldown, so a
|
|
698
807
|
// burst against a slow upstream cannot re-enter here per rejection, and
|
|
699
|
-
// no idempotency key minted for a request never made.
|
|
808
|
+
// no idempotency key minted for a request never made. Per-account only —
|
|
809
|
+
// nothing was sent, so the fleet has nothing to be careful of.
|
|
810
|
+
const why = fresh.error || 'ran out of the redeem budget';
|
|
700
811
|
state.cooldownUntil = now + RETRY_COOLDOWN_MS;
|
|
701
|
-
this.log(`[TeamClaude] Codex rate-limit reset on "${name}"
|
|
702
|
-
return { redeemed: false, reason:
|
|
812
|
+
this.log(`[TeamClaude] Codex rate-limit reset on "${name}" stopped before redeeming — ${why}`);
|
|
813
|
+
return { redeemed: false, reason: why };
|
|
703
814
|
}
|
|
704
815
|
|
|
705
816
|
// Reused across retries of the SAME logical attempt: an attempt that failed
|
|
@@ -713,9 +824,18 @@ export class ResetCreditRedeemer {
|
|
|
713
824
|
{ timeoutMs });
|
|
714
825
|
|
|
715
826
|
if (result?.error) {
|
|
716
|
-
//
|
|
827
|
+
// We do not know whether that POST was acted on, and this is the one
|
|
828
|
+
// place in the file where not knowing is expensive. Upstream states a
|
|
829
|
+
// refusal as a 200 with a `code`, so an error here is never "upstream
|
|
830
|
+
// said no" — it is a verdict we never read, over a request that may well
|
|
831
|
+
// have spent the credit. So the key is deliberately kept (the retry
|
|
832
|
+
// replays it, and `already_redeemed` is upstream answering for it), and
|
|
833
|
+
// the hold is fleet-wide: a pool walk would otherwise move straight on to
|
|
834
|
+
// a sibling, and a second credit spent over an uncertain first is exactly
|
|
835
|
+
// how one becomes two.
|
|
717
836
|
state.cooldownUntil = now + RETRY_COOLDOWN_MS;
|
|
718
|
-
this.
|
|
837
|
+
this.fleetCooldownUntil = Math.max(this.fleetCooldownUntil, now + RETRY_COOLDOWN_MS);
|
|
838
|
+
this.log(`[TeamClaude] Codex rate-limit reset failed on "${name}" — ${safeLine(result.error, 120)}; it may still have been spent, so the pool holds off`);
|
|
719
839
|
return { redeemed: false, reason: `redeem failed (${result.error})` };
|
|
720
840
|
}
|
|
721
841
|
|
|
@@ -727,6 +847,14 @@ export class ResetCreditRedeemer {
|
|
|
727
847
|
// that did land, so it is a success with the windows already reset.
|
|
728
848
|
if (result?.code === 'reset' || result?.code === 'already_redeemed') {
|
|
729
849
|
state.cooldownUntil = now + SUCCESS_COOLDOWN_MS;
|
|
850
|
+
// And the fleet with it. Not because this account might redeem again —
|
|
851
|
+
// its own cooldown answers that — but because the only thing that would
|
|
852
|
+
// otherwise stop the next trigger spending a SIBLING's credit is this
|
|
853
|
+
// account reading available again, which is the re-read below: the one
|
|
854
|
+
// step here that is allowed to fail. A hold that does not depend on that
|
|
855
|
+
// reading is what keeps "at most one credit per dry pool" true when it
|
|
856
|
+
// does fail.
|
|
857
|
+
this.fleetCooldownUntil = Math.max(this.fleetCooldownUntil, now + SUCCESS_COOLDOWN_MS);
|
|
730
858
|
const windows = result.windowsReset || 0;
|
|
731
859
|
this.log(`[TeamClaude] Redeemed a free Codex rate-limit reset on "${name}" — ${windows} window(s) reset (${result.code})`);
|
|
732
860
|
await this._refresh(account, name, deadline);
|