@timo972/cc-router 0.9.0 → 0.10.0-rc.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,16 +4,19 @@ import { ServerResponse } from "http";
4
4
  import { timingSafeEqual } from "crypto";
5
5
  import { TokenPool } from "./token-pool.js";
6
6
  import { needsRefresh, refreshAccountIfCurrent, saveAccounts, startRefreshLoop } from "./token-refresher.js";
7
- import { loadAccounts, loadOpenAIAccounts, saveOpenAIAccounts, accountsFileExists, readAccountsFromPath, readConfig, writeConfig, getProxyRequestTimeoutMs, migrateLegacyAccountProviders, setProviderAccountsEnabled } from "../config/manager.js";
7
+ import { loadAccounts, loadOpenAIAccounts, saveOpenAIAccountsToPath, accountsFileExists, readAccountsFromPath, readConfig, writeConfig, getProxyRequestTimeoutMs, migrateLegacyAccountProviders, setProviderAccountsEnabled } from "../config/manager.js";
8
8
  import { checkForUpdate, performUpdate, restartSelf, printUpdateBanner } from "../utils/self-update.js";
9
9
  import { trackEvent, startHeartbeat } from "../utils/telemetry.js";
10
10
  import { loadTelemetryState } from "../config/telemetry.js";
11
11
  import { logRoute, logError, logStartup } from "./logger.js";
12
12
  import { createLocalRoutingErrorLog, stats } from "./stats.js";
13
- import { PROXY_PORT, LITELLM_URL } from "../config/paths.js";
13
+ import { PROXY_PORT, LITELLM_URL, ACCOUNTS_PATH } from "../config/paths.js";
14
14
  import { writePid, removePid } from "../daemon/pid.js";
15
- import { createOpenAIAccountPicker } from "../providers/openai/account-pool.js";
16
- import { prepareOpenAIAccountForRequest, startOpenAIRefreshLoop } from "../providers/openai/token-refresher.js";
15
+ import { applyOpenAIAccountPatch, validateAccountPatchBody } from "./account-patch.js";
16
+ import { hasPendingCredentialWrite, markOpenAICredentialsPersisted, prepareOpenAIAccountForRequest, refreshAndPersistOpenAIAccount, startOpenAIRefreshLoop, } from "../providers/openai/token-refresher.js";
17
+ import { createOpenAIAccount } from "../providers/openai/account-state.js";
18
+ import { OpenAITokenPool } from "../providers/openai/token-pool.js";
19
+ import { DEFAULT_CODEX_LIMIT_ID } from "../providers/openai/usage.js";
17
20
  import { mountResponsesRoutes } from "./responses-server.js";
18
21
  import { mountMessagesCrossProviderRoute } from "./messages-cross-route.js";
19
22
  import { mountModelsRoute } from "./models-server.js";
@@ -68,10 +71,10 @@ export function createOperationalStatus(opts) {
68
71
  },
69
72
  };
70
73
  }
71
- export function createHealthAccountViews(anthropicAccounts, openAIAccounts, resolveRoutingMetrics = zeroRoutingMetrics) {
74
+ export function createHealthAccountViews(anthropicAccounts, openAIAccounts, resolveRoutingMetrics = zeroRoutingMetrics, resolveOpenAIRouting) {
72
75
  return [
73
76
  ...anthropicAccounts.map(account => (publicAnthropicAccountView(account, resolveRoutingMetrics(account.id)))),
74
- ...openAIAccounts.map(publicOpenAIAccountView),
77
+ ...openAIAccounts.map(account => publicOpenAIAccountView(account, resolveOpenAIRouting?.(account.id) ?? { metrics: zeroRoutingMetrics(account.id), cooldowns: { globalUntilMs: 0, bucketCooldowns: [] } })),
75
78
  ];
76
79
  }
77
80
  function publicAnthropicAccountView(a, metrics) {
@@ -182,23 +185,87 @@ function publicRepresentativeClaim(claim) {
182
185
  }
183
186
  return "unknown";
184
187
  }
185
- function publicOpenAIAccountView(a) {
188
+ function publicOpenAIAccountView(a, routing) {
186
189
  const expiresInMs = a.expiresAt - Date.now();
187
190
  return {
188
191
  id: a.id,
189
192
  provider: "openai_subscription",
190
193
  enabled: a.enabled !== false,
191
- healthy: a.enabled !== false && expiresInMs > 0,
192
- busy: false,
193
- inFlightRequests: 0,
194
- activeSessions: 0,
195
- requestCount: 0,
196
- errorCount: 0,
194
+ sessionLimitPercent: a.sessionLimitPercent,
195
+ weeklyLimitPercent: a.weeklyLimitPercent,
196
+ healthy: a.enabled !== false && a.healthy && expiresInMs > 0,
197
+ busy: routing.metrics.coolingDown,
198
+ cooldownUntilMs: routing.metrics.cooldownUntilMs ?? 0,
199
+ globalCooldownUntilMs: routing.cooldowns.globalUntilMs,
200
+ inFlightRequests: routing.metrics.inFlightRequests,
201
+ activeSessions: routing.metrics.activeSessions,
202
+ requestCount: a.requestCount,
203
+ errorCount: a.errorCount,
197
204
  expiresInMs,
198
- lastUsedMs: 0,
199
- lastRefreshMs: 0,
205
+ lastUsedMs: a.lastUsed,
206
+ lastRefreshMs: a.lastRefresh,
207
+ codexRateLimits: publicCodexRateLimits(a, routing.cooldowns),
208
+ ...(hasPendingCredentialWrite(a) ? { credentialsPendingWrite: true } : {}),
209
+ };
210
+ }
211
+ function publicCodexRateLimits(a, cooldowns) {
212
+ const rl = a.rateLimits;
213
+ // A bucket cooldown can exist without any snapshot for that bucket: a
214
+ // header-only 429 (an `x-codex-active-limit` with no accompanying
215
+ // rate-limit headers) sets a cooldown the pool enforces in `hardBlock`.
216
+ // Synthesizing a window-less entry for those keeps the health view honest —
217
+ // otherwise the account renders fully available while it is actually being
218
+ // skipped for that model.
219
+ const cooldownOnly = cooldowns.bucketCooldowns
220
+ .filter(cooldown => !rl.buckets.has(cooldown.limitId))
221
+ .map(cooldown => ({ limitId: cooldown.limitId }));
222
+ const buckets = [...rl.buckets.values(), ...cooldownOnly]
223
+ .sort((left, right) => left.limitId === DEFAULT_CODEX_LIMIT_ID ? -1
224
+ : right.limitId === DEFAULT_CODEX_LIMIT_ID ? 1
225
+ : left.limitId.localeCompare(right.limitId))
226
+ .slice(0, 8)
227
+ .map(bucket => ({
228
+ limitId: publicCodexLimitId(bucket.limitId),
229
+ label: publicCodexLabel(bucket),
230
+ ...(bucket.primary ? { primary: publicCodexWindow(bucket.primary) } : {}),
231
+ ...(bucket.secondary ? { secondary: publicCodexWindow(bucket.secondary) } : {}),
232
+ cooldownUntilMs: publicTimestamp(cooldowns.bucketCooldowns.find(c => c.limitId === bucket.limitId)?.untilMs ?? 0),
233
+ }));
234
+ const credits = rl.credits;
235
+ const balance = typeof credits?.balance === "string"
236
+ ? credits.balance.replace(/[\x00-\x1f\x7f]/g, "").trim().slice(0, 32)
237
+ : "";
238
+ return {
239
+ status: rl.status === "rate_limited" ? "rate_limited" : "ok",
240
+ plan: publicCodexPlan(rl.plan),
241
+ buckets,
242
+ ...(credits ? {
243
+ credits: {
244
+ hasCredits: credits.hasCredits === true,
245
+ unlimited: credits.unlimited === true,
246
+ ...(balance ? { balance } : {}),
247
+ },
248
+ } : {}),
249
+ lastUpdated: publicTimestamp(rl.lastUpdated),
250
+ };
251
+ }
252
+ function publicCodexWindow(window) {
253
+ return {
254
+ utilization: publicUtilization(window.utilization),
255
+ resetAt: publicTimestamp(window.resetAt),
256
+ windowMinutes: publicNonNegativeInteger(window.windowMinutes),
200
257
  };
201
258
  }
259
+ function publicCodexLimitId(value) {
260
+ return /^[a-z0-9_]{1,64}$/.test(value) ? value : "unknown";
261
+ }
262
+ function publicCodexLabel(bucket) {
263
+ const name = bucket.limitName?.replace(/[\u0000-\u001f\u007f]/g, "").trim().slice(0, 64);
264
+ return name || publicCodexLimitId(bucket.limitId);
265
+ }
266
+ function publicCodexPlan(value) {
267
+ return typeof value === "string" && /^[a-z0-9_-]{1,32}$/.test(value) ? value : "";
268
+ }
202
269
  function providerStatus(accounts) {
203
270
  return {
204
271
  configured: accounts.length > 0,
@@ -258,6 +325,23 @@ export function applyRateLimitHeaders(account, headers) {
258
325
  account.rateLimits = { ...account.rateLimits, ...rateLimits };
259
326
  return true;
260
327
  }
328
+ /**
329
+ * Build the single function through which this server writes OpenAI accounts.
330
+ *
331
+ * Two things have to be true of every such write, so they live together here
332
+ * rather than at each call site. It must land in the file the process was
333
+ * started against — a custom `--accounts <path>` must never silently fall back
334
+ * to the default accounts.json. And it must report its own durability: an add,
335
+ * patch, or delete rewrites the same file from the same live array, so it puts
336
+ * a rotation that failed to persist earlier on disk even though no refresh was
337
+ * involved, and the pending-write bookkeeping has to clear with it.
338
+ */
339
+ export function createOpenAIPersister(accountsPath) {
340
+ return (accountsToSave) => {
341
+ saveOpenAIAccountsToPath(accountsToSave, accountsPath ?? ACCOUNTS_PATH);
342
+ markOpenAICredentialsPersisted(accountsToSave);
343
+ };
344
+ }
261
345
  export async function startServer(opts = {}) {
262
346
  const port = opts.port ?? PROXY_PORT;
263
347
  // Direct-to-Anthropic (standalone) or via LiteLLM (full mode).
@@ -266,6 +350,7 @@ export async function startServer(opts = {}) {
266
350
  const target = litellmUrl ?? "https://api.anthropic.com";
267
351
  const mode = litellmUrl ? "litellm" : "standalone";
268
352
  const accountsPath = opts.accountsPath;
353
+ const persistOpenAIAccounts = createOpenAIPersister(accountsPath);
269
354
  if (!accountsFileExists(accountsPath)) {
270
355
  console.error(chalk.red("\n✗ accounts.json not found."));
271
356
  console.error(chalk.yellow(" Run: cc-router setup\n"));
@@ -273,7 +358,7 @@ export async function startServer(opts = {}) {
273
358
  }
274
359
  migrateLegacyAccountProviders(accountsPath);
275
360
  const accounts = accountsPath ? readAccountsFromPath(accountsPath) : loadAccounts();
276
- const openAIAccounts = loadOpenAIAccounts(accountsPath);
361
+ const openAIAccounts = loadOpenAIAccounts(accountsPath).map(createOpenAIAccount);
277
362
  if (accounts.length === 0 && openAIAccounts.length === 0) {
278
363
  console.error(chalk.red("\n✗ No accounts found in accounts.json."));
279
364
  console.error(chalk.yellow(" Run: cc-router setup\n"));
@@ -295,7 +380,28 @@ export async function startServer(opts = {}) {
295
380
  };
296
381
  };
297
382
  };
298
- const pickOpenAIAccount = createOpenAIAccountPicker(openAIAccounts);
383
+ const openAIPool = new OpenAITokenPool(openAIAccounts);
384
+ const openAIRouter = new SessionRouter(openAIPool);
385
+ // Factory, mirroring `createRoutingMetricsResolver`: the active-session
386
+ // snapshot is taken once per request rather than rebuilt (sweeping every
387
+ // binding and copying the map) for each account, and the cooldown view is
388
+ // computed once per account instead of three times.
389
+ const createOpenAIRoutingResolver = () => {
390
+ const activeSessionCounts = openAIRouter.getActiveSessionCountsSnapshot();
391
+ return (accountId) => {
392
+ const cooldowns = openAIPool.getCooldownView(accountId);
393
+ return {
394
+ metrics: {
395
+ inFlightRequests: openAIPool.getInFlight(accountId),
396
+ activeSessions: activeSessionCounts.get(accountId) ?? 0,
397
+ coolingDown: cooldowns.globalUntilMs > 0,
398
+ cooldownUntilMs: cooldowns.globalUntilMs,
399
+ },
400
+ cooldowns,
401
+ };
402
+ };
403
+ };
404
+ const resolveOpenAIRouting = (accountId) => createOpenAIRoutingResolver()(accountId);
299
405
  const initialConfig = readConfig();
300
406
  const modelRouting = initialConfig.modelRouting ?? {};
301
407
  // Log when the pool falls back to a capped account — makes the cap bypass
@@ -311,8 +417,15 @@ export async function startServer(opts = {}) {
311
417
  const msg = `${a.id} cooldown expired — rate limit cleared`;
312
418
  stats.addLog({ ts: Date.now(), accountId: a.id, model: "-", type: "route", details: msg });
313
419
  };
420
+ openAIPool.onCapBypass = (a) => {
421
+ const msg = `all OpenAI accounts capped — routing to ${a.id}`;
422
+ stats.addLog({ ts: Date.now(), accountId: a.id, model: "-", type: "error", details: msg });
423
+ };
424
+ openAIPool.onCooldownExpired = (a) => {
425
+ stats.addLog({ ts: Date.now(), accountId: a.id, model: "-", type: "route", details: `${a.id} cooldown expired — rate limit cleared` });
426
+ };
314
427
  startRefreshLoop(accounts);
315
- startOpenAIRefreshLoop(openAIAccounts, saveOpenAIAccounts);
428
+ startOpenAIRefreshLoop(openAIAccounts, persistOpenAIAccounts);
316
429
  const usageRefresher = new AnthropicUsageRefresher(pool);
317
430
  usageRefresher.start();
318
431
  const app = express();
@@ -364,8 +477,9 @@ export async function startServer(opts = {}) {
364
477
  // Sweep expired cooldowns on each poll so the dashboard reflects recovery
365
478
  // even during idle periods when no /v1 request would trigger getNext().
366
479
  pool.sweepExpiredCooldowns();
480
+ openAIPool.sweepExpiredCooldowns();
367
481
  const resolveRoutingMetrics = createRoutingMetricsResolver();
368
- const accountViews = createHealthAccountViews(pool.getAll(), openAIAccounts, resolveRoutingMetrics);
482
+ const accountViews = createHealthAccountViews(pool.getAll(), openAIAccounts, resolveRoutingMetrics, createOpenAIRoutingResolver());
369
483
  const status = accountViews.some(a => a.healthy) ? "ok" : "degraded";
370
484
  if (secretBuf && !secretMatches(presentedSecret(req))) {
371
485
  res.json({ status });
@@ -404,7 +518,7 @@ export async function startServer(opts = {}) {
404
518
  accountsRouter.get("/", (_req, res) => {
405
519
  const resolveRoutingMetrics = createRoutingMetricsResolver();
406
520
  res.json({
407
- accounts: createHealthAccountViews(pool.getAll(), openAIAccounts, resolveRoutingMetrics),
521
+ accounts: createHealthAccountViews(pool.getAll(), openAIAccounts, resolveRoutingMetrics, createOpenAIRoutingResolver()),
408
522
  });
409
523
  });
410
524
  accountsRouter.patch("/providers/:provider", (req, res) => {
@@ -448,12 +562,31 @@ export async function startServer(opts = {}) {
448
562
  };
449
563
  applyRuntime(enabled);
450
564
  try {
565
+ const isAnthropic = provider === "anthropic_subscription";
451
566
  const changed = persistProviderEnabledState({
452
567
  provider,
453
568
  enabled,
454
- accountIds: pool.getAll().map(account => account.id),
455
- persist: () => setProviderAccountsEnabled(provider, enabled, accountsPath),
456
- invalidateAccount: accountId => sessionRouter.invalidateAccount(accountId),
569
+ accountIds: isAnthropic
570
+ ? pool.getAll().map(account => account.id)
571
+ : openAIAccounts.map(account => account.id),
572
+ // OpenAI accounts are written from the live pool rather than by
573
+ // rewriting whatever is on disk. A refresh may have rotated
574
+ // credentials that never reached a file, and re-serializing that
575
+ // stale record would report this PATCH a success while the only
576
+ // valid refresh token stayed in memory — one crash from forcing a
577
+ // re-login. Going through the live persister saves it and clears the
578
+ // pending marker. Anthropic has no such in-memory rotation to lose.
579
+ persist: () => {
580
+ if (isAnthropic)
581
+ return setProviderAccountsEnabled(provider, enabled, accountsPath);
582
+ persistOpenAIAccounts(openAIAccounts);
583
+ // Same count `setProviderAccountsEnabled` reports, so the response
584
+ // keeps its shape for both providers: every account of this provider
585
+ // the write covered, not only those whose flag flipped.
586
+ return openAIAccounts.length;
587
+ },
588
+ invalidateAccount: accountId => (isAnthropic ? sessionRouter : openAIRouter)
589
+ .invalidateAccount(accountId),
457
590
  });
458
591
  res.json({ provider, enabled, changed });
459
592
  }
@@ -488,51 +621,69 @@ export async function startServer(opts = {}) {
488
621
  accountsRouter.patch("/:id", (req, res) => {
489
622
  const { id } = req.params;
490
623
  const body = (req.body ?? {});
491
- const patch = {};
492
- if (body.enabled !== undefined) {
493
- if (typeof body.enabled !== "boolean") {
494
- res.status(400).json({ error: "enabled must be boolean" });
624
+ const validation = validateAccountPatchBody(body);
625
+ if (!validation.ok) {
626
+ res.status(400).json({ error: validation.error });
627
+ return;
628
+ }
629
+ const patch = validation.patch;
630
+ // Snapshot the previous values so we can roll back on persistence failure
631
+ const existing = pool.findById(id);
632
+ if (existing) {
633
+ const prev = {
634
+ enabled: existing.enabled,
635
+ sessionLimitPercent: existing.sessionLimitPercent,
636
+ weeklyLimitPercent: existing.weeklyLimitPercent,
637
+ };
638
+ const updated = pool.updateAccount(id, patch);
639
+ if (!updated) {
640
+ res.status(404).json({ error: `Account "${id}" not found` });
495
641
  return;
496
642
  }
497
- patch.enabled = body.enabled;
498
- }
499
- for (const key of ["sessionLimitPercent", "weeklyLimitPercent"]) {
500
- const v = body[key];
501
- if (v === undefined)
502
- continue;
503
- if (typeof v !== "number" || !Number.isFinite(v) || v < 0 || v > 100) {
504
- res.status(400).json({ error: `${key} must be a number between 0 and 100` });
643
+ const result = tryPersist(() => {
644
+ pool.updateAccount(id, prev);
645
+ });
646
+ if (!result.ok) {
647
+ res.status(500).json({ error: `Failed to persist accounts.json: ${result.message}` });
505
648
  return;
506
649
  }
507
- patch[key] = v;
508
- }
509
- // Snapshot the previous values so we can roll back on persistence failure
510
- const existing = pool.findById(id);
511
- if (!existing) {
512
- res.status(404).json({ error: `Account "${id}" not found` });
650
+ if (patch.enabled === false)
651
+ sessionRouter.invalidateAccount(id);
652
+ res.json({
653
+ account: publicAnthropicAccountView(updated, createRoutingMetricsResolver()(updated.id)),
654
+ });
513
655
  return;
514
656
  }
515
- const prev = {
516
- enabled: existing.enabled,
517
- sessionLimitPercent: existing.sessionLimitPercent,
518
- weeklyLimitPercent: existing.weeklyLimitPercent,
519
- };
520
- const updated = pool.updateAccount(id, patch);
521
- if (!updated) {
522
- res.status(404).json({ error: `Account "${id}" not found` });
657
+ // Not a Claude account — try the OpenAI pool. `OpenAITokenPool` has no
658
+ // `updateAccount` of its own, so the runtime `OpenAIAccount` is patched
659
+ // and persisted directly via the same transaction contract used to add
660
+ // and delete OpenAI accounts.
661
+ let updatedOpenAI;
662
+ try {
663
+ updatedOpenAI = applyOpenAIAccountPatch({
664
+ id,
665
+ patch,
666
+ accounts: openAIAccounts,
667
+ persist: persistOpenAIAccounts,
668
+ });
669
+ }
670
+ catch (err) {
671
+ const message = err instanceof Error ? err.message : String(err);
672
+ logError("accounts", 0, `Failed to persist accounts.json: ${message}`);
673
+ res.status(500).json({ error: `Failed to persist accounts.json: ${message}` });
523
674
  return;
524
675
  }
525
- const result = tryPersist(() => {
526
- pool.updateAccount(id, prev);
527
- });
528
- if (!result.ok) {
529
- res.status(500).json({ error: `Failed to persist accounts.json: ${result.message}` });
676
+ if (!updatedOpenAI) {
677
+ res.status(404).json({ error: `Account "${id}" not found` });
530
678
  return;
531
679
  }
680
+ // Same contract as the Anthropic branch above: disabling an account drops
681
+ // its sticky bindings immediately instead of leaving them (and their
682
+ // active-session counts) attributed to it until the binding TTL expires.
532
683
  if (patch.enabled === false)
533
- sessionRouter.invalidateAccount(id);
684
+ openAIRouter.invalidateAccount(id);
534
685
  res.json({
535
- account: publicAnthropicAccountView(updated, createRoutingMetricsResolver()(updated.id)),
686
+ account: publicOpenAIAccountView(updatedOpenAI, resolveOpenAIRouting(updatedOpenAI.id)),
536
687
  });
537
688
  });
538
689
  accountsRouter.post("/", (req, res) => {
@@ -549,8 +700,18 @@ export async function startServer(opts = {}) {
549
700
  res.status(400).json({ error: "Invalid field types on account record" });
550
701
  return;
551
702
  }
703
+ // Same cap validation the PATCH endpoint applies, so the two writers of
704
+ // these fields agree instead of POST silently clamping a bad value to 100.
705
+ const capValidation = validateAccountPatchBody({
706
+ sessionLimitPercent: body.sessionLimitPercent,
707
+ weeklyLimitPercent: body.weeklyLimitPercent,
708
+ });
709
+ if (!capValidation.ok) {
710
+ res.status(400).json({ error: capValidation.error });
711
+ return;
712
+ }
552
713
  // IDs are unique across providers, so a new account may not collide with an
553
- // existing account in either the Claude pool or the OpenAI picker.
714
+ // existing account in either the Claude pool or the OpenAI pool.
554
715
  if (pool.findById(body.id) || openAIAccounts.some(a => a.id === body.id)) {
555
716
  res.status(409).json({ error: `Account "${body.id}" already exists` });
556
717
  return;
@@ -565,9 +726,11 @@ export async function startServer(opts = {}) {
565
726
  refreshToken: body.refreshToken,
566
727
  expiresAt: body.expiresAt,
567
728
  enabled: body.enabled,
729
+ sessionLimitPercent: body.sessionLimitPercent,
730
+ weeklyLimitPercent: body.weeklyLimitPercent,
568
731
  },
569
732
  accounts: openAIAccounts,
570
- persist: saveOpenAIAccounts,
733
+ persist: persistOpenAIAccounts,
571
734
  });
572
735
  }
573
736
  catch (err) {
@@ -576,7 +739,7 @@ export async function startServer(opts = {}) {
576
739
  res.status(500).json({ error: `Failed to persist accounts.json: ${message}` });
577
740
  return;
578
741
  }
579
- res.status(201).json({ account: publicOpenAIAccountView(addedOpenAI) });
742
+ res.status(201).json({ account: publicOpenAIAccountView(addedOpenAI, resolveOpenAIRouting(addedOpenAI.id)) });
580
743
  return;
581
744
  }
582
745
  const record = {
@@ -622,7 +785,9 @@ export async function startServer(opts = {}) {
622
785
  id,
623
786
  accounts: openAIAccounts,
624
787
  otherAccountCount: pool.getAll().length,
625
- persist: saveOpenAIAccounts,
788
+ persist: persistOpenAIAccounts,
789
+ forgetAccount: account => openAIPool.forgetAccount(account),
790
+ invalidateAccount: accountId => { openAIRouter.invalidateAccount(accountId); },
626
791
  });
627
792
  }
628
793
  catch (err) {
@@ -678,17 +843,28 @@ export async function startServer(opts = {}) {
678
843
  Object.assign(modelRouting, next);
679
844
  writeConfig({ ...readConfig(), modelRouting: next });
680
845
  },
681
- prepareOpenAIAccount: (account) => prepareOpenAIAccountForRequest(account, openAIAccounts, saveOpenAIAccounts),
846
+ prepareOpenAIAccount: (account) => prepareOpenAIAccountForRequest(account, openAIAccounts, persistOpenAIAccounts),
682
847
  });
848
+ // A relayed upstream 401 means the subscription token is stale — kick off a
849
+ // refresh in the background so the *next* request succeeds without making
850
+ // this client wait on it. Best-effort: failures are swallowed here since
851
+ // the ingress lifecycle has already recorded the 401 for this request.
852
+ const onOpenAIUpstreamAuthFailure = (account) => {
853
+ refreshAndPersistOpenAIAccount(account, openAIAccounts, persistOpenAIAccounts).catch(() => { });
854
+ };
683
855
  mountResponsesRoutes(app, {
684
- getOpenAIAccount: pickOpenAIAccount,
685
- prepareOpenAIAccount: (account) => prepareOpenAIAccountForRequest(account, openAIAccounts, saveOpenAIAccounts),
856
+ openAIRouter,
857
+ openAIPool,
858
+ prepareOpenAIAccount: (account) => prepareOpenAIAccountForRequest(account, openAIAccounts, persistOpenAIAccounts),
686
859
  modelRouting,
860
+ onUpstreamAuthFailure: onOpenAIUpstreamAuthFailure,
687
861
  });
688
862
  mountMessagesCrossProviderRoute(app, {
689
- getOpenAIAccount: pickOpenAIAccount,
690
- prepareOpenAIAccount: (account) => prepareOpenAIAccountForRequest(account, openAIAccounts, saveOpenAIAccounts),
863
+ openAIRouter,
864
+ openAIPool,
865
+ prepareOpenAIAccount: (account) => prepareOpenAIAccountForRequest(account, openAIAccounts, persistOpenAIAccounts),
691
866
  modelRouting,
867
+ onUpstreamAuthFailure: onOpenAIUpstreamAuthFailure,
692
868
  });
693
869
  // ─── Proxy middleware ──────────────────────────────────────────────────────
694
870
  // IMPORTANT: selfHandleResponse must be false (default) for SSE streaming to
@@ -1,9 +1,27 @@
1
+ /**
2
+ * Longest model identifier retained in an activity entry. Model names arrive
3
+ * in request bodies the JSON parsers accept up to megabytes, and every entry
4
+ * stays resident until MAX_LOG_ENTRIES newer ones push it out — and is
5
+ * re-serialized into each health response meanwhile. Retaining one verbatim
6
+ * would let a handful of requests pin hundreds of megabytes. Real model names
7
+ * are far shorter than this; the same 64 that `normalizeModelSlug` and
8
+ * `normalizeModelFamily` already clamp their identifiers to.
9
+ */
10
+ const MAX_LOG_MODEL_LENGTH = 64;
11
+ /**
12
+ * Clamp a caller-supplied model identifier to a length that is safe to retain.
13
+ * Truncation only — an empty model stays empty so routing lookups behave
14
+ * exactly as they did on the untruncated value.
15
+ */
16
+ export function boundModelId(model) {
17
+ return model.length > MAX_LOG_MODEL_LENGTH ? model.slice(0, MAX_LOG_MODEL_LENGTH) : model;
18
+ }
1
19
  /** Build a bounded diagnostic for a request rejected before account selection. */
2
20
  export function createLocalRoutingErrorLog(reason, modelFamily, now = Date.now()) {
3
21
  return {
4
22
  ts: now,
5
23
  accountId: "proxy",
6
- model: modelFamily ?? "-",
24
+ model: modelFamily ? boundModelId(modelFamily) : "-",
7
25
  type: "error",
8
26
  details: `no-eligible:${reason.replace("_", "-")}`,
9
27
  statusCode: reason === "rate_limited" ? 429 : 503,
@@ -34,3 +52,14 @@ class ProxyStats {
34
52
  }
35
53
  // Singleton — shared across server and health endpoint
36
54
  export const stats = new ProxyStats();
55
+ /** Record Codex token usage on both the request's log entry and the running totals. */
56
+ export function applyCodexUsage(entry, usage) {
57
+ if (!usage)
58
+ return;
59
+ entry.inputTokens = usage.inputTokens;
60
+ entry.outputTokens = usage.outputTokens;
61
+ entry.cacheReadTokens = usage.cachedInputTokens;
62
+ stats.totalInputTokens += usage.inputTokens;
63
+ stats.totalOutputTokens += usage.outputTokens;
64
+ stats.totalCacheReadTokens += usage.cachedInputTokens;
65
+ }
@@ -1,24 +1,8 @@
1
1
  import { DEFAULT_RATE_LIMITS, ACCOUNT_USER_DEFAULTS, clampPercent } from "./types.js";
2
2
  import { canUseExtraUsage, normalizeModelFamily } from "../providers/anthropic/usage.js";
3
- export class EmptyPoolError extends Error {
4
- constructor(message) {
5
- super(message);
6
- this.name = "EmptyPoolError";
7
- }
8
- }
9
- export class NoEligibleAccountError extends Error {
10
- reason;
11
- retryAtMs;
12
- blockedAccounts;
13
- constructor(reason, blockedAccounts, retryAtMs) {
14
- super("no account is currently eligible for routing");
15
- this.name = "NoEligibleAccountError";
16
- this.reason = reason;
17
- this.blockedAccounts = blockedAccounts;
18
- if (retryAtMs !== undefined)
19
- this.retryAtMs = retryAtMs;
20
- }
21
- }
3
+ import { EmptyPoolError, NoEligibleAccountError, } from "./account-pool.js";
4
+ // Re-export so existing importers (anthropic-routing.ts, tests) keep working.
5
+ export { EmptyPoolError, NoEligibleAccountError };
22
6
  const MAX_TRUSTED_RATE_LIMIT_RESET_MS = 8 * 24 * 60 * 60 * 1_000;
23
7
  /**
24
8
  * Returns the reset timestamp (seconds) that must pass before the account