@timo972/cc-router 0.12.0 → 0.12.1-rc.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,6 +6,7 @@ import { stats, boundModelId, createLocalRoutingErrorLog } from "./stats.js";
6
6
  import { logError, logRoute } from "./logger.js";
7
7
  import { EmptyPoolError, NoEligibleAccountError } from "./account-pool.js";
8
8
  import { acquireRequestRoute, routeReasonDetails, routeFailureDetails } from "./lease-lifecycle.js";
9
+ import { createCorrelationId, formatTransportDiagnostic, safeCauseCode } from "./transport-diagnostics.js";
9
10
  import { MAX_UPSTREAM_ATTEMPTS, RETRY_REFRESH_TIMEOUT_MS, SAME_ACCOUNT_RETRY_DELAY_MS, boundedWait, isRetryableUpstreamStatus, retryDelay, } from "./upstream-retry.js";
10
11
  /**
11
12
  * Mirrors `anthropic-routing.ts`'s `requestTerminated` check. This ingress
@@ -100,13 +101,14 @@ export function mirrorUpstreamHeaders(source, apply) {
100
101
  * an unhandled rejection behind.
101
102
  */
102
103
  export async function runOpenAIIngress(opts) {
103
- const { res, sessionKey, path, openAIRouter, openAIPool, prepareOpenAIAccount, forwardOpenAI, forwardBody, recordActivity, now, envelope, relay, onUpstreamAuthFailure, } = opts;
104
+ const { res, sessionKey, path, openAIRouter, openAIPool, prepareOpenAIAccount, forwardOpenAI, forwardBody, recordActivity, now, envelope, relay, onUpstreamAuthFailure, timeoutMs, } = opts;
104
105
  // The model comes from a client-controlled body and is retained in the
105
106
  // activity ring buffer below. Bound it once, here, so every activity entry,
106
107
  // routing context and bucket lookup on this path carries an identifier that
107
108
  // cannot grow with the request. The body forwarded upstream is untouched —
108
109
  // it still carries whatever model the caller asked for.
109
110
  const requestedModel = boundModelId(opts.requestedModel);
111
+ const correlationId = createCorrelationId();
110
112
  // A client that hangs up must take the upstream request with it. Releasing
111
113
  // the lease (which the response's own close listener does) only returns the
112
114
  // account's *local* capacity — without this the Codex request keeps
@@ -145,8 +147,11 @@ export async function runOpenAIIngress(opts) {
145
147
  // this handler's promise — no account lease was taken, so there is
146
148
  // nothing to release.
147
149
  stats.totalErrors++;
148
- const message = error instanceof Error ? error.message : String(error);
149
- logError("proxy", 500, `unexpected routing failure: ${message}`);
150
+ logError("proxy", 500, formatTransportDiagnostic({
151
+ correlationId,
152
+ operation: "forward",
153
+ causeCode: safeCauseCode(error),
154
+ }));
150
155
  recordActivity({
151
156
  ts: now(),
152
157
  accountId: "proxy",
@@ -155,6 +160,7 @@ export async function runOpenAIIngress(opts) {
155
160
  statusCode: 500,
156
161
  path,
157
162
  details: "proxy_error:acquire",
163
+ correlationId,
158
164
  });
159
165
  res.status(500).json(envelope.wrap("proxy_error", "Unexpected routing error"));
160
166
  return;
@@ -166,6 +172,8 @@ export async function runOpenAIIngress(opts) {
166
172
  // reason string — keep it that way.
167
173
  logRoute(selected.route.account.id, selected.route.account.requestCount, Math.round((selected.route.account.expiresAt - now()) / 60_000), selected.details);
168
174
  const startedAt = now();
175
+ const refreshDurations = new Map();
176
+ let headerDurationMs = 0;
169
177
  const maxAttempts = Math.max(1, opts.maxAttempts ?? MAX_UPSTREAM_ATTEMPTS);
170
178
  const sameAccountDelayMs = opts.sameAccountRetryDelayMs ?? SAME_ACCOUNT_RETRY_DELAY_MS;
171
179
  const retryRefreshTimeoutMs = opts.retryRefreshTimeoutMs ?? RETRY_REFRESH_TIMEOUT_MS;
@@ -178,6 +186,8 @@ export async function runOpenAIIngress(opts) {
178
186
  const prepareRoute = async (routed) => {
179
187
  const account = routed.route.account;
180
188
  const needed = needsOpenAIRefresh(account);
189
+ const refreshStartedAt = now();
190
+ let refreshCauseCode;
181
191
  let ready;
182
192
  try {
183
193
  ready = await prepareOpenAIAccount(account);
@@ -185,14 +195,13 @@ export async function runOpenAIIngress(opts) {
185
195
  catch (error) {
186
196
  // A throwing refresh must behave exactly like a `false` return, never
187
197
  // crash the request (or the daemon).
188
- const message = error instanceof Error ? error.message : String(error);
189
- logError(account.id, 401, `openai token refresh threw: ${message}`);
198
+ refreshCauseCode = safeCauseCode(error);
190
199
  ready = false;
191
200
  }
201
+ const refreshDurationMs = now() - refreshStartedAt;
202
+ refreshDurations.set(account, refreshDurationMs);
192
203
  if (!ready) {
193
204
  routed.release();
194
- account.errorCount++;
195
- stats.totalErrors++;
196
205
  // Intentionally does not touch `account.healthy`: a single failed
197
206
  // refresh fails only this request. Disabling the account here would
198
207
  // hard-block it from every future request until a manual recovery, even
@@ -206,6 +215,13 @@ export async function runOpenAIIngress(opts) {
206
215
  openAIRouter.invalidate(routed.route.sessionId, account.id, routed.route.bindingGeneration);
207
216
  }
208
217
  openAIPool.setGlobalCooldownForAccount(account, REFRESH_FAILURE_COOLDOWN_MS, "unavailable");
218
+ if (clientGone.signal.aborted || responseTerminated(res))
219
+ return false;
220
+ account.errorCount++;
221
+ stats.totalErrors++;
222
+ logError(account.id, 401, formatTransportDiagnostic({
223
+ correlationId, operation: "prepare", status: 401, causeCode: refreshCauseCode,
224
+ }));
209
225
  recordActivity({
210
226
  ts: now(),
211
227
  accountId: account.id,
@@ -213,7 +229,9 @@ export async function runOpenAIIngress(opts) {
213
229
  type: "error",
214
230
  statusCode: 401,
215
231
  path,
216
- details: "openai token refresh failed",
232
+ details: `openai token refresh failed;correlation=${correlationId};auth=${account.authFailure ?? "unknown"}`,
233
+ correlationId,
234
+ refreshDurationMs,
217
235
  });
218
236
  return false;
219
237
  }
@@ -223,6 +241,8 @@ export async function runOpenAIIngress(opts) {
223
241
  return true;
224
242
  };
225
243
  if (!(await prepareRoute(selected))) {
244
+ if (clientGone.signal.aborted || responseTerminated(res))
245
+ return;
226
246
  res.status(401).json(envelope.wrap("authentication_error", "OpenAI subscription token refresh failed"));
227
247
  return;
228
248
  }
@@ -248,6 +268,7 @@ export async function runOpenAIIngress(opts) {
248
268
  body: forwardBody,
249
269
  stream: forwardBody.stream === true,
250
270
  signal: clientGone.signal,
271
+ timeoutMs,
251
272
  });
252
273
  }
253
274
  catch (error) {
@@ -267,21 +288,28 @@ export async function runOpenAIIngress(opts) {
267
288
  // own finish/close lifecycle once this response is sent.
268
289
  account.errorCount++;
269
290
  stats.totalErrors++;
270
- const message = error instanceof Error ? error.message : String(error);
271
- logError(account.id, 502, `openai request failed: ${message}`);
291
+ const timeout = error instanceof Error && error.name === "TimeoutError";
292
+ const diagnostic = formatTransportDiagnostic({
293
+ correlationId, operation: "forward", causeCode: safeCauseCode(error),
294
+ });
295
+ logError(account.id, timeout ? 504 : 502, diagnostic);
272
296
  recordActivity({
273
297
  ts: startedAt,
274
298
  accountId: account.id,
275
299
  model: requestedModel,
276
300
  type: "error",
277
- statusCode: 502,
301
+ statusCode: timeout ? 504 : 502,
278
302
  path,
279
- details: "upstream_error:network",
303
+ details: `upstream_error:${timeout ? "timeout" : "network"};${diagnostic}`,
280
304
  durationMs: now() - startedAt,
305
+ correlationId,
306
+ refreshDurationMs: refreshDurations.get(account),
307
+ headerDurationMs: now() - attemptStartedAt,
281
308
  });
282
- res.status(502).json(envelope.wrap("upstream_error", `OpenAI request failed: ${message}`));
309
+ res.status(timeout ? 504 : 502).json(envelope.wrap("upstream_error", timeout ? "OpenAI request timed out" : "OpenAI request failed"));
283
310
  return;
284
311
  }
312
+ headerDurationMs = now() - attemptStartedAt;
285
313
  // Cooldown/eligibility react to the raw upstream signal — this must not
286
314
  // change based on how the relay later renders the response to the client.
287
315
  upstreamFailed = upstream.status === 401 || upstream.status === 429 || upstream.status >= 500;
@@ -314,11 +342,15 @@ export async function runOpenAIIngress(opts) {
314
342
  : "upstream-error", applied.limitingScope);
315
343
  if (upstream.status === 401)
316
344
  onUpstreamAuthFailure?.(account);
345
+ logError(account.id, upstream.status, formatTransportDiagnostic({
346
+ correlationId, operation: "upstream", status: upstream.status,
347
+ }));
317
348
  }
318
349
  }
319
350
  catch (error) {
320
- const message = error instanceof Error ? error.message : String(error);
321
- logError(account.id, upstream.status, `openai response classification failed: ${message}`);
351
+ logError(account.id, upstream.status, formatTransportDiagnostic({
352
+ correlationId, operation: "upstream", status: upstream.status, causeCode: safeCauseCode(error),
353
+ }));
322
354
  }
323
355
  // Retryable statuses (429 || >= 500) are a strict subset of
324
356
  // `upstreamFailed`, so this predicate alone decides the loop.
@@ -380,6 +412,9 @@ export async function runOpenAIIngress(opts) {
380
412
  path,
381
413
  ...(opts.method !== undefined ? { method: opts.method } : {}),
382
414
  ...(opts.source !== undefined ? { source: opts.source } : {}),
415
+ correlationId,
416
+ refreshDurationMs: refreshDurations.get(account),
417
+ headerDurationMs,
383
418
  details: `${details}:will-retry`,
384
419
  durationMs: now() - attemptStartedAt,
385
420
  });
@@ -405,12 +440,15 @@ export async function runOpenAIIngress(opts) {
405
440
  path,
406
441
  ...(opts.method !== undefined ? { method: opts.method } : {}),
407
442
  ...(opts.source !== undefined ? { source: opts.source } : {}),
443
+ refreshDurationMs: refreshDurations.get(account),
444
+ headerDurationMs,
445
+ correlationId,
408
446
  details,
409
447
  };
410
448
  let finalStatus = upstream.status;
411
449
  let relayFailed = false;
412
450
  let relayFailureMessage = "";
413
- const relayReport = { upstreamReportedFailure: false };
451
+ const relayReport = { upstreamReportedFailure: false, routerFailure: false };
414
452
  try {
415
453
  const result = await relay(upstream, res, entry, relayReport);
416
454
  finalStatus = result.statusCode;
@@ -421,8 +459,17 @@ export async function runOpenAIIngress(opts) {
421
459
  // otherwise the client already has a partial response and the best we
422
460
  // can do is tear the connection down.
423
461
  relayFailed = true;
424
- const message = error instanceof Error ? error.message : String(error);
425
- relayFailureMessage = message;
462
+ // The same reader rejection is produced by two very different events:
463
+ // an upstream/idle failure, and the caller closing its socket. Capture
464
+ // the origin before any router-side destroy emits another `close`.
465
+ const downstreamAlreadyGone = clientGone.signal.aborted || responseTerminated(res);
466
+ relayReport.routerFailure = !downstreamAlreadyGone;
467
+ relayFailureMessage = formatTransportDiagnostic({
468
+ correlationId,
469
+ operation: "relay",
470
+ causeCode: error instanceof Error && error.name === "TimeoutError"
471
+ ? "UPSTREAM_IDLE_TIMEOUT" : safeCauseCode(error),
472
+ });
426
473
  // The recorded status is what this request *became*, which is a failure
427
474
  // whether or not another HTTP response can still be sent. Leaving it at
428
475
  // the upstream's 200 in the headers-already-sent case produced an
@@ -430,8 +477,11 @@ export async function runOpenAIIngress(opts) {
430
477
  // that contradicts itself, and one that reads as a success in any view
431
478
  // that keys off the status.
432
479
  finalStatus = 502;
433
- if (!res.headersSent) {
434
- res.status(502).json(envelope.wrap("upstream_error", `OpenAI response relay failed: ${message}`));
480
+ if (downstreamAlreadyGone) {
481
+ // There is no peer left to receive an error response.
482
+ }
483
+ else if (!res.headersSent) {
484
+ res.status(502).json(envelope.wrap("upstream_error", "OpenAI response relay failed"));
435
485
  }
436
486
  else {
437
487
  if (!res.writableEnded && !res.destroyed)
@@ -451,7 +501,9 @@ export async function runOpenAIIngress(opts) {
451
501
  // a 200 stream — the client can truncate a stream, but it cannot make
452
502
  // upstream announce a failure. Only the truncation is the disconnect's to
453
503
  // explain away.
454
- const clientCancelled = clientGone.signal.aborted && !relayReport.upstreamReportedFailure;
504
+ const clientCancelled = clientGone.signal.aborted
505
+ && !relayReport.upstreamReportedFailure
506
+ && !relayReport.routerFailure;
455
507
  // Logged only now, after the cancellation classification: the Codex CLI
456
508
  // aborts streams routinely (a superseded turn, Ctrl-C, a pane closing),
457
509
  // and each abort rejects the relay's body read with "This operation was
@@ -459,7 +511,8 @@ export async function runOpenAIIngress(opts) {
459
511
  // every one — eight hours of them in one unattended overnight session —
460
512
  // while the stats correctly ignored them. The log now matches the stats.
461
513
  if (relayFailed && !clientCancelled) {
462
- logError(account.id, 502, `openai response relay failed: ${relayFailureMessage}`);
514
+ logError(account.id, 502, relayFailureMessage);
515
+ entry.details = details ? `${details};${relayFailureMessage}` : relayFailureMessage;
463
516
  }
464
517
  // Activity/stats must reflect what the client actually received, not just
465
518
  // the raw upstream signal: the non-streaming collector can synthesize a
@@ -0,0 +1,80 @@
1
+ /**
2
+ * One operator-triggered "reload everything" pass over the live pools —
3
+ * what a router restart gives you, without dropping in-flight requests or
4
+ * sticky sessions: expired cooldowns are swept, due (or quarantined) tokens
5
+ * go through the same refresh path the background loops use, and every
6
+ * account's usage is re-fetched immediately instead of waiting for its
7
+ * scheduled slot.
8
+ *
9
+ * Provider-agnostic on purpose: each provider hands in its own token refresh
10
+ * and usage hook, so the endpoint never learns OAuth or usage details.
11
+ */
12
+ export async function refreshAllAccounts(sources, options = {}) {
13
+ const now = options.now ?? Date.now;
14
+ const startedAt = now();
15
+ options.sweepCooldowns?.();
16
+ const providers = await Promise.all(sources.map(async (source) => {
17
+ let tokenRefreshFailed = 0;
18
+ try {
19
+ tokenRefreshFailed = Math.max(0, (await source.refreshTokens()).failed);
20
+ }
21
+ catch (error) {
22
+ tokenRefreshFailed = source.getAll().length;
23
+ options.onError?.(source.provider, error);
24
+ }
25
+ // Snapshot AFTER the token pass — a refresh may have quarantined or
26
+ // recovered accounts — and as a copy: the pools hand out their live
27
+ // array, which an add/delete can mutate while usage fetches settle.
28
+ const accounts = [...source.getAll()];
29
+ const outcomes = await Promise.allSettled(accounts.map(account => source.refreshUsage(account)));
30
+ let usageRefreshed = 0;
31
+ for (const outcome of outcomes) {
32
+ if (outcome.status === "fulfilled" && outcome.value.ok)
33
+ usageRefreshed++;
34
+ else if (outcome.status === "rejected")
35
+ options.onError?.(source.provider, outcome.reason);
36
+ }
37
+ return {
38
+ provider: source.provider,
39
+ accounts: accounts.length,
40
+ usageRefreshed,
41
+ usageFailed: accounts.length - usageRefreshed,
42
+ tokenRefreshFailed,
43
+ };
44
+ }));
45
+ const total = (pick) => providers.reduce((sum, p) => sum + pick(p), 0);
46
+ return {
47
+ accounts: total(p => p.accounts),
48
+ usageRefreshed: total(p => p.usageRefreshed),
49
+ usageFailed: total(p => p.usageFailed),
50
+ tokenRefreshFailed: total(p => p.tokenRefreshFailed),
51
+ durationMs: Math.max(0, now() - startedAt),
52
+ providers,
53
+ };
54
+ }
55
+ /**
56
+ * Single-flight wrapper: a second reload request while one is running joins
57
+ * the running one instead of starting another pass. Needed because the
58
+ * dashboard's client deadline is shorter than the worst case of many
59
+ * accounts all timing out — a client that gave up and pressed again must
60
+ * not stack a second sweep on the pools.
61
+ */
62
+ export function createRefreshAllRunner(run) {
63
+ let inFlight = null;
64
+ return () => {
65
+ if (inFlight)
66
+ return inFlight;
67
+ const operation = run().finally(() => {
68
+ if (inFlight === operation)
69
+ inFlight = null;
70
+ });
71
+ inFlight = operation;
72
+ return operation;
73
+ };
74
+ }
75
+ /** Human-readable one-liner for the activity log and the dashboard banner. */
76
+ export function describeRefreshAll(summary) {
77
+ const failed = summary.usageFailed > 0 ? `, ${summary.usageFailed} usage fetch${summary.usageFailed === 1 ? "" : "es"} failed` : "";
78
+ const tokens = summary.tokenRefreshFailed > 0 ? `, ${summary.tokenRefreshFailed} token refresh${summary.tokenRefreshFailed === 1 ? "" : "es"} failed` : "";
79
+ return `reloaded ${summary.accounts} account${summary.accounts === 1 ? "" : "s"}, usage fresh for ${summary.usageRefreshed}${failed}${tokens} (${summary.durationMs}ms)`;
80
+ }
@@ -6,6 +6,7 @@ import { stats, applyCodexUsage, boundModelId } from "./stats.js";
6
6
  import { logWarn } from "./logger.js";
7
7
  import { extractCodexSessionKey, sendOpenAINoEligibleResponse } from "./openai-routing.js";
8
8
  import { runOpenAIIngress, mirrorUpstreamHeaders, } from "./openai-ingress.js";
9
+ import { waitForWritable } from "./transport-timing.js";
9
10
  const RESPONSES_ENVELOPE = {
10
11
  wrap: (type, message) => ({ error: { type, message } }),
11
12
  sendNoEligible: (error, res, nowMs) => sendOpenAINoEligibleResponse(error, res, nowMs),
@@ -38,30 +39,45 @@ async function sendUpstreamResponse(upstream, res, onChunk) {
38
39
  // releases a stalled stream, so the account's upstream slot is not held for
39
40
  // a response nobody will receive.
40
41
  const DISCONNECTED = Symbol("client-disconnected");
42
+ let resolveDisconnected;
43
+ const onClose = () => resolveDisconnected?.(DISCONNECTED);
41
44
  const disconnected = new Promise(resolve => {
45
+ resolveDisconnected = resolve;
42
46
  if (res.destroyed)
43
47
  resolve(DISCONNECTED);
44
48
  else
45
- res.once("close", () => resolve(DISCONNECTED));
49
+ res.once("close", onClose);
46
50
  });
51
+ let completed = false;
47
52
  try {
48
53
  while (true) {
49
54
  const next = await Promise.race([reader.read(), disconnected]);
50
55
  if (next === DISCONNECTED) {
51
56
  await reader.cancel().catch(() => { });
52
- break;
57
+ return;
53
58
  }
54
59
  const { value, done } = next;
55
60
  if (done)
56
61
  break;
57
62
  if (value) {
58
- res.write(Buffer.from(value));
59
63
  onChunk?.(value);
64
+ if (!res.write(Buffer.from(value))) {
65
+ // Backpressure is a contract: do not consume another upstream chunk
66
+ // until Node drains, or until the peer goes away.
67
+ const drained = await waitForWritable(res);
68
+ if (!drained) {
69
+ await reader.cancel().catch(() => { });
70
+ return;
71
+ }
72
+ }
60
73
  }
61
74
  }
75
+ completed = true;
62
76
  }
63
77
  finally {
64
- res.end();
78
+ res.removeListener("close", onClose);
79
+ if (completed && !res.destroyed && !res.writableEnded)
80
+ res.end();
65
81
  }
66
82
  }
67
83
  export function mountResponsesRoutes(app, opts) {
@@ -137,6 +153,7 @@ export function mountResponsesRoutes(app, opts) {
137
153
  now,
138
154
  envelope: RESPONSES_ENVELOPE,
139
155
  onUpstreamAuthFailure: opts.onUpstreamAuthFailure,
156
+ timeoutMs: opts.timeoutMs,
140
157
  ...(opts.maxAttempts !== undefined ? { maxAttempts: opts.maxAttempts } : {}),
141
158
  ...(opts.sameAccountRetryDelayMs !== undefined
142
159
  ? { sameAccountRetryDelayMs: opts.sameAccountRetryDelayMs }
@@ -160,6 +177,8 @@ export function mountResponsesRoutes(app, opts) {
160
177
  // (or be cut short) after upstream has already announced a failure,
161
178
  // and the verdict has to survive that.
162
179
  await sendUpstreamResponse(upstream, res, chunk => {
180
+ if (entry.firstByteDurationMs === undefined)
181
+ entry.firstByteDurationMs = now() - entry.ts;
163
182
  observer.push(chunk);
164
183
  if (observer.explicitFailure() !== undefined)
165
184
  report.upstreamReportedFailure = true;
@@ -3,8 +3,9 @@ import { createProxyMiddleware } from "http-proxy-middleware";
3
3
  import { ServerResponse } from "http";
4
4
  import { timingSafeEqual } from "crypto";
5
5
  import { TokenPool } from "./token-pool.js";
6
- import { needsRefresh, refreshAccountIfCurrent, saveAccounts, startRefreshLoop } from "./token-refresher.js";
6
+ import { needsRefresh, refreshAccountIfCurrent, refreshAccountsOnce, saveAccounts, startRefreshLoop } from "./token-refresher.js";
7
7
  import { loadAccounts, loadOpenAIAccounts, saveOpenAIAccountsToPath, accountsFileExists, readAccountsFromPath, readConfig, writeConfig, getAutoFailoverEnabled, getProxyRequestTimeoutMs, migrateLegacyAccountProviders, setProviderAccountsEnabled, upsertAccountRecord, removeAccountRecordById } from "../config/manager.js";
8
+ import { createRefreshAllRunner, describeRefreshAll, refreshAllAccounts } from "./pool-refresh.js";
8
9
  import { checkForUpdate, performUpdate, restartSelf, printUpdateBanner, getCurrentVersion } from "../utils/self-update.js";
9
10
  import { trackEvent, startHeartbeat } from "../utils/telemetry.js";
10
11
  import { loadTelemetryState } from "../config/telemetry.js";
@@ -17,7 +18,7 @@ import { loadGrokHealthSnapshots } from "../providers/xai/overview.js";
17
18
  import { writePid, removePid, managesPidFile } from "../daemon/pid.js";
18
19
  import { applyOpenAIAccountPatch, validateAccountPatchBody } from "./account-patch.js";
19
20
  import { AccountRenameConflictError, renameAccountTransaction } from "./account-rename.js";
20
- import { hasPendingCredentialWrite, markOpenAICredentialsPersisted, prepareOpenAIAccountForRequest, refreshAndPersistOpenAIAccount, startOpenAIRefreshLoop, } from "../providers/openai/token-refresher.js";
21
+ import { hasPendingCredentialWrite, markOpenAICredentialsPersisted, prepareOpenAIAccountForRequest, refreshOpenAIAccountsOnce, refreshAndPersistOpenAIAccount, startOpenAIRefreshLoop, } from "../providers/openai/token-refresher.js";
21
22
  import { createOpenAIAccount } from "../providers/openai/account-state.js";
22
23
  import { OpenAITokenPool } from "../providers/openai/token-pool.js";
23
24
  import { DEFAULT_CODEX_LIMIT_ID } from "../providers/openai/usage.js";
@@ -61,6 +62,7 @@ export function createOperationalStatus(opts) {
61
62
  health: "/cc-router/health",
62
63
  accounts: "/cc-router/accounts",
63
64
  allowance: "/cc-router/allowance",
65
+ refresh: "/cc-router/refresh",
64
66
  messages: "/v1/messages",
65
67
  responses: "/v1/responses",
66
68
  models: "/v1/models",
@@ -222,7 +224,7 @@ function publicOpenAIAccountView(a, routing) {
222
224
  enabled: a.enabled !== false,
223
225
  sessionLimitPercent: a.sessionLimitPercent,
224
226
  weeklyLimitPercent: a.weeklyLimitPercent,
225
- healthy: a.enabled !== false && a.healthy && expiresInMs > 0,
227
+ healthy: a.enabled !== false && a.healthy && a.authState !== "quarantined" && expiresInMs > 0,
226
228
  busy: routing.metrics.coolingDown,
227
229
  cooldownUntilMs: routing.metrics.cooldownUntilMs ?? 0,
228
230
  globalCooldownUntilMs: routing.cooldowns.globalUntilMs,
@@ -235,6 +237,8 @@ function publicOpenAIAccountView(a, routing) {
235
237
  lastRefreshMs: a.lastRefresh,
236
238
  codexRateLimits: publicCodexRateLimits(a, routing.cooldowns),
237
239
  ...(hasPendingCredentialWrite(a) ? { credentialsPendingWrite: true } : {}),
240
+ ...(a.authState === "quarantined" ? { authState: "quarantined" } : {}),
241
+ ...(a.authFailure ? { authFailure: a.authFailure } : {}),
238
242
  };
239
243
  }
240
244
  function publicCodexRateLimits(a, cooldowns) {
@@ -532,6 +536,59 @@ export async function startServer(opts = {}) {
532
536
  const views = createHealthAccountViews(pool.getAll(), openAIAccounts, resolveRoutingMetrics, createOpenAIRoutingResolver(), loadGrokHealthSnapshots());
533
537
  res.json(createAllowanceView(views, Date.now()));
534
538
  });
539
+ // ─── Manual reload (authenticated) ────────────────────────────────────────
540
+ // The dashboard's "reload everything" key. Gives the operator what a
541
+ // restart would — swept cooldowns, due/quarantined tokens re-tried, every
542
+ // account's usage re-fetched — without dropping in-flight requests or
543
+ // sticky sessions. Each provider contributes its own hooks; the route
544
+ // itself knows nothing about OAuth or usage formats.
545
+ const runRefreshAll = createRefreshAllRunner(() => {
546
+ const onError = (provider, error) => {
547
+ const message = error instanceof Error ? error.message : String(error);
548
+ logError(provider, 0, `manual refresh: ${message}`);
549
+ };
550
+ return refreshAllAccounts([
551
+ {
552
+ provider: "anthropic",
553
+ getAll: () => pool.getAll(),
554
+ refreshTokens: async () => {
555
+ let failed = 0;
556
+ await refreshAccountsOnce(pool.getAll(), { onError: error => { failed++; onError("anthropic", error); } });
557
+ return { failed };
558
+ },
559
+ refreshUsage: account => usageRefresher.refreshNow(account),
560
+ },
561
+ {
562
+ provider: "openai",
563
+ getAll: () => openAIAccounts,
564
+ refreshTokens: () => refreshOpenAIAccountsOnce(openAIAccounts, persistOpenAIAccounts, {
565
+ onError: error => onError("openai", error),
566
+ }),
567
+ refreshUsage: account => openAIUsageRefresher.refreshNow(account),
568
+ },
569
+ ], {
570
+ sweepCooldowns: () => {
571
+ pool.sweepExpiredCooldowns();
572
+ openAIPool.sweepExpiredCooldowns();
573
+ },
574
+ onError,
575
+ });
576
+ });
577
+ app.post("/cc-router/refresh", async (_req, res) => {
578
+ let summary;
579
+ try {
580
+ summary = await runRefreshAll();
581
+ }
582
+ catch (err) {
583
+ const message = err instanceof Error ? err.message : String(err);
584
+ logError("refresh", 0, `manual refresh failed: ${message}`);
585
+ res.status(500).json({ error: `Refresh failed: ${message}` });
586
+ return;
587
+ }
588
+ const details = `manual refresh — ${describeRefreshAll(summary)}`;
589
+ stats.addLog({ ts: Date.now(), accountId: "proxy", model: "-", type: "refresh", details });
590
+ res.json({ refresh: summary });
591
+ });
535
592
  // ─── Account management endpoints (authenticated) ─────────────────────────
536
593
  // These are mounted BEFORE the /v1/* proxy middleware so they don't get
537
594
  // forwarded to Anthropic. express.json() is scoped to this sub-router so
@@ -989,6 +1046,7 @@ export async function startServer(opts = {}) {
989
1046
  prepareOpenAIAccount: (account) => prepareOpenAIAccountForRequest(account, openAIAccounts, persistOpenAIAccounts),
990
1047
  modelRouting,
991
1048
  onUpstreamAuthFailure: onOpenAIUpstreamAuthFailure,
1049
+ timeoutMs: proxyRequestTimeoutMs,
992
1050
  ...upstreamAttempts,
993
1051
  });
994
1052
  mountMessagesCrossProviderRoute(app, {
@@ -997,6 +1055,7 @@ export async function startServer(opts = {}) {
997
1055
  prepareOpenAIAccount: (account) => prepareOpenAIAccountForRequest(account, openAIAccounts, persistOpenAIAccounts),
998
1056
  modelRouting,
999
1057
  onUpstreamAuthFailure: onOpenAIUpstreamAuthFailure,
1058
+ timeoutMs: proxyRequestTimeoutMs,
1000
1059
  ...upstreamAttempts,
1001
1060
  });
1002
1061
  // Shared between the retrying /v1/messages route and the generic /v1 chain
@@ -0,0 +1,32 @@
1
+ import { randomUUID } from "node:crypto";
2
+ const SAFE_CODE = /^[A-Za-z0-9_.:-]{1,48}$/;
3
+ const SAFE_CORRELATION = /^[A-Za-z0-9-]{1,48}$/;
4
+ export function createCorrelationId() {
5
+ return `oai-${randomUUID().slice(0, 8)}`;
6
+ }
7
+ export function safeCauseCode(error) {
8
+ let cause = error instanceof Error ? error.cause : undefined;
9
+ for (let depth = 0; depth < 3 && typeof cause === "object" && cause !== null; depth++) {
10
+ const code = cause.code;
11
+ if (typeof code === "string" && SAFE_CODE.test(code))
12
+ return code;
13
+ cause = cause.cause;
14
+ }
15
+ return undefined;
16
+ }
17
+ export function formatTransportDiagnostic(diagnostic) {
18
+ const correlationId = SAFE_CORRELATION.test(diagnostic.correlationId)
19
+ ? diagnostic.correlationId
20
+ : "invalid";
21
+ const fields = [
22
+ `correlation=${correlationId}`,
23
+ `operation=${diagnostic.operation}`,
24
+ ];
25
+ if (diagnostic.status !== undefined && Number.isInteger(diagnostic.status)) {
26
+ fields.push(`status=${diagnostic.status}`);
27
+ }
28
+ if (diagnostic.causeCode !== undefined && SAFE_CODE.test(diagnostic.causeCode)) {
29
+ fields.push(`cause=${diagnostic.causeCode}`);
30
+ }
31
+ return fields.join(" ");
32
+ }