maxpool 1.5.36 → 1.5.38

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "maxpool",
3
- "version": "1.5.36",
3
+ "version": "1.5.38",
4
4
  "description": "Multi-account Claude Code proxy with adaptive, rate-aware load balancing across Claude accounts",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -15,7 +15,8 @@
15
15
  "start": "node src/index.js",
16
16
  "test": "bash scripts/run-tests.sh",
17
17
  "lint": "eslint src/ test/",
18
- "release": "bash scripts/release.sh"
18
+ "release": "bash scripts/release.sh",
19
+ "version": "bash scripts/update-changelog.sh"
19
20
  },
20
21
  "keywords": [
21
22
  "claude",
@@ -38,5 +39,8 @@
38
39
  },
39
40
  "engines": {
40
41
  "node": ">=20.3.0"
42
+ },
43
+ "devDependencies": {
44
+ "git-cliff": "2.13.1"
41
45
  }
42
46
  }
@@ -15,6 +15,15 @@ const BOUNDED_REPOLL_HOLD_MS = 60_000;
15
15
  // session can never ping-pong between two similarly-loaded accounts.
16
16
  const REBALANCE_SCORE_MARGIN = 0.5; // candidate score must be ≤ 50% of the bound account's
17
17
  const REBALANCE_MIN_ABS_GAP = 0.5; // …with a small absolute floor so near-zero scores don't micro-churn
18
+ // Warmup-pull: a freshly-ADDED account (mid-session, no reload) would otherwise idle
19
+ // until a reload clears bindings, because bound sessions only leave a HOT account. So
20
+ // for a bounded window pull migration-safe sessions onto it. Keyed on the account's
21
+ // own onboarding state (addedAt + completedRequests), NOT relative load — so it
22
+ // PROVABLY TERMINATES (once it has served WARMUP_REQUESTS, or WARMUP_MS elapses, it
23
+ // stops warming and never fires again) and cannot oscillate for any #sessions/#accounts
24
+ // ratio, unlike a share/concurrency-gap rebalance (which flaps on the lagging load signal).
25
+ const WARMUP_MS = 5 * 60 * 1000; // a just-added account stays "warming" this long…
26
+ const WARMUP_REQUESTS = 5; // …or until it has served this many requests, whichever comes first
18
27
  // Weekly-pressure tiers, healthiest first. A fresh (unknown) account is the best
19
28
  // migration target; migration requires the candidate be a STRICTLY healthier tier.
20
29
  const WEEKLY_TIER = { unknown: 0, normal: 0, soft: 1, reserve: 2, critical: 3, exhausted: 4 };
@@ -1138,6 +1147,54 @@ export class AccountManager {
1138
1147
  * All gates must hold: thinking-safe + not mid queue-admission + bound is hot +
1139
1148
  * a genuinely-healthy alternative that is BOTH much cheaper AND a strictly
1140
1149
  * healthier weekly tier (so concurrency jitter alone can never trigger a move). */
1150
+ /** Is this account still in its post-add onboarding window? A freshly-ADDED
1151
+ * account (addedAt set only by addAccount, never on boot) that has served fewer
1152
+ * than WARMUP_REQUESTS within WARMUP_MS. Providers are never "warming" targets. */
1153
+ _isWarming(account, now = Date.now()) {
1154
+ if (!account || account.type === 'provider') return false;
1155
+ if (account.addedAt == null) return false; // boot/config account → never warming
1156
+ if (account.completedRequests >= WARMUP_REQUESTS) return false; // onboarded → terminates the pull
1157
+ return (now - account.addedAt) < WARMUP_MS;
1158
+ }
1159
+
1160
+ /** Warmup-pull target: the best (lowest-score) healthy, migration-eligible,
1161
+ * still-WARMING NON-provider account to onboard a freshly-added account WITHOUT a
1162
+ * reload, or null. Returned DIRECTLY (not via the score-loop fallthrough) so the
1163
+ * destination is GUARANTEED non-provider even under 'always' cross-provider policy
1164
+ * — a signed-thinking session can never be shuttled onto a provider here (which
1165
+ * the shared candidate loop, keyed only on _matchesRequest, would not prevent).
1166
+ * Fires only when the bound account is itself established (not warming) AND is
1167
+ * actually carrying recent load — relieving a real carrier onto the fresh account,
1168
+ * never churning fresh↔fresh or re-homing an idle session. */
1169
+ _warmupPullTarget(bound, profile, excludedIndexes, requestInfo, now = Date.now()) {
1170
+ // Cheapest early-out first: the overwhelmingly common steady state has NO warming
1171
+ // account, so bail before the migration-safety / load / fleet-scoring work. This
1172
+ // runs on every bound request's selection — keep the no-warming path near-free.
1173
+ if (!this.accounts.some(a => this._isWarming(a, now))) return null;
1174
+ if (!this._migrationSafeForRequest(requestInfo)) return null;
1175
+ if (requestInfo.queueTicket || requestInfo.queueAdmitted) return null;
1176
+ if (this._isWarming(bound, now)) return null;
1177
+ if (this._loadSummary(bound, this.scheduler.spreadWindowMs, now).weight <= 0) return null;
1178
+ const ctx = this._scoringContext();
1179
+ let best = null;
1180
+ let bestScore = Infinity;
1181
+ for (const account of this.accounts) {
1182
+ if (account.index === bound.index) continue;
1183
+ if (account.type === 'provider') continue; // GUARANTEED non-provider destination
1184
+ if (excludedIndexes.has(account.index)) continue;
1185
+ if (!this._isWarming(account, now)) continue;
1186
+ if (!this._matchesRequest(account, profile, requestInfo)) continue;
1187
+ // Genuinely-healthy target only (same bar as the hot-rebalance candidate scan).
1188
+ if (!this._isAvailable(account, { allowWeeklyReserve: false, allowWeeklyCritical: false, model: requestInfo.model })) continue;
1189
+ const score = this._scoreAccount(account, requestInfo, ctx);
1190
+ if (score < bestScore) {
1191
+ bestScore = score;
1192
+ best = account;
1193
+ }
1194
+ }
1195
+ return best;
1196
+ }
1197
+
1141
1198
  _shouldRebalanceBoundSession(bound, profile, excludedIndexes, requestInfo, scoringCtx) {
1142
1199
  if (!this._migrationSafeForRequest(requestInfo)) return false;
1143
1200
  if (requestInfo.queueTicket || requestInfo.queueAdmitted) return false;
@@ -1350,10 +1407,24 @@ export class AccountManager {
1350
1407
  }
1351
1408
  }
1352
1409
  const bound = this._boundAccount(requestInfo.sessionKey, profile, excludedIndexes, requestInfo);
1353
- if (bound
1354
- && !this._hasHigherPriorityAvailable(bound, profile, excludedIndexes, requestInfo)
1355
- && !this._shouldRebalanceBoundSession(bound, profile, excludedIndexes, requestInfo, scoringCtx)) {
1356
- return bound;
1410
+ if (bound) {
1411
+ // Warmup-pull: onboard a freshly-ADDED account (added mid-session, no reload)
1412
+ // by DIRECTLY re-homing this migration-safe session onto the warming account,
1413
+ // before the sticky "stay bound" return. Directed (not via the score loop) so
1414
+ // the destination is guaranteed non-provider. Bounded + self-terminating via
1415
+ // _isWarming; each session re-homes at most once → no flap. Skipped for a
1416
+ // preferred/pinned request (handled above) and any non-migration-safe request.
1417
+ const warmupTarget = this._warmupPullTarget(
1418
+ bound, profile, excludedIndexes, requestInfo, scoringCtx?.now,
1419
+ );
1420
+ if (warmupTarget) {
1421
+ this.currentIndex = warmupTarget.index;
1422
+ return warmupTarget; // _bindSession re-homes the session on acquire
1423
+ }
1424
+ if (!this._hasHigherPriorityAvailable(bound, profile, excludedIndexes, requestInfo)
1425
+ && !this._shouldRebalanceBoundSession(bound, profile, excludedIndexes, requestInfo, scoringCtx)) {
1426
+ return bound;
1427
+ }
1357
1428
  }
1358
1429
  // Else fall through to the candidate score loop, which re-homes the session
1359
1430
  // onto the best healthy account via _bindSession on acquire.
@@ -2488,6 +2559,10 @@ export class AccountManager {
2488
2559
  provisionalRateLimitFingerprint: null,
2489
2560
  recoveredAt: null,
2490
2561
  lastQuotaLogKey: null,
2562
+ // Onboarding clock for the warmup-pull — set ONLY here (mid-session add),
2563
+ // never on the boot/config construction path, so an established fleet account
2564
+ // is never "warming". Lets a just-added account draw load without a reload.
2565
+ addedAt: Date.now(),
2491
2566
  });
2492
2567
  return index;
2493
2568
  }
package/src/server.js CHANGED
@@ -56,6 +56,22 @@ const DEFAULT_QUEUE = {
56
56
  // Non-streaming requests have no SSE heartbeat to keep them alive, so a long
57
57
  // hold would die on the client timeout anyway. Cap their wait conservatively.
58
58
  nonStreamMaxWaitMs: 5 * 60 * 1000,
59
+ // Proactive streaming keep-alive: a STREAMING request gets ZERO client bytes
60
+ // while maxpool selects a route, cycles failover, and waits for the upstream's
61
+ // first byte (up to UPSTREAM_TTFB_MS=120s) — the queue heartbeat only starts
62
+ // once a request is QUEUED. If that client-silent window exceeds the client's
63
+ // own idle timeout (Claude Code aborts "Stream idle timeout - no chunks
64
+ // received", observed as low as ~23s), the client gives up on a request maxpool
65
+ // is still patiently serving. If the upstream hasn't delivered bytes within this
66
+ // grace, commit the SSE stream + start the heartbeat so the client never idles
67
+ // out. Kept above a normal fast TTFB (1-5s → common case never early-commits,
68
+ // behavior unchanged) and well below the client idle floor. Env-overridable.
69
+ streamForwardGraceMs: Math.max(1000, Number(process.env.MAXPOOL_STREAM_FORWARD_GRACE_MS) || 10000),
70
+ // count_tokens is cheap non-streaming metadata — cap its queue wait VERY low so it
71
+ // fast-fails with a retryable 429 instead of hanging silently past the client's idle
72
+ // window. Kept well under any plausible client idle timeout (observed errors as low
73
+ // as ~23s). Env-overridable for tuning as the "Stream idle timeout" reports resolve.
74
+ countTokensMaxWaitMs: Math.max(1000, Number(process.env.MAXPOOL_COUNT_TOKENS_MAX_WAIT_MS) || 8000),
59
75
  // Backpressure: holds used to be 0ms, now they can be hours. Bound the queue
60
76
  // so 22 retrying agents can't grow the heap without limit.
61
77
  maxConcurrentQueued: 64,
@@ -487,6 +503,28 @@ async function forwardRequest(
487
503
  UPSTREAM_TTFB_MS,
488
504
  );
489
505
  ttfbTimer.unref?.();
506
+ // Proactive streaming keep-alive. UPSTREAM_TTFB_MS (120s) is FAR above the
507
+ // client's own idle timeout (~23-60s), and the queue heartbeat only starts once
508
+ // a request is QUEUED — so a streaming request whose selected upstream is slow to
509
+ // first byte, or that burns the client's timeout cycling failover, sends the
510
+ // client ZERO bytes and is aborted ("Stream idle timeout - no chunks received")
511
+ // on a request maxpool is still serving. If bytes haven't arrived within the
512
+ // grace, commit the SSE stream + heartbeat so the client stays alive. The
513
+ // deadline is anchored ONCE per request (??=) so it also bounds CUMULATIVE
514
+ // failover time across re-forwards, not each attempt independently.
515
+ const streamForwardGraceMs = queueConfig?.streamForwardGraceMs == null
516
+ ? 10000
517
+ : Math.max(0, Number(queueConfig.streamForwardGraceMs) || 0);
518
+ let streamGraceTimer = null;
519
+ if (requestInfo.stream && !res.headersSent) {
520
+ requestInfo.streamGraceDeadline ??= Date.now() + streamForwardGraceMs;
521
+ const graceDelay = Math.max(0, requestInfo.streamGraceDeadline - Date.now());
522
+ streamGraceTimer = setTimeout(
523
+ () => commitStreamGraceHeartbeat(res, requestInfo, queueConfig, accountManager),
524
+ graceDelay,
525
+ );
526
+ streamGraceTimer.unref?.();
527
+ }
490
528
  try {
491
529
  let upstreamRes;
492
530
  try {
@@ -499,6 +537,9 @@ async function forwardRequest(
499
537
  });
500
538
  } finally {
501
539
  clearTimeout(ttfbTimer);
540
+ // Stop the grace timer for THIS attempt — but LEAVE streamGraceDeadline set so
541
+ // a re-forward (failover) re-arms for the REMAINING time to the shared deadline.
542
+ if (streamGraceTimer) clearTimeout(streamGraceTimer);
502
543
  }
503
544
  // Response arrived — the pre-response leak window is over. Stop guarding for
504
545
  // client-disconnect via abort (streamResponse handles mid-stream disconnects).
@@ -1095,6 +1136,7 @@ function formatRetryDuration(seconds) {
1095
1136
  function computeQueueWindowMs({
1096
1137
  cause, stream, retryPlanCause,
1097
1138
  maxWaitMs, capacityMaxWaitMs, nonStreamMaxWaitMs, streamHoldMaxMs,
1139
+ isCountTokens, countTokensMaxWaitMs,
1098
1140
  }) {
1099
1141
  let windowMs;
1100
1142
  if (!stream) {
@@ -1105,6 +1147,10 @@ function computeQueueWindowMs({
1105
1147
  windowMs = streamHoldMaxMs;
1106
1148
  }
1107
1149
  if (retryPlanCause === 'concurrency_cap') windowMs = Math.min(windowMs, capacityMaxWaitMs);
1150
+ // count_tokens: cap the QUEUE wait low (bounds only the wait-for-an-account, never
1151
+ // the upstream processing once acquired) so a non-heartbeated metadata call fast-
1152
+ // fails with a retryable 429 instead of hanging past the client's idle window.
1153
+ if (isCountTokens && countTokensMaxWaitMs != null) windowMs = Math.min(windowMs, countTokensMaxWaitMs);
1108
1154
  return windowMs;
1109
1155
  }
1110
1156
 
@@ -1143,7 +1189,7 @@ function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRe
1143
1189
  return `All ${n} accounts exhausted. Retry in ${retryAfter}s.`;
1144
1190
  }
1145
1191
 
1146
- export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, streamResponse, startIdleRequestReaper };
1192
+ export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, commitStreamGraceHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, streamResponse, startIdleRequestReaper };
1147
1193
 
1148
1194
  async function readErrorBody(upstreamRes, limitBytes = 64 * 1024) {
1149
1195
  if (!upstreamRes.body) return '';
@@ -1392,6 +1438,9 @@ async function queueAndRetry(
1392
1438
  const nonStreamMaxWaitMs = queueConfig.nonStreamMaxWaitMs == null
1393
1439
  ? 5 * 60_000
1394
1440
  : Math.max(0, Number(queueConfig.nonStreamMaxWaitMs) || 0);
1441
+ const countTokensMaxWaitMs = queueConfig.countTokensMaxWaitMs == null
1442
+ ? 8000
1443
+ : Math.max(0, Number(queueConfig.countTokensMaxWaitMs) || 0);
1395
1444
  const retryPlan = accountManager.nextRetryForRequest?.(requestInfo, new Set()) || {
1396
1445
  retryAfterMs: Infinity,
1397
1446
  cause: 'unavailable',
@@ -1426,6 +1475,8 @@ async function queueAndRetry(
1426
1475
  capacityMaxWaitMs,
1427
1476
  nonStreamMaxWaitMs,
1428
1477
  streamHoldMaxMs,
1478
+ isCountTokens: Boolean(requestInfo.isCountTokens),
1479
+ countTokensMaxWaitMs,
1429
1480
  });
1430
1481
 
1431
1482
  if (queueWindowMs <= 0) return finishQueuedStreamIfNeeded(res, requestInfo, honestMessage);
@@ -1492,26 +1543,56 @@ async function queueAndRetry(
1492
1543
  ).then(() => true);
1493
1544
  }
1494
1545
 
1546
+ // The stream-forward grace-timer callback (extracted so its logic is unit-testable
1547
+ // in isolation). Fires when a STREAMING request's upstream is slow to first byte:
1548
+ // commits the SSE stream + heartbeat so the client never idle-times-out. Client-
1549
+ // abort-safe — an async timer has NO synchronous liveness precondition (unlike the
1550
+ // queue caller), so if the client vanished during the forward window (or the stream
1551
+ // is already committed) it reaps the queue slot and bails WITHOUT touching the dead
1552
+ // socket. ensureQueueHeartbeat is itself hardened against a throwing write, but this
1553
+ // guard is the cheaper first line of defense against the worker-bounce race.
1554
+ function commitStreamGraceHeartbeat(res, requestInfo, queueConfig, accountManager) {
1555
+ if (res.destroyed || res.writableEnded || res.headersSent) {
1556
+ clearQueueHeartbeat(requestInfo);
1557
+ accountManager.removeQueuedRequest?.(requestInfo);
1558
+ return;
1559
+ }
1560
+ ensureQueueHeartbeat(res, requestInfo, queueConfig, accountManager);
1561
+ }
1562
+
1495
1563
  function ensureQueueHeartbeat(res, requestInfo, queueConfig, accountManager) {
1496
1564
  if (!requestInfo.stream || requestInfo.queueHeartbeatActive || res.headersSent) return;
1497
1565
  const heartbeatMs = Math.max(1000, Number(queueConfig.heartbeatMs) || 10_000);
1498
- res.writeHead(200, {
1499
- 'Content-Type': 'text/event-stream',
1500
- 'Cache-Control': 'no-cache',
1501
- Connection: 'keep-alive',
1502
- 'X-Accel-Buffering': 'no',
1503
- });
1504
- res.flushHeaders?.();
1505
- res.write(': maxpool queued\n\n');
1506
- requestInfo.queueHeartbeatActive = true;
1507
1566
  // The heartbeat is the liveness probe: if the client is gone (socket
1508
- // destroyed/ended, or the write throws EPIPE/ERR_STREAM_DESTROYED), release
1567
+ // destroyed/ended, or a write throws EPIPE/ERR_STREAM_DESTROYED), release
1509
1568
  // the queue slot + bytes IMMEDIATELY rather than letting a dead ticket occupy
1510
1569
  // the queue until its (up to 7d) deadline — the ghost-leak guard.
1511
1570
  const reapDead = () => {
1512
1571
  clearQueueHeartbeat(requestInfo);
1513
1572
  accountManager?.removeQueuedRequest?.(requestInfo);
1514
1573
  };
1574
+ // The INITIAL commit can throw if the socket died between the caller's last
1575
+ // liveness check and here. The synchronous queue caller (queueAndRetry) bails
1576
+ // on res.destroyed right before calling, but the ASYNC stream-forward grace
1577
+ // timer has no such precondition — the client can vanish mid-window. An
1578
+ // unguarded throw here has NO 'error' listener → uncaughtException → the worker
1579
+ // process.exits and bounces EVERY in-flight stream. Guard the initial write
1580
+ // exactly like the interval callback below so ensureQueueHeartbeat is safe from
1581
+ // ANY caller, sync or async.
1582
+ try {
1583
+ res.writeHead(200, {
1584
+ 'Content-Type': 'text/event-stream',
1585
+ 'Cache-Control': 'no-cache',
1586
+ Connection: 'keep-alive',
1587
+ 'X-Accel-Buffering': 'no',
1588
+ });
1589
+ res.flushHeaders?.();
1590
+ res.write(': maxpool queued\n\n');
1591
+ } catch {
1592
+ reapDead();
1593
+ return;
1594
+ }
1595
+ requestInfo.queueHeartbeatActive = true;
1515
1596
  requestInfo.queueHeartbeatTimer = setInterval(() => {
1516
1597
  if (res.destroyed || res.writableEnded) { reapDead(); return; }
1517
1598
  try {
@@ -1596,6 +1677,11 @@ function describeRequest(req, body) {
1596
1677
  path: req.url,
1597
1678
  bodyBytes: body.length,
1598
1679
  weight,
1680
+ // count_tokens is cheap non-streaming metadata with NO SSE heartbeat, so a long
1681
+ // queue-hold just dies on the client's idle window ("Stream idle timeout - no
1682
+ // chunks received"). Flag it (URL-based, so it's set regardless of body parse) to
1683
+ // cap its queue wait short and fast-fail with a retryable 429 instead of hanging.
1684
+ isCountTokens: /\/v1\/messages\/count_tokens\b/.test(req.url),
1599
1685
  };
1600
1686
  try {
1601
1687
  const json = JSON.parse(body.toString());
package/src/tui.js CHANGED
@@ -26,14 +26,27 @@ const vw = s => strip(s).length;
26
26
 
27
27
  // ── Accounts-table columns ───────────────────────────────────
28
28
  // Fixed column widths shared by the header row (acctHeader) AND every data row, so
29
- // the header labels stay aligned with the columns they name. The Account/Type/
30
- // Status/Quota start offsets (4/17/26/40) are pure functions of these widths + the
29
+ // the header labels stay aligned with the columns they name. The Account/Provider/
30
+ // Status/Quota start offsets (4/17/27/41) are pure functions of these widths + the
31
31
  // 4-col row prefix, independent of the quota-bar width.
32
32
  const NAME_W = 12; // a.name.slice(0, NAME_W).padEnd(NAME_W)
33
- const TYPE_W = 8; // a.type.padEnd(TYPE_W) — fits "provider"
33
+ const PROVIDER_W = 9; // providerLabel(a).padEnd(PROVIDER_W) — fits "Anthropic"
34
34
  const STATUS_W = 13; // rpad(status, STATUS_W) — fits "throttled 59s"
35
35
  const ROW_PREFIX = ' '; // ' ' + sel(1) + cur(1) + ' ' — 4 cols before the name
36
36
 
37
+ // Human provider name for the accounts-table "Provider" column. account.provider
38
+ // is 'anthropic' for oauth/apikey accounts (they ARE Anthropic — the oauth-vs-key
39
+ // billing split is still shown by the Ses/Wk vs Tok/Req quota-bar labels, not here),
40
+ // and 'zai'/'kimi' for the GLM/Kimi fallback providers. Explicit map + a graceful
41
+ // Titlecase default so a future provider (Codex/Grok) renders sanely; truncated to
42
+ // the column width so a long name can never misalign the row.
43
+ const PROVIDER_LABELS = { anthropic: 'Anthropic', zai: 'z.ai', kimi: 'Moonshot', openai: 'OpenAI', codex: 'Codex', grok: 'Grok', xai: 'Grok' };
44
+ function providerLabel(a) {
45
+ const key = a.provider || (a.type === 'provider' ? 'provider' : 'anthropic');
46
+ const label = PROVIDER_LABELS[key] || (key.charAt(0).toUpperCase() + key.slice(1));
47
+ return label.slice(0, PROVIDER_W);
48
+ }
49
+
37
50
  /**
38
51
  * Aligned column header for the accounts table. Names the three columns that carry
39
52
  * NO inline label (Account / Type / Status) plus a group label over the two quota
@@ -45,7 +58,7 @@ function acctHeader(W) {
45
58
  const quota = W >= 88 ? 'Quota (used% · resets-in)' : 'Quota';
46
59
  return ROW_PREFIX
47
60
  + 'Account'.padEnd(NAME_W) + ' '
48
- + 'Type'.padEnd(TYPE_W) + ' '
61
+ + 'Provider'.padEnd(PROVIDER_W) + ' '
49
62
  + 'Status'.padEnd(STATUS_W) + ' '
50
63
  + quota;
51
64
  }
@@ -221,7 +234,7 @@ function emptyBar(label, w = 10) {
221
234
  return `${ESC}100m${' '.repeat(lp)}${text}${' '.repeat(rp)}${RESET}`;
222
235
  }
223
236
 
224
- export const __tuiTest = { formatReset, quotaLabel, bar, emptyBar, strip, loadText, countdown, acctHeader, fitLine };
237
+ export const __tuiTest = { formatReset, quotaLabel, bar, emptyBar, strip, loadText, countdown, acctHeader, fitLine, providerLabel };
225
238
 
226
239
  function timestamp() {
227
240
  return new Date().toLocaleTimeString('en-US', { hour12: false });
@@ -554,9 +567,25 @@ export class TUI {
554
567
  this.selIdx = selectable.includes(this.am.currentIndex) ? this.am.currentIndex : selectable[0];
555
568
  }
556
569
 
570
+ // Real am.accounts indices in DISPLAY order: non-provider (Claude/OAuth/apikey)
571
+ // accounts first, providers (GLM/Kimi fallback) last, stable within each group.
572
+ // The single source of truth for row order — the render loop AND selection nav
573
+ // both iterate it, so the highlight can never desync from the visible list. The
574
+ // canonical am.accounts array order is never mutated (routing/index actions safe).
575
+ _displayOrder() {
576
+ const nonProv = [];
577
+ const prov = [];
578
+ this.am.accounts.forEach((a, i) => (a.type === 'provider' ? prov : nonProv).push(i));
579
+ return [...nonProv, ...prov];
580
+ }
581
+
557
582
  _selectableIndexes(action) {
558
- return this.am.accounts
559
- .map((account, index) => ({ account, index }))
583
+ // Map over _displayOrder() BEFORE filtering so the nav array is in DISPLAY order
584
+ // with the existing selectability filter intact — _keySelect steps this array, so
585
+ // visual order == nav order automatically and it can never land on a provider
586
+ // (prefer) or a non-configurable runtime account.
587
+ return this._displayOrder()
588
+ .map(index => ({ account: this.am.accounts[index], index }))
560
589
  .filter(({ account }) => {
561
590
  if (action === 'prefer') return account.type !== 'provider' && account.enabled;
562
591
  return this._configAccountIndex(account) >= 0;
@@ -991,11 +1020,17 @@ export class TUI {
991
1020
  // misaligned second header).
992
1021
  lines.push(dimUnderline(acctHeader(W)));
993
1022
  const showBoth = W >= 70;
1023
+ // 57/46 = the fixed pre-bar column span (prefix + Account + Provider + Status +
1024
+ // gaps); grew by 1 with the Provider column (was 56/45 under the 8-wide Type col).
994
1025
  const bw = showBoth
995
- ? Math.max(5, Math.min(20, Math.floor((W - 56) / 2)))
996
- : Math.max(5, Math.min(20, W - 45));
997
-
998
- for (let i = 0; i < this.am.accounts.length; i++) {
1026
+ ? Math.max(5, Math.min(20, Math.floor((W - 57) / 2)))
1027
+ : Math.max(5, Math.min(20, W - 46));
1028
+
1029
+ // Claude/OAuth accounts first, providers (GLM/Kimi fallback) last — a stable
1030
+ // display order over the canonical am.accounts array (which stays untouched so
1031
+ // routing/index-keyed actions are unaffected). Selection navigation shares the
1032
+ // SAME order via _selectableIndexes → _displayOrder.
1033
+ for (const i of this._displayOrder()) {
999
1034
  lines.push(this._renderAcct(i, bw, showBoth));
1000
1035
  }
1001
1036
  // Glossary FOOTER (expands the abbreviations the header + inline labels can't
@@ -1076,10 +1111,11 @@ export class TUI {
1076
1111
  const rawName = a.name.slice(0, NAME_W).padEnd(NAME_W);
1077
1112
  const name = isSel ? bold(rawName) : rawName;
1078
1113
 
1079
- // Type — pad to 8 so "provider" (8 chars) doesn't overflow a 7-wide column and
1080
- // shift the whole provider row (incl. its quota bars) 1 char out of alignment
1081
- // with the "oauth"/"apikey" rows.
1082
- const type = gray(a.type.padEnd(TYPE_W));
1114
+ // Provider column — the real vendor (Anthropic / z.ai / Moonshot), padded to
1115
+ // PROVIDER_W so "Anthropic" (9 chars) never overflows and shifts the row (incl.
1116
+ // its quota bars) out of alignment with the shorter provider labels. This single
1117
+ // cell is reused by both the oauth/apikey row below and _renderProviderAcct.
1118
+ const type = gray(providerLabel(a).padEnd(PROVIDER_W));
1083
1119
 
1084
1120
  // Status
1085
1121
  let status;