maxpool 1.5.63 → 1.5.65

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "maxpool",
3
- "version": "1.5.63",
3
+ "version": "1.5.65",
4
4
  "description": "Multi-account Claude Code proxy with adaptive, rate-aware load balancing across Claude accounts",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -50,6 +50,10 @@ function emptyQuota() {
50
50
  // { fable:{utilization,resetAt,severity,isActive}, opus:{...}, sonnet:{...} }.
51
51
  scopedWeekly: {},
52
52
  unifiedStatus: null, // allowed | allowed_warning | rejected
53
+ // TRUE when a SUCCESSFUL probe reported no weekly cap for this account — i.e.
54
+ // "this plan has no weekly limit", NOT "we couldn't read one". Without it both
55
+ // states render as an empty bar and a healthy account reads as broken.
56
+ weeklyAbsent: false,
53
57
  resetsAt: null,
54
58
  // Provider (z.ai / Kimi) quota — kept SEPARATE from unified* so a provider
55
59
  // reading never leaks into the OAuth quota gates (_isAvailable / _weeklyRawState
@@ -1396,6 +1400,41 @@ export class AccountManager {
1396
1400
  return { cause: 'unavailable', retryAt: null, queueable: false };
1397
1401
  }
1398
1402
 
1403
+ /**
1404
+ * Why is every Claude account unavailable RIGHT NOW? Counts the per-account causes
1405
+ * the retry oracle already computes, so the 429 can name the real one.
1406
+ *
1407
+ * "All N accounts are at their limit" was wrong whenever the true cause was a short
1408
+ * NETWORK cooldown (5s each, applied on a connection drop): the user reads a quota
1409
+ * problem, waits, and considers adding accounts — when the fleet is actually fine and
1410
+ * recovers in seconds. Returns { total, quota, transient, network, dominant }.
1411
+ */
1412
+ unavailabilityCensus(model = null) {
1413
+ const QUOTA = new Set(['exhausted', 'weekly_exhausted', 'session_limit', 'token_limit', 'request_limit', 'rate_limited']);
1414
+ const TRANSIENT = new Set(['cooldown', 'upstream_failure', 'concurrency_cap', 'weekly_critical']);
1415
+ const c = { total: 0, quota: 0, transient: 0, network: 0, disabled: 0, error: 0, other: 0 };
1416
+ for (const a of this.accounts) {
1417
+ if (a.type === 'provider') continue;
1418
+ c.total++;
1419
+ const cause = this._retryInfo(a, model)?.cause || 'unavailable';
1420
+ if (cause === 'disabled') c.disabled++;
1421
+ else if (cause === 'error') c.error++;
1422
+ else if (QUOTA.has(cause)) c.quota++;
1423
+ else if (TRANSIENT.has(cause)) {
1424
+ c.transient++;
1425
+ // A cooldown whose window matches the NETWORK cooldown is a connectivity blip,
1426
+ // not congestion — the distinction the user's report was about.
1427
+ if (cause === 'cooldown' && a.cooldownUntil
1428
+ && (a.cooldownUntil - Date.now()) <= this.scheduler.networkCooldownMs) c.network++;
1429
+ } else c.other++;
1430
+ }
1431
+ const eligible = c.total - c.disabled - c.error;
1432
+ c.dominant = eligible <= 0 ? 'none'
1433
+ : c.quota >= Math.max(1, Math.ceil(eligible / 2)) ? 'quota'
1434
+ : c.transient > 0 ? 'transient' : 'other';
1435
+ return c;
1436
+ }
1437
+
1399
1438
  // The soonest active short-term (non-weekly) blocker for an account, or null if
1400
1439
  // none is active. Ordered most-specific-first; each entry is a {cause, retryAt,
1401
1440
  // queueable} the retry oracle can hold on. Kept separate from the weekly state
@@ -2212,6 +2251,11 @@ export class AccountManager {
2212
2251
  if (usage.sevenDay.utilization != null) q.unified7d = clamp01(usage.sevenDay.utilization);
2213
2252
  if (usage.sevenDay.resetAt != null) q.unified7dReset = usage.sevenDay.resetAt;
2214
2253
  }
2254
+ // Only a SUCCESSFUL probe carrying the flag speaks to this. A header-driven update
2255
+ // can't see the limits[] array, and a FAILED read knows nothing about the account's
2256
+ // caps — either one claiming "uncapped" would mislabel a capped account as having no
2257
+ // weekly limit, which is the exact wrong direction (it reads as free capacity).
2258
+ if (!usage.error && usage.weeklyAbsent !== undefined) q.weeklyAbsent = Boolean(usage.weeklyAbsent);
2215
2259
  // Per-model weekly sub-limits (Fable, Opus, ...). Replace wholesale with the
2216
2260
  // fresh probe set so a family that dropped out of the response doesn't linger
2217
2261
  // stale; expiry on reset is a backstop for the between-probe window. EXCEPTION:
@@ -2320,11 +2364,16 @@ export class AccountManager {
2320
2364
  if (usage.wk) {
2321
2365
  if (usage.wk.utilization != null) q.providerWk = clamp01(usage.wk.utilization);
2322
2366
  if (usage.wk.resetAt != null) q.providerWkReset = usage.wk.resetAt;
2367
+ q.weeklyAbsent = false;
2323
2368
  } else {
2324
2369
  // Weekly window absent from this plan/response — clear so a stale weekly
2325
- // reading doesn't linger after a plan/window change.
2370
+ // reading doesn't linger after a plan/window change. A SUCCESSFUL poll that
2371
+ // carries no weekly is positive knowledge that the plan has none: measured
2372
+ // 2026-08-06, z.ai `max` returns exactly one TOKENS_LIMIT (unit 3 = the 5h
2373
+ // session) and no unit-6 weekly, so GLM's blank Wk is correct, not a gap.
2326
2374
  q.providerWk = null;
2327
2375
  q.providerWkReset = null;
2376
+ q.weeklyAbsent = true;
2328
2377
  }
2329
2378
  q.lastProbeOkAt = Date.now();
2330
2379
  }
@@ -2773,6 +2822,8 @@ export class AccountManager {
2773
2822
  modelMap: acctData.modelMap || null,
2774
2823
  stripBetaHeaders: Boolean(acctData.stripBetaHeaders),
2775
2824
  runtime: Boolean(acctData.runtime),
2825
+ configSourced: Boolean(acctData.configSourced),
2826
+ secretName: acctData.secretName || null,
2776
2827
  enabled: acctData.enabled !== false,
2777
2828
  refreshToken: acctData.refreshToken || null,
2778
2829
  expiresAt: acctData.expiresAt || null,
@@ -2827,6 +2878,8 @@ export class AccountManager {
2827
2878
  account.modelMap = acctData.modelMap || account.modelMap;
2828
2879
  account.stripBetaHeaders = Boolean(acctData.stripBetaHeaders);
2829
2880
  account.runtime = true;
2881
+ if (acctData.configSourced !== undefined) account.configSourced = acctData.configSourced;
2882
+ if (acctData.secretName !== undefined) account.secretName = acctData.secretName;
2830
2883
  // Restore path carries an explicit enabled (persisted disable); honor it. The `cc
2831
2884
  // all` header path (prepareRuntimeProviders) omits enabled, so a re-sent token
2832
2885
  // NEVER silently re-enables a provider the user benched in the TUI.
@@ -2851,7 +2904,7 @@ export class AccountManager {
2851
2904
  */
2852
2905
  exportRuntimeProviders() {
2853
2906
  return this.accounts
2854
- .filter(a => a.runtime && a.type === 'provider' && a.credential)
2907
+ .filter(a => a.runtime && a.type === 'provider' && a.credential && !a.configSourced)
2855
2908
  .map(a => ({
2856
2909
  name: a.name,
2857
2910
  type: a.type,
@@ -2877,6 +2930,10 @@ export class AccountManager {
2877
2930
  * live `cc all` request refreshes the token before routing (prepareRuntimeProviders
2878
2931
  * runs ahead of account selection), so a stale restored token never serves a
2879
2932
  * request. Idempotent — upsertRuntimeAccount matches by name.
2933
+ *
2934
+ * Config-sourced providers (resolved from GCP Secret Manager at startup) are
2935
+ * EXCLUDED — their source of truth is config + GCP, not state.json, so persisting
2936
+ * their token to disk would duplicate the secret and survive a GCP deletion.
2880
2937
  */
2881
2938
  restoreRuntimeProviders(list) {
2882
2939
  if (!Array.isArray(list)) return;
@@ -2886,6 +2943,69 @@ export class AccountManager {
2886
2943
  }
2887
2944
  }
2888
2945
 
2946
+ /**
2947
+ * Load config-sourced provider accounts — resolved from GCP Secret Manager at
2948
+ * startup, NOT from per-request headers. Each config entry is { name, provider,
2949
+ * secretName, upstream?, priority?, modelMap? }. The secret is resolved by the
2950
+ * caller (index.js) and passed as `token`; if null the provider is created but
2951
+ * marked error so the TUI shows WHY it's broken.
2952
+ *
2953
+ * These are marked configSourced so they're excluded from state.json persistence
2954
+ * (their source of truth is config + GCP, not disk) and so the header path can
2955
+ * dedup against them (same token → skip creating a duplicate runtime provider).
2956
+ */
2957
+ loadConfigProviders(entries) {
2958
+ if (!Array.isArray(entries)) return;
2959
+ // Remove existing config-sourced providers that are no longer in the config
2960
+ // (handles a config edit that removes an entry).
2961
+ const wantedNames = new Set(entries.map(e => e.name).filter(Boolean));
2962
+ for (const a of this.accounts) {
2963
+ if (a.configSourced && !wantedNames.has(a.name)) {
2964
+ const idx = this.accounts.indexOf(a);
2965
+ if (idx >= 0 && this.accounts[idx].inFlight === 0) this.removeAccount(idx);
2966
+ }
2967
+ }
2968
+ for (const entry of entries) {
2969
+ if (!entry || !entry.name || !entry.provider) continue;
2970
+ this.upsertRuntimeAccount({
2971
+ name: entry.name,
2972
+ type: 'provider',
2973
+ provider: entry.provider,
2974
+ authToken: entry.token || null,
2975
+ upstream: entry.upstream || (entry.provider === 'kimi'
2976
+ ? 'https://api.kimi.com/coding'
2977
+ : 'https://api.z.ai/api/anthropic'),
2978
+ authHeader: 'authorization',
2979
+ profiles: ['all'],
2980
+ priority: Number.isFinite(entry.priority) ? entry.priority : 10,
2981
+ modelMap: entry.modelMap,
2982
+ model: entry.model,
2983
+ stripBetaHeaders: true,
2984
+ configSourced: true,
2985
+ });
2986
+ if (!entry.token) {
2987
+ const a = this.accounts.find(a => a.name === entry.name);
2988
+ if (a) { a.status = 'error'; a.lastError = 'secret-unresolved'; }
2989
+ }
2990
+ }
2991
+ }
2992
+
2993
+ /**
2994
+ * Return the config-sourced provider definitions (name + provider + secretName)
2995
+ * for the TUI and for config serialization. Tokens are NEVER included.
2996
+ */
2997
+ configProviderDefs() {
2998
+ return this.accounts
2999
+ .filter(a => a.configSourced && a.type === 'provider')
3000
+ .map(a => ({
3001
+ name: a.name,
3002
+ provider: a.provider,
3003
+ secretName: a.secretName || null,
3004
+ upstream: a.upstream,
3005
+ priority: a.priority,
3006
+ }));
3007
+ }
3008
+
2889
3009
  /**
2890
3010
  * Remove an account by index.
2891
3011
  */
package/src/config.js CHANGED
@@ -154,6 +154,11 @@ export function createDefaultConfig() {
154
154
  heartbeatMs: 10_000,
155
155
  },
156
156
  accounts: [],
157
+ // Config-sourced provider accounts (GLM/Kimi). Each entry references a GCP
158
+ // Secret Manager secret by NAME — the key is resolved at startup and held in
159
+ // memory only. See src/secret-resolver.js.
160
+ // [{ name, provider:'zai'|'kimi', secretName, upstream?, priority? }]
161
+ providers: [],
157
162
  };
158
163
  }
159
164
 
package/src/index.js CHANGED
@@ -593,6 +593,40 @@ async function serverWorkerCommand() {
593
593
  // self-healing window as an abrupt SIGKILL. A cold restart reads the clean-shutdown
594
594
  // final flush and restores fully (the reported zero-request case).
595
595
  if (savedState?.runtimeProviders) accountManager.restoreRuntimeProviders(savedState.runtimeProviders);
596
+
597
+ // ── config-sourced providers (GCP Secret Manager) ──────────────────────────
598
+ // Provider keys (GLM, Kimi) defined in the config's `providers` section, each
599
+ // referencing a GCP Secret Manager secret by NAME — the key is resolved at
600
+ // startup and held in memory only, never written to config or state.json.
601
+ // Team members add a provider by storing the key in GCP, then pointing maxpool
602
+ // at the secret name via the TUI. Deleting the GCP secret disables the provider
603
+ // on the next restart — no key to hunt down in config files.
604
+ if (Array.isArray(config.providers) && config.providers.length > 0) {
605
+ try {
606
+ const { resolveSecrets } = await import('./secret-resolver.js');
607
+ // Split: GCP-sourced (secretName) vs direct (apiKey). Both produce a token
608
+ // the same way — the resolution path is the only difference.
609
+ const gcpEntries = config.providers.filter(p => p.secretName);
610
+ const directEntries = config.providers.filter(p => p.apiKey && !p.secretName);
611
+ const secretNames = gcpEntries.map(p => p.secretName);
612
+ const resolved = await resolveSecrets(secretNames);
613
+ const entries = config.providers.map(p => ({
614
+ ...p,
615
+ token: p.secretName ? (resolved[p.secretName] || null) : (p.apiKey || null),
616
+ }));
617
+ accountManager.loadConfigProviders(entries);
618
+ const ok = entries.filter(e => e.token).length;
619
+ const fail = entries.length - ok;
620
+ console.log(`[Maxpool] Config providers: ${ok} active${fail ? `, ${fail} unresolved` : ''}`);
621
+ for (const p of config.providers) {
622
+ const a = accountManager.accounts.find(a => a.name === p.name);
623
+ if (a && p.secretName) a.secretName = p.secretName;
624
+ }
625
+ } catch (err) {
626
+ console.error(`[Maxpool] Config provider resolution failed: ${err.message}`);
627
+ }
628
+ }
629
+
596
630
  // Track the state-file generation we last observed so a stale flush is refused.
597
631
  let stateGeneration = Number(savedState?._generation) || 0;
598
632
 
package/src/oauth.js CHANGED
@@ -267,10 +267,18 @@ export async function fetchUsage(accessToken) {
267
267
  }
268
268
  }
269
269
 
270
+ const sevenDay = limitsWeeklyAll || normalizeUsageBucket(data?.seven_day);
270
271
  return {
271
272
  fiveHour: limitsSession || normalizeUsageBucket(data?.five_hour),
272
- sevenDay: limitsWeeklyAll || normalizeUsageBucket(data?.seven_day),
273
+ sevenDay,
273
274
  scopedWeekly,
275
+ // A SUCCESSFUL read that carries no weekly bucket is positive knowledge: this
276
+ // account HAS no weekly cap, as opposed to "we haven't managed to read one yet".
277
+ // Both render as a blank bar otherwise, so a perfectly healthy account looks
278
+ // broken. Measured 2026-08-06: privacy@gomokka.com returns limits=[session,
279
+ // weekly_scoped(inactive)] and `seven_day: null`, while every other account
280
+ // returns a `weekly_all` — it is genuinely uncapped, not unread.
281
+ weeklyAbsent: sevenDay == null || sevenDay.utilization == null,
274
282
  };
275
283
  } catch (err) {
276
284
  return { error: err.message || String(err), status: null };
@@ -0,0 +1,49 @@
1
+ /**
2
+ * Resolve a GCP Secret Manager secret name to its plaintext value.
3
+ *
4
+ * Uses `gcloud secrets versions access` — the same path load-secrets.sh uses.
5
+ * The key is resolved ONCE at startup (or when a provider is added via the TUI)
6
+ * and held in maxpool's process memory, never written to disk or logs.
7
+ *
8
+ * Returns null on any failure (missing secret, no gcloud, auth error) so a
9
+ * broken provider degrades to "disabled" instead of crashing the proxy.
10
+ */
11
+ import { execFile } from 'node:child_process';
12
+ import { promisify } from 'node:util';
13
+
14
+ const execFileAsync = promisify(execFile);
15
+
16
+ const DEFAULT_PROJECT = 'mokka-business-automations';
17
+
18
+ export async function resolveSecret(secretName, { project = DEFAULT_PROJECT, timeoutMs = 10_000 } = {}) {
19
+ if (!secretName || typeof secretName !== 'string') return null;
20
+ try {
21
+ const { stdout } = await execFileAsync(
22
+ 'gcloud',
23
+ ['secrets', 'versions', 'access', 'latest', '--secret', secretName, '--project', project],
24
+ { timeout: timeoutMs, maxBuffer: 1024 * 1024 },
25
+ );
26
+ const value = stdout.trim();
27
+ return value || null;
28
+ } catch {
29
+ // Missing secret, no gcloud, auth expired, no network — all degrade to null.
30
+ // The TUI shows the provider as "error: secret-unresolved" so the operator
31
+ // knows which secret failed, not just that "the provider is broken".
32
+ return null;
33
+ }
34
+ }
35
+
36
+ /**
37
+ * Resolve multiple secrets in parallel. Returns { name → value } for the ones
38
+ * that resolved; failed ones are simply absent (caller treats absence as
39
+ * "provider can't activate"). Parallel because N secrets at ~500ms each
40
+ * serialized would add seconds to startup.
41
+ */
42
+ export async function resolveSecrets(secretNames, opts = {}) {
43
+ const unique = [...new Set(secretNames.filter(Boolean))];
44
+ if (!unique.length) return {};
45
+ const entries = await Promise.all(
46
+ unique.map(async name => [name, await resolveSecret(name, opts)]),
47
+ );
48
+ return Object.fromEntries(entries.filter(([, v]) => v != null));
49
+ }
package/src/server.js CHANGED
@@ -403,6 +403,13 @@ async function forwardRequest(
403
403
  ) {
404
404
  const configuredAttempts = Number(retryConfig.maxAttemptsPerRequest) || accountManager.accounts.length;
405
405
  const maxAttempts = Math.max(1, configuredAttempts);
406
+ // A body REPAIR (strip a block, convert a tool pair, downgrade effort) is not an
407
+ // account failover — it re-sends a FIXED body and consumes no account. Charging both
408
+ // to one budget sized by ACCOUNT COUNT meant a 1-account fleet could never repair
409
+ // anything and a 2-account fleet got exactly one repair, while the chain is now six
410
+ // deep. Each repair latches its own flag, so this budget is a backstop, not the bound.
411
+ const repairCount = requestInfo.repairCount || 0;
412
+ const canRepairBody = repairCount < 6 && !res.headersSent;
406
413
 
407
414
  // PRE-STRIP a session already known to carry provider-authored thinking. The client
408
415
  // resends the whole poisoned history every turn, so without this each turn pays another
@@ -1076,7 +1083,8 @@ async function forwardRequest(
1076
1083
  // text + tool_use preserved. Tried once per request (thinkingStripped guard); if it
1077
1084
  // still fails, the provider pin below is the fallback.
1078
1085
  if (isSignatureRejection && !requestInfo.thinkingStripped
1079
- && canRetryBufferedBody && retryCount + 1 < maxAttempts && !res.headersSent) {
1086
+ && canRetryBufferedBody && canRepairBody) {
1087
+ console.log(`[Maxpool] Anthropic rejected a block: ${describeRejectedBlock(body, errorBody)}`);
1080
1088
  const { body: cleanBody, removed, converted } = stripForeignThinkingBlocks(body);
1081
1089
  if (cleanBody) {
1082
1090
  // Latch it so EVERY later turn is stripped up front instead of re-paying this
@@ -1085,7 +1093,35 @@ async function forwardRequest(
1085
1093
  console.log(`[Maxpool] Recovering session on Claude: stripped ${removed} provider thinking block(s), converted ${converted} provider search block(s) to text`);
1086
1094
  return forwardRequest(
1087
1095
  req, res, cleanBody, accountManager, upstream, retryCount + 1, hooks, reqId, ctx, logDir,
1088
- retryConfig, queueConfig, { ...requestInfo, thinkingStripped: true },
1096
+ retryConfig, queueConfig, { ...requestInfo, thinkingStripped: true, repairCount: repairCount + 1 },
1097
+ canRetryBufferedBody, canQueueBufferedBody, excludedIndexes,
1098
+ );
1099
+ }
1100
+ }
1101
+
1102
+ // COORDINATE REPAIR (runs after the broad strip found nothing, or found the wrong
1103
+ // thing). Anthropic pointed at an exact block index; trust that over our own shape
1104
+ // model. Its own flag, so it still fires on a request whose broad strip already ran.
1105
+ if (isSignatureRejection && !requestInfo.rejectedBlockStripped && canRetryBufferedBody && canRepairBody) {
1106
+ const { body: fixedBody, removed, type } = stripRejectedBlockClass(body, errorBody);
1107
+ if (fixedBody) {
1108
+ // Latch the session ONLY for the class the pre-strip can actually repair up
1109
+ // front. `stripForeignThinkingBlocks` never touches `redacted_thinking`, so
1110
+ // latching on it would make every later turn pay a rejected round-trip that
1111
+ // the pre-strip cannot prevent — and would mislabel the give-up message as a
1112
+ // GLM/Kimi story. Same reason `thinkingStripped` is set only for `thinking`:
1113
+ // setting it here would bar the broad strip on the retry.
1114
+ const preStripCanRepeat = type === 'thinking';
1115
+ if (preStripCanRepeat) accountManager.markSessionThinkingContaminated?.(requestInfo.sessionKey);
1116
+ console.log(`[Maxpool] Recovering session on Claude: Anthropic rejected a "${type}" block by index; removed ${removed} block(s) of that type`);
1117
+ return forwardRequest(
1118
+ req, res, fixedBody, accountManager, upstream, retryCount + 1, hooks, reqId, ctx, logDir,
1119
+ retryConfig, queueConfig, {
1120
+ ...requestInfo,
1121
+ rejectedBlockStripped: true,
1122
+ thinkingStripped: preStripCanRepeat || requestInfo.thinkingStripped,
1123
+ repairCount: repairCount + 1,
1124
+ },
1089
1125
  canRetryBufferedBody, canQueueBufferedBody, excludedIndexes,
1090
1126
  );
1091
1127
  }
@@ -1147,9 +1183,24 @@ async function forwardRequest(
1147
1183
  // Say only what ACTUALLY happened. `thinkingStripped` is set only when the strip
1148
1184
  // ran; without it we found nothing provider-shaped, so claiming we stripped —
1149
1185
  // or blaming GLM/Kimi — would misdirect the user.
1150
- const what = requestInfo.thinkingStripped
1151
- ? 'This session ran on GLM/Kimi earlier, and Anthropic will not accept parts of what they wrote. Maxpool repaired what it could and retried on Claude, but Anthropic still rejected the history.'
1152
- : "Anthropic rejected part of this session's history that maxpool could not repair automatically.";
1186
+ // Log the shape on the way out: this is the ONE place a give-up is observable,
1187
+ // and without it the two surviving explanations (a block the strip cannot see
1188
+ // vs. a body over the retry buffer) are indistinguishable in the log.
1189
+ console.log(`[Maxpool] Unrepaired signature 400: ${describeRejectedBlock(body, errorBody)} bufferable=${canRetryBufferedBody} stripped=${!!requestInfo.thinkingStripped}`);
1190
+ // A body over the retry buffer bars EVERY repair above without a word — the user
1191
+ // then sees "could not repair" for a transcript maxpool never even tried to fix.
1192
+ // Name that separately so the reason is actionable rather than mysterious.
1193
+ // `peek` (not the full class-strip) because on THIS path the body may be the very
1194
+ // one just declared too large to rewrite — a full parse+rebuild there costs 15ms
1195
+ // and a discarded 4.8MB Buffer on a 9.6MB body, to read one string.
1196
+ const rejectedType = canRetryBufferedBody ? peekRejectedBlockType(body, errorBody) : null;
1197
+ const what = !canRetryBufferedBody
1198
+ ? `This session's history is too large for maxpool to rewrite automatically (over ${Math.round(retryConfig.maxRetryBufferBytes / (1024 * 1024))}MB). Run /compact and it will keep going.`
1199
+ : rejectedType && rejectedType !== 'thinking' && rejectedType !== 'redacted_thinking'
1200
+ ? `Anthropic rejected a "${rejectedType}" block in this session's history, which maxpool cannot remove without losing conversation content.`
1201
+ : requestInfo.thinkingStripped || requestInfo.rejectedBlockStripped
1202
+ ? 'This session ran on GLM/Kimi earlier, and Anthropic will not accept parts of what they wrote. Maxpool repaired what it could and retried on Claude, but Anthropic still rejected the history.'
1203
+ : "Anthropic rejected part of this session's history that maxpool could not repair automatically.";
1153
1204
  const hint = provs.length === 0
1154
1205
  ? ''
1155
1206
  : provs.every(a => a.enabled === false)
@@ -1160,7 +1211,12 @@ async function forwardRequest(
1160
1211
  type: 'error',
1161
1212
  error: {
1162
1213
  type: 'invalid_request_error',
1163
- message: `Start a new session to keep working — this one cannot continue. ${what}${hint}`,
1214
+ // The lead depends on whether the session is actually RECOVERABLE. Telling a
1215
+ // user "this one cannot continue" and then "run /compact and it will keep
1216
+ // going" is two mutually exclusive remedies in one sentence.
1217
+ message: canRetryBufferedBody
1218
+ ? `Start a new session to keep working — this one cannot continue. ${what}${hint}`
1219
+ : `${what}${hint}`,
1164
1220
  },
1165
1221
  });
1166
1222
  return;
@@ -1506,6 +1562,20 @@ function unavailableMessage(accountManager, requestInfo = {}, retryAfter, willRe
1506
1562
  // MINUTES (not momentary), and raw seconds are unreadable. Scale the wording to the
1507
1563
  // actual wait and render it in human units, the way every other branch already does.
1508
1564
  const waitLong = Number.isFinite(retryAfter) && retryAfter >= 120;
1565
+
1566
+ // "at their limit" is a QUOTA claim. When the accounts are actually in short network
1567
+ // cooldowns (a connectivity blip drops every in-flight connection and cools each
1568
+ // account for ~5s), that claim sends the user to check quota and add accounts while
1569
+ // the fleet is healthy and recovers in seconds. Name the real cause.
1570
+ const census = accountManager.unavailabilityCensus?.(requestInfo.model);
1571
+ if (census && census.dominant === 'transient') {
1572
+ const netly = census.network > 0;
1573
+ const what = netly
1574
+ ? `${census.network} of ${census.total} Claude accounts are in a brief reconnect cooldown`
1575
+ : `all ${census.total} Claude accounts are momentarily busy`;
1576
+ return `No account can take this request right now — ${what}, not out of quota. Retry in ~${formatRetryDuration(retryAfter)}; it clears on its own.${providersBarredHint}`;
1577
+ }
1578
+
1509
1579
  return `No account can take this request right now — all ${claudeCount} Claude accounts${providersClause} are ${waitLong ? 'at their limit' : 'momentarily at their limit'}. Retry in ~${formatRetryDuration(retryAfter)}.${providersBarredHint}`;
1510
1580
  }
1511
1581
 
@@ -1520,7 +1590,7 @@ function isContextLengthError(errorBody) {
1520
1590
  return /exceeded model token limit|maximum context length|context length exceeded|context window (?:size )?(?:exceeded|too)|prompt is too long|input is too long|reduce the length of|too many (?:input )?tokens|request too large/i.test(errorBody);
1521
1591
  }
1522
1592
 
1523
- export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, classifyEffortRejection, repairEffort, isCapacitySignalStatus, isStrippableThinkingBlock, stripForeignThinkingBlocks, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, commitStreamGraceHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, isContextLengthError, streamResponse, startIdleRequestReaper };
1593
+ export const __serverTest = { unavailableMessage, computeQueueWindowMs, isRetriableUpstreamStatus, classifyEffortRejection, repairEffort, isCapacitySignalStatus, isStrippableThinkingBlock, stripForeignThinkingBlocks, parseRejectedBlockPath, stripRejectedBlockClass, peekRejectedBlockType, describeRejectedBlock, headerValue, getMaxpoolProfile, ensureQueueHeartbeat, clearQueueHeartbeat, commitStreamGraceHeartbeat, describeRequest, classifyRateLimit, detectTranscriptOrigin, isAnthropicIncompatBody, isContextLengthError, streamResponse, startIdleRequestReaper };
1524
1594
 
1525
1595
  async function readErrorBody(upstreamRes, limitBytes = 64 * 1024) {
1526
1596
  if (!upstreamRes.body) return '';
@@ -1708,6 +1778,119 @@ function isStrippableThinkingBlock(block) {
1708
1778
  return block?.type === 'thinking';
1709
1779
  }
1710
1780
 
1781
+ /**
1782
+ * Anthropic's signature 400 names the EXACT block it rejected:
1783
+ * "messages.29.content.58: Invalid `signature` in `thinking` block"
1784
+ * That coordinate is GROUND TRUTH. Every other repair here depends on maxpool's own
1785
+ * model of what a provider-authored block looks like, and that model is what silently
1786
+ * failed — the role gate above meant `stripForeignThinkingBlocks` returned "nothing to
1787
+ * remove" for a block Anthropic had just pointed at by index.
1788
+ *
1789
+ * Returns { mi, ci } or null.
1790
+ */
1791
+ function parseRejectedBlockPath(errorBody) {
1792
+ const s = String(errorBody || '');
1793
+ const m = /messages\.(\d+)\.content\.(\d+)/.exec(s);
1794
+ if (!m) return null;
1795
+ // A NESTED path — `messages.29.content.58.content.3` — points INSIDE the block at
1796
+ // [58], not at it. Taking the outer coordinate would name the wrong block: the user
1797
+ // would be told a "tool_result" was rejected when a thinking block nested in it is
1798
+ // the real culprit, and a class-strip keyed on that type would be wrong too.
1799
+ if (/^\.content\./.test(s.slice(m.index + m[0].length))) return null;
1800
+ return { mi: Number(m[1]), ci: Number(m[2]) };
1801
+ }
1802
+
1803
+ /**
1804
+ * The rejected block's TYPE only — no parse-and-rebuild. Used on the give-up path,
1805
+ * where the body may be the very one we just declared too large to rewrite (measured:
1806
+ * a full stripRejectedBlockClass on a 9.6MB body costs 15ms and allocates a 4.8MB
1807
+ * Buffer that is discarded, because only `.type` is ever read).
1808
+ */
1809
+ function peekRejectedBlockType(body, errorBody) {
1810
+ const path = parseRejectedBlockPath(errorBody);
1811
+ if (!path) return null;
1812
+ try {
1813
+ const json = JSON.parse(Buffer.isBuffer(body) ? body.toString('utf8') : String(body));
1814
+ return json?.messages?.[path.mi]?.content?.[path.ci]?.type || null;
1815
+ } catch {
1816
+ return null;
1817
+ }
1818
+ }
1819
+
1820
+ /**
1821
+ * One line naming exactly what Anthropic rejected: coordinate, role, block type, and
1822
+ * whether the body was even parseable. Without it, the two remaining explanations for a
1823
+ * silent give-up (a block shape the strip cannot see vs. a body over the retry buffer)
1824
+ * are indistinguishable in the log — and once the repair starts working, the successful
1825
+ * path logs the same line as the already-working one, so the question becomes
1826
+ * unanswerable. Types and roles only; no transcript content.
1827
+ */
1828
+ function describeRejectedBlock(body, errorBody) {
1829
+ const path = parseRejectedBlockPath(errorBody);
1830
+ if (!path) return 'coordinate=unparsed';
1831
+ try {
1832
+ const json = JSON.parse(Buffer.isBuffer(body) ? body.toString('utf8') : String(body));
1833
+ const msg = json?.messages?.[path.mi];
1834
+ const block = msg?.content?.[path.ci];
1835
+ return `coordinate=messages.${path.mi}.content.${path.ci} role=${msg?.role ?? 'MISSING'} type=${block?.type ?? 'MISSING'} blocks=${Array.isArray(msg?.content) ? msg.content.length : 'n/a'}`;
1836
+ } catch {
1837
+ return `coordinate=messages.${path.mi}.content.${path.ci} body=UNPARSEABLE`;
1838
+ }
1839
+ }
1840
+
1841
+ /**
1842
+ * Last-resort repair driven by the upstream's own coordinate, for a rejected block no
1843
+ * shape heuristic here recognised. Removes every block sharing the rejected block's
1844
+ * TYPE, on any role — fixing the whole class in ONE round-trip rather than replaying
1845
+ * once per bad block (a 47-block transcript would otherwise cost 47 rejected requests).
1846
+ *
1847
+ * Restricted to the thinking family on purpose: `text` / `tool_use` / `tool_result`
1848
+ * carry conversation content and tool pairing, so removing them would corrupt the
1849
+ * transcript rather than repair it. A rejected block outside that family returns null
1850
+ * and the 400 surfaces with its real cause intact.
1851
+ *
1852
+ * Returns { body, removed, type } — `body` is null when nothing was safe to remove.
1853
+ */
1854
+ function stripRejectedBlockClass(body, errorBody) {
1855
+ const path = parseRejectedBlockPath(errorBody);
1856
+ if (!path) return { body: null, removed: 0, type: null };
1857
+ try {
1858
+ const json = JSON.parse(Buffer.isBuffer(body) ? body.toString('utf8') : String(body));
1859
+ if (!Array.isArray(json?.messages)) return { body: null, removed: 0, type: null };
1860
+ const target = json.messages[path.mi]?.content?.[path.ci];
1861
+ const type = target?.type;
1862
+ // Only the thinking family is safe to drop wholesale (verified 2026-07-25: a
1863
+ // history with thinking blocks removed replays 200 OK, text + tool_use preserved).
1864
+ if (type !== 'thinking' && type !== 'redacted_thinking') {
1865
+ return { body: null, removed: 0, type: type || null };
1866
+ }
1867
+ let removed = 0;
1868
+ const messages = [];
1869
+ for (const msg of json.messages) {
1870
+ if (!Array.isArray(msg?.content)) { messages.push(msg); continue; }
1871
+ const kept = msg.content.filter(b => {
1872
+ if (b?.type !== type) return true;
1873
+ removed++;
1874
+ return false;
1875
+ });
1876
+ if (kept.length === msg.content.length) { messages.push(msg); continue; }
1877
+ // A turn stripping empties is DROPPED — an empty content array is itself invalid,
1878
+ // and keeping the original would resend the exact body that just 400'd. Except
1879
+ // messages[0], which must survive as a `user` turn (see the same guard above).
1880
+ if (kept.length === 0) {
1881
+ if (messages.length === 0) { messages.push({ ...msg, content: [{ type: 'text', text: '(content removed)' }] }); }
1882
+ continue;
1883
+ }
1884
+ messages.push({ ...msg, content: kept });
1885
+ }
1886
+ if (!removed) return { body: null, removed: 0, type };
1887
+ json.messages = messages;
1888
+ return { body: Buffer.from(JSON.stringify(json)), removed, type };
1889
+ } catch {
1890
+ return { body: null, removed: 0, type: null };
1891
+ }
1892
+ }
1893
+
1711
1894
  /**
1712
1895
  * Recovery for a provider-contaminated transcript: drop the assistant `thinking` /
1713
1896
  * `redacted_thinking` blocks whose signature Anthropic can't validate, so the session
@@ -1847,10 +2030,12 @@ function stripForeignThinkingBlocks(body) {
1847
2030
  // assistant turn but its result can be carried on the following user turn.
1848
2031
  const tools = convertForeignServerTools(msg.content, foreignToolIds);
1849
2032
  converted += tools.converted;
1850
- if (msg.role !== 'assistant') {
1851
- messages.push(tools.converted ? { ...msg, content: tools.content } : msg);
1852
- continue;
1853
- }
2033
+ // Thinking blocks are stripped on EVERY role, not just `assistant`. Anthropic
2034
+ // validates the signature wherever the block sits, so a role gate here made the
2035
+ // repair silently find NOTHING to remove — which left `thinkingStripped` false,
2036
+ // barred the session latch, and surfaced the 400 to the user as "maxpool could
2037
+ // not repair automatically" while healthy Claude accounts sat idle. Measured
2038
+ // 2026-08-06: 315 signature 400s, the broad strip finding nothing on a subset.
1854
2039
  let localRemoved = 0;
1855
2040
  const kept = tools.content.filter(block => {
1856
2041
  if (!isStrippableThinkingBlock(block)) return true;
@@ -1866,7 +2051,14 @@ function stripForeignThinkingBlocks(body) {
1866
2051
  // leaving it would resend the exact body that just 400'd while reporting success
1867
2052
  // and burning the single recovery attempt. Verified against the live API — the
1868
2053
  // resulting consecutive user messages are accepted (200 OK).
1869
- if (kept.length === 0) continue;
2054
+ // EXCEPT messages[0]: Anthropic requires the first message to be role `user`, so
2055
+ // dropping it leaves an `assistant`-first transcript that is rejected outright.
2056
+ // Reachable only since the role gate was removed — before that, non-assistant
2057
+ // turns were never dropped at all.
2058
+ if (kept.length === 0) {
2059
+ if (messages.length === 0) { messages.push({ ...msg, content: [{ type: 'text', text: '(content removed)' }] }); }
2060
+ continue;
2061
+ }
1870
2062
  messages.push({ ...msg, content: kept });
1871
2063
  }
1872
2064
  if (!removed && !converted) return { body: null, removed: 0, converted: 0 };
@@ -2381,8 +2573,18 @@ function getMaxpoolProfile(headers) {
2381
2573
  function prepareRuntimeProviders(accountManager, headers) {
2382
2574
  if (getMaxpoolProfile(headers) !== 'all') return;
2383
2575
 
2576
+ // Config-sourced providers already exist (resolved from GCP at startup). A `cc all`
2577
+ // session sends the SAME token as a header → dedup by token so we don't create a
2578
+ // duplicate provider for every request. The config provider wins (it persists, has
2579
+ // the right name, and carries quota state from earlier requests).
2580
+ const configTokens = new Set(
2581
+ (accountManager.accounts || [])
2582
+ .filter(a => a.configSourced && a.credential)
2583
+ .map(a => a.credential),
2584
+ );
2585
+
2384
2586
  const zaiToken = headerValue(headers, 'x-maxpool-zai-token');
2385
- if (zaiToken) {
2587
+ if (zaiToken && !configTokens.has(zaiToken)) {
2386
2588
  const opus = headerValue(headers, 'x-maxpool-zai-opus-model') || headerValue(headers, 'x-maxpool-zai-model') || 'glm-5.2';
2387
2589
  const sonnet = headerValue(headers, 'x-maxpool-zai-sonnet-model') || headerValue(headers, 'x-maxpool-zai-model') || opus;
2388
2590
  const haiku = headerValue(headers, 'x-maxpool-zai-haiku-model') || 'glm-5.1';
@@ -2401,7 +2603,7 @@ function prepareRuntimeProviders(accountManager, headers) {
2401
2603
  }
2402
2604
 
2403
2605
  const kimiToken = headerValue(headers, 'x-maxpool-kimi-token');
2404
- if (kimiToken) {
2606
+ if (kimiToken && !configTokens.has(kimiToken)) {
2405
2607
  // Fallback only — `cc all` always sends x-maxpool-kimi-model from the llm_config SSOT,
2406
2608
  // so this is what a bare/older client gets. Kept current deliberately: it read
2407
2609
  // 'kimi-k2.7' while the fleet had moved to k3.
package/src/tui.js CHANGED
@@ -405,6 +405,7 @@ export class TUI {
405
405
  case 'accounts': this._keyAccounts(k); break;
406
406
  case 'routing': this._keyRouting(k); break;
407
407
  case 'updates': this._keyUpdates(k); break;
408
+ case 'providers': this._keyProviders(k); break;
408
409
  case 'select': this._keySelect(k); break;
409
410
  case 'input': this._keyInput(k); break;
410
411
  case 'confirm': this._keyConfirm(k); break;
@@ -442,6 +443,8 @@ export class TUI {
442
443
  );
443
444
  } else if (k === 'u') {
444
445
  this.mode = 'updates';
446
+ } else if (k === 'p') {
447
+ this.mode = 'providers';
445
448
  }
446
449
  // Enable/disable lives ONLY under [a] Accounts now (with rename/delete/login) —
447
450
  // one home for every account mutation, instead of a duplicate top-level toggle.
@@ -582,6 +585,140 @@ export class TUI {
582
585
  }
583
586
  }
584
587
 
588
+ _keyProviders(k) {
589
+ if (k === 'a') {
590
+ this._providerAddStep('name');
591
+ } else if (k === 'd') {
592
+ this._startProviderSelection('delete');
593
+ } else if (k === 't') {
594
+ this._startProviderSelection('toggle');
595
+ } else if (k === 'esc' || k === 'q') {
596
+ this.mode = 'normal';
597
+ }
598
+ }
599
+
600
+ // Multi-step input for adding a provider. Steps: name → type → secret name.
601
+ _providerAddStep(step, prev) {
602
+ if (step === 'name') {
603
+ this.mode = 'input';
604
+ this.inputPrompt = 'Provider display name (e.g. glm-ahmed)';
605
+ this.inputBuf = '';
606
+ this.inputSensitive = false;
607
+ this.inputCb = value => {
608
+ const name = String(value || '').trim();
609
+ if (!name) { this.mode = 'providers'; return; }
610
+ if (this.am.accounts.some(a => a.name === name)) {
611
+ this._addLog(`Account "${name}" already exists`); this.mode = 'providers'; return;
612
+ }
613
+ this._providerAddStep('type', { name });
614
+ };
615
+ } else if (step === 'type') {
616
+ this.mode = 'input';
617
+ this.inputPrompt = `Type for ${prev.name} (zai or kimi)`;
618
+ this.inputBuf = 'zai';
619
+ this.inputSensitive = false;
620
+ this.inputCb = value => {
621
+ const provider = String(value || '').trim().toLowerCase();
622
+ if (provider !== 'zai' && provider !== 'kimi') { this._addLog('Type must be zai or kimi'); this.mode = 'providers'; return; }
623
+ this._providerAddStep('secret', { ...prev, provider });
624
+ };
625
+ } else if (step === 'secret') {
626
+ this.mode = 'input';
627
+ this.inputPrompt = `${prev.name}: GCP secret name OR paste API key directly`;
628
+ this.inputBuf = '';
629
+ this.inputSensitive = false;
630
+ this.inputCb = async value => {
631
+ const input = String(value || '').trim();
632
+ if (!input) { this.mode = 'providers'; return; }
633
+ // Heuristic: a GCP secret name is uppercase/dashes/underscores and short.
634
+ // An API key is long and contains dots/mixed-case/alphanumeric.
635
+ const looksLikeSecretName = /^[A-Z][A-Z0-9_-]{2,60}$/.test(input) && !input.includes('.');
636
+ if (looksLikeSecretName) {
637
+ await this._doAddProvider({ ...prev, secretName: input });
638
+ } else {
639
+ // Direct key paste — store in config (0600, same protection as OAuth tokens).
640
+ await this._doAddProvider({ ...prev, apiKey: input });
641
+ }
642
+ };
643
+ }
644
+ }
645
+
646
+ async _doAddProvider({ name, provider, secretName, apiKey }) {
647
+ this.mode = 'providers';
648
+ const isDirect = !secretName && apiKey;
649
+ if (secretName) this._addLog(`Resolving secret "${secretName}" from GCP…`);
650
+ else this._addLog(`Adding "${name}" with direct API key…`);
651
+ this.render();
652
+ try {
653
+ let token = null;
654
+ if (secretName) {
655
+ const { resolveSecret } = await import('../secret-resolver.js');
656
+ token = await resolveSecret(secretName);
657
+ if (!token) {
658
+ this._addLog(`✗ Secret "${secretName}" not found in GCP — create it first: gcloud secrets create ${secretName} --data-file=-`);
659
+ return;
660
+ }
661
+ } else {
662
+ token = apiKey;
663
+ }
664
+ // Add to config (persistence) and activate immediately.
665
+ const { atomicConfigUpdate } = await import('../config.js');
666
+ await atomicConfigUpdate(cfg => {
667
+ if (!Array.isArray(cfg.providers)) cfg.providers = [];
668
+ const entry = { name, provider };
669
+ if (secretName) entry.secretName = secretName;
670
+ else entry.apiKey = apiKey;
671
+ cfg.providers.push(entry);
672
+ });
673
+ // Activate in the running process.
674
+ this.am.loadConfigProviders([{ name, provider, secretName, token }]);
675
+ const a = this.am.accounts.find(a => a.name === name);
676
+ if (a && secretName) a.secretName = secretName;
677
+ this._addLog(`✓ Added "${name}" (${provider})${secretName ? ` — GCP secret "${secretName}"` : ' — direct key'}`);
678
+ } catch (err) {
679
+ this._addLog(`✗ Failed to add provider: ${err.message}`);
680
+ }
681
+ }
682
+
683
+ _startProviderSelection(action) {
684
+ const providers = this.am.accounts.filter(a => a.type === 'provider');
685
+ if (!providers.length) { this._addLog('No providers to manage'); return; }
686
+ this._selOptions = providers.map(a => a.index);
687
+ this._selLabels = providers.map(a => {
688
+ const enabled = a.enabled !== false;
689
+ const tag = a.configSourced ? ' (GCP)' : ' (header)';
690
+ const secret = a.secretName ? ` [${a.secretName}]` : '';
691
+ return `${a.name}${tag}${secret}${enabled ? '' : ' ✕'}`;
692
+ });
693
+ this.selAction = action;
694
+ this.mode = 'select';
695
+ }
696
+
697
+ _renderProviders(buf, width) {
698
+ const providers = this.am.accounts.filter(a => a.type === 'provider');
699
+ buf.push(`${bold('Providers (GLM / Kimi)')} ${dim('— managed via GCP Secret Manager')}`);
700
+ buf.push('');
701
+ if (!providers.length) {
702
+ buf.push(dim(' No providers configured.'));
703
+ buf.push('');
704
+ buf.push(dim(' Press ') + bold('a') + dim(' to add one. You\'ll need:'));
705
+ buf.push(dim(' 1. An API key from z.ai (GLM) or Moonshot (Kimi)'));
706
+ buf.push(dim(' 2. The key stored in GCP: ') + 'gcloud secrets create <name> --data-file=-');
707
+ buf.push(dim(' 3. The GCP secret name (e.g. RESTRICTED_MAXPOOL_ZAI_NEW)'));
708
+ return;
709
+ }
710
+ for (const a of providers) {
711
+ const enabled = a.enabled !== false;
712
+ const tag = a.configSourced ? dim(' (GCP)') : dim(' (header)');
713
+ const secret = a.secretName ? dim(` [${a.secretName}]`) : '';
714
+ const status = a.status === 'error' ? red(a.lastError || 'error')
715
+ : enabled ? green('active') : red('✕');
716
+ const q = a.quota;
717
+ const ses = q?.providerSes != null ? ` Ses ${Math.round(q.providerSes * 100)}%` : '';
718
+ buf.push(` ${enabled ? '' : dim('')} ${bold(a.name)} ${a.provider}${tag}${secret} ${status}${ses}`);
719
+ }
720
+ }
721
+
585
722
  _keyRouting(k) {
586
723
  if (k === 'a') {
587
724
  this._confirm(
@@ -641,6 +778,51 @@ export class TUI {
641
778
  }
642
779
 
643
780
  _keySelect(k) {
781
+ // Provider selection (from the providers screen) uses its own option list.
782
+ if (this._selOptions && (this.selAction === 'delete' || this.selAction === 'toggle')
783
+ && this.mode === 'select' && this.am.accounts[this._selOptions[0]]?.type === 'provider') {
784
+ const opts = this._selOptions;
785
+ const position = Math.max(0, opts.indexOf(this.selIdx));
786
+ if (k === 'up' || k === 'k') this.selIdx = opts[Math.max(0, position - 1)] ?? this.selIdx;
787
+ else if (k === 'down' || k === 'j') this.selIdx = opts[Math.min(opts.length - 1, position + 1)] ?? this.selIdx;
788
+ else if (k === 'enter') {
789
+ const account = this.am.accounts[this.selIdx];
790
+ if (!account) { this.mode = 'providers'; return; }
791
+ if (this.selAction === 'toggle') {
792
+ const enable = !account.enabled;
793
+ this._confirm(
794
+ `${enable ? 'Enable' : 'Disable'} "${account.name}"?`,
795
+ enable ? 'Allow this provider to receive requests again.' : 'Stop routing to it. Active requests continue.',
796
+ () => { this._doToggle(this.selIdx, enable); this.mode = 'providers'; },
797
+ );
798
+ } else if (this.selAction === 'delete') {
799
+ this._confirm(
800
+ `Delete provider "${account.name}"?`,
801
+ account.configSourced
802
+ ? 'Removes it from config and GCP reference. The GCP secret itself stays — delete it separately if needed.'
803
+ : 'Removes the runtime provider. It returns on the next request that sends its token.',
804
+ async () => {
805
+ if (account.configSourced) {
806
+ try {
807
+ const { atomicConfigUpdate } = await import('../config.js');
808
+ await atomicConfigUpdate(cfg => {
809
+ if (Array.isArray(cfg.providers)) {
810
+ cfg.providers = cfg.providers.filter(p => p.name !== account.name);
811
+ }
812
+ });
813
+ } catch (err) { this._addLog(`Config update failed: ${err.message}`); }
814
+ }
815
+ this.am.removeAccount(this.selIdx);
816
+ this._addLog(`Deleted provider "${account.name}"`);
817
+ this.mode = 'providers';
818
+ },
819
+ );
820
+ }
821
+ }
822
+ else if (k === 'esc' || k === 'q') { this.mode = 'providers'; }
823
+ return;
824
+ }
825
+
644
826
  const selectable = this._selectableIndexes(this.selAction);
645
827
  const position = Math.max(0, selectable.indexOf(this.selIdx));
646
828
  if (k === 'up' || k === 'k') this.selIdx = selectable[Math.max(0, position - 1)] ?? this.selIdx;
@@ -1283,6 +1465,13 @@ export class TUI {
1283
1465
  // Pad to fill
1284
1466
  while (lines.length < H - footerH) lines.push('');
1285
1467
 
1468
+ // Providers panel — shown when the user is on the providers screen
1469
+ if (this.mode === 'providers' || this.mode === 'select') {
1470
+ const pLines = [];
1471
+ this._renderProviders(pLines, W);
1472
+ lines.push(...pLines);
1473
+ }
1474
+
1286
1475
  // ── Footer
1287
1476
  lines.push(' ' + dim('─'.repeat(W - 2)));
1288
1477
  if (this.mode === 'confirm') lines.push(` ${this.confirmDetail}`);
@@ -1405,7 +1594,12 @@ export class TUI {
1405
1594
 
1406
1595
  let line = ` ${sel}${cur} ${name} ${type} ${status} ${l1} ${bar(r1, bw, t1)}`;
1407
1596
  if (showBoth) {
1408
- line += ` ${l2} ${bar(r2, bw, t2)}`;
1597
+ // "no weekly cap on this plan" is a DIFFERENT state from "not read yet", and
1598
+ // both render as an empty bar. Say which, so a healthy uncapped account
1599
+ // (privacy@gomokka.com, measured 2026-08-06) doesn't read as broken.
1600
+ line += (r2 == null && q.weeklyAbsent)
1601
+ ? ` ${l2} ${emptyBar('none', bw)}`
1602
+ : ` ${l2} ${bar(r2, bw, t2)}`;
1409
1603
  }
1410
1604
  const weekly = weeklyPolicyText(this.am, a);
1411
1605
  if (weekly) line += ` ${weekly}`;
@@ -1481,7 +1675,10 @@ export class TUI {
1481
1675
  let sesCell, wkCell, note = '';
1482
1676
  if (q.providerSes != null || q.providerWk != null) {
1483
1677
  sesCell = q.providerSes != null ? bar(q.providerSes, bw, q.providerSesReset) : emptyBar('—', bw);
1484
- wkCell = q.providerWk != null ? bar(q.providerWk, bw, q.providerWkReset) : emptyBar('—', bw);
1678
+ // z.ai's `max` plan genuinely has NO weekly token window (measured 2026-08-06:
1679
+ // one TOKENS_LIMIT, unit 3 = 5h). "none" says that; "—" read as a broken probe.
1680
+ wkCell = q.providerWk != null ? bar(q.providerWk, bw, q.providerWkReset)
1681
+ : emptyBar(q.weeklyAbsent ? 'none' : '—', bw);
1485
1682
  note = this._probeHealthNote(a);
1486
1683
  } else if (q.providerQuotaSource === 'console-only') {
1487
1684
  sesCell = emptyBar('n/a', bw);
@@ -1527,13 +1724,15 @@ export class TUI {
1527
1724
  _renderFooter() {
1528
1725
  switch (this.mode) {
1529
1726
  case 'normal':
1530
- return ` ${bold('a')} Accounts ${bold('m')} Routing ${bold('s')} Sync ${bold('u')} Updates ${bold('r')} Restart ${bold('q')} Stop`;
1727
+ return ` ${bold('a')} Accounts ${bold('p')} Providers ${bold('m')} Routing ${bold('s')} Sync ${bold('u')} Updates ${bold('r')} Restart ${bold('q')} Stop`;
1531
1728
  case 'updates': {
1532
1729
  const state = this._autoUpdateOn() ? green('on') : dim('off');
1533
1730
  return ` ${bold('c')} Check & apply now ${bold('t')} Automatic updates: ${state} ↻ ${bold('Esc')} Back`;
1534
1731
  }
1535
1732
  case 'accounts':
1536
1733
  return ` ${bold('l')} Login/re-auth (browser) ${bold('k')} API key ${bold('n')} Rename ${bold('t')} Enable/disable ${bold('d')} Delete ${bold('Esc')} Back`;
1734
+ case 'providers':
1735
+ return ` ${bold('a')} Add provider ${bold('d')} Delete ${bold('t')} Enable/disable ${bold('Esc')} Back`;
1537
1736
  case 'routing': {
1538
1737
  // Show the CURRENT cross-provider policy inline so pressing f visibly changes it
1539
1738
  // right here at the footer (the policy also renders in the header, far from the