pi-multikey 1.7.1 → 1.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/pool.ts CHANGED
@@ -6,20 +6,23 @@
6
6
  * concurrent subagents across different keys automatically.
7
7
  */
8
8
 
9
- import type { PoolConfig } from "./config.ts";
9
+ import type { KeyCredential, PoolConfig } from "./config.ts";
10
10
  import { maskKey } from "./config.ts";
11
11
 
12
- export type KeyOutcome = "ok" | "rate_limited" | "invalid" | "error";
12
+ export type KeyOutcome = "ok" | "rate_limited" | "quota_exhausted" | "invalid" | "error";
13
13
 
14
14
  export interface Lease {
15
15
  key: string;
16
16
  label: string;
17
17
  acquiredAt: number;
18
+ /** OAuth credential behind this key, when it has one (Cline accounts). */
19
+ credential?: KeyCredential;
18
20
  }
19
21
 
20
22
  interface KeyStat {
21
23
  ok: number;
22
24
  rateLimited: number;
25
+ quotaLimited: number;
23
26
  invalid: number;
24
27
  errors: number;
25
28
  inflight: number;
@@ -29,7 +32,7 @@ interface KeyStat {
29
32
  }
30
33
 
31
34
  function newStat(): KeyStat {
32
- return { ok: 0, rateLimited: 0, invalid: 0, errors: 0, inflight: 0, cooldownUntil: 0, lastUsed: 0 };
35
+ return { ok: 0, rateLimited: 0, quotaLimited: 0, invalid: 0, errors: 0, inflight: 0, cooldownUntil: 0, lastUsed: 0 };
33
36
  }
34
37
 
35
38
  export class KeyPool {
@@ -49,10 +52,10 @@ export class KeyPool {
49
52
  return s;
50
53
  }
51
54
 
52
- private enabledKeys(): { key: string; label: string }[] {
55
+ private enabledKeys(): { key: string; label: string; credential?: KeyCredential }[] {
53
56
  return this.config.keys
54
57
  .filter((k) => k.enabled !== false)
55
- .map((k) => ({ key: k.key, label: k.label ?? maskKey(k.key) }));
58
+ .map((k) => ({ key: k.key, label: k.label ?? maskKey(k.key), credential: k.credential }));
56
59
  }
57
60
 
58
61
  get size(): number {
@@ -85,7 +88,7 @@ export class KeyPool {
85
88
  const chosen = candidates[0]!;
86
89
  chosen.stat.inflight++;
87
90
  chosen.stat.lastUsed = now;
88
- return { key: chosen.key, label: chosen.label, acquiredAt: now };
91
+ return { key: chosen.key, label: chosen.label, acquiredAt: now, credential: chosen.credential };
89
92
  }
90
93
 
91
94
  // All keys cooling: wait for the earliest recovery (bounded to 60s).
@@ -115,6 +118,14 @@ export class KeyPool {
115
118
  s.cooldownUntil = Math.max(s.cooldownUntil, now + (cooldownMs ?? this.config.cooldownMs ?? 20_000));
116
119
  s.cooldownReason = "429";
117
120
  break;
121
+ case "quota_exhausted":
122
+ // Daily per-account quota (Cline free models): the server tells us when
123
+ // it resets ("Try again in 23h 59m"), so the cooldown is hours, not the
124
+ // 20s rate-limit rotation. A sane fallback if the parse ever fails.
125
+ s.quotaLimited++;
126
+ s.cooldownUntil = Math.max(s.cooldownUntil, now + (cooldownMs ?? 30 * 60_000));
127
+ s.cooldownReason = "daily limit";
128
+ break;
118
129
  case "invalid":
119
130
  s.invalid++;
120
131
  s.cooldownUntil = Math.max(s.cooldownUntil, now + (this.config.invalidKeyCooldownMs ?? 600_000));
@@ -147,6 +158,34 @@ export class KeyPool {
147
158
  });
148
159
  }
149
160
 
161
+ /**
162
+ * Persist a refreshed OAuth credential onto the key's config entry (the
163
+ * access token also becomes the key value so every consumer sees the same
164
+ * token). Live stats move with the rotation — they are keyed by the key
165
+ * string, so the entry's counters (including its in-flight count) must be
166
+ * re-keyed to the new token. The caller saves the config to disk.
167
+ */
168
+ applyClineCredential(lease: Lease, update: { accessToken: string; refreshToken: string; expiresAt?: number }): void {
169
+ const entry = this.config.keys.find(
170
+ (k) => k.credential === lease.credential || (k.credential && lease.credential && k.credential.refreshToken === lease.credential.refreshToken),
171
+ );
172
+ if (!entry?.credential) return;
173
+ const oldKey = entry.key;
174
+ entry.credential.refreshToken = update.refreshToken;
175
+ entry.credential.accessToken = update.accessToken;
176
+ entry.credential.expiresAt = update.expiresAt;
177
+ entry.key = update.accessToken;
178
+ if (oldKey !== update.accessToken) {
179
+ const stat = this.stats.get(oldKey);
180
+ if (stat) {
181
+ this.stats.delete(oldKey);
182
+ this.stats.set(update.accessToken, stat);
183
+ }
184
+ }
185
+ lease.key = update.accessToken;
186
+ lease.credential = entry.credential;
187
+ }
188
+
150
189
  /** One row per configured key, for TUI status display. */
151
190
  statusRows(): {
152
191
  label: string;
@@ -155,10 +194,12 @@ export class KeyPool {
155
194
  inflight: number;
156
195
  ok: number;
157
196
  rateLimited: number;
197
+ quotaLimited: number;
158
198
  invalid: number;
159
199
  errors: number;
160
200
  cooldownRemainingMs: number;
161
201
  cooldownReason?: string;
202
+ credential?: KeyCredential;
162
203
  }[] {
163
204
  const now = Date.now();
164
205
  return this.config.keys.map((k) => {
@@ -170,10 +211,12 @@ export class KeyPool {
170
211
  inflight: s.inflight,
171
212
  ok: s.ok,
172
213
  rateLimited: s.rateLimited,
214
+ quotaLimited: s.quotaLimited,
173
215
  invalid: s.invalid,
174
216
  errors: s.errors,
175
217
  cooldownRemainingMs: Math.max(0, s.cooldownUntil - now),
176
218
  cooldownReason: s.cooldownReason,
219
+ credential: k.credential,
177
220
  };
178
221
  });
179
222
  }
package/presets.ts CHANGED
@@ -6,6 +6,8 @@
6
6
  * live probes — see README "Presets" for the evidence trail.
7
7
  */
8
8
 
9
+ import { createHash } from "node:crypto";
10
+
9
11
  import type { PoolConfig, PoolModelConfig } from "./config.ts";
10
12
 
11
13
  export interface Preset {
@@ -55,34 +57,13 @@ export const PRESETS: Preset[] = [
55
57
  {
56
58
  id: "b-ai",
57
59
  name: "B.AI",
58
- description: "api.b.ai — DeepSeek V4 Flash (+vision), Hunyuan Hy3, MiMo V2.5, Qwen3.8, GLM 5.3 (6 models)",
60
+ description: "api.b.ai — Hunyuan Hy3, MiMo V2.5, Qwen3.8, GLM 5.3 (4 models)",
59
61
  defaultPoolId: "bai",
60
62
  baseUrl: "https://api.b.ai/v1",
61
63
  api: "openai-completions",
62
64
  compat: BAI_COMPAT,
63
65
  keyHint: "https://www.b.ai/ → API Keys (one entry per key; multiple keys share the load)",
64
66
  models: [
65
- {
66
- // docs.b.ai/llmservice/models/deepseek-v4-flash + api-docs.deepseek.com/guides/thinking_mode
67
- // Native effort tiers: low/high/max (medium & xhigh alias to high server-side).
68
- id: "deepseek-v4-flash",
69
- name: "DeepSeek V4 Flash",
70
- reasoning: true,
71
- input: ["text"],
72
- contextWindow: 1_000_000,
73
- maxTokens: 384_000,
74
- thinkingLevelMap: levels({ off: "none", low: "low", high: "high", max: "max" }),
75
- },
76
- {
77
- // No public card; experimental vision build on the same V4 Flash backbone.
78
- id: "deepseek-v4-flash-vision-exp",
79
- name: "DeepSeek V4 Flash Vision (exp)",
80
- reasoning: true,
81
- input: ["text", "image"],
82
- contextWindow: 1_000_000,
83
- maxTokens: 384_000,
84
- thinkingLevelMap: levels({ off: "none", low: "low", high: "high", max: "max" }),
85
- },
86
67
  {
87
68
  // docs.b.ai/llmservice/models/hy3 — Tencent modes: no_think / think_low / think_high.
88
69
  id: "hy3",
@@ -131,7 +112,7 @@ export const PRESETS: Preset[] = [
131
112
  {
132
113
  id: "opencode-zen",
133
114
  name: "OpenCode Zen",
134
- description: "opencode.ai/zen free tier — Big Pickle, DeepSeek V4 Flash, MiMo V2.5, Hy3, Ling 3.0 Fin, Nemotron 3 Ultra/Lightning, Muse Spark 1.2 (8 free models)",
115
+ description: "opencode.ai/zen free tier — Big Pickle, MiMo V2.5, Ling 3.0 Fin, Nemotron 3 Ultra/Lightning, Muse Spark 1.3 (6 free models)",
135
116
  defaultPoolId: "zen",
136
117
  baseUrl: "https://opencode.ai/zen/v1",
137
118
  api: "openai-completions",
@@ -148,22 +129,10 @@ export const PRESETS: Preset[] = [
148
129
  maxTokens: 32_000,
149
130
  compat: ZEN_CHAT_COMPAT,
150
131
  },
151
- {
152
- // DeepSeek V4 Flash on the Zen FREE tier — id confirmed via GET /zen/v1/models
153
- // (deepseek-v4-flash-free); serving limits 200K/128K from models.dev `opencode`
154
- // provider. Thinking levels mirror the b.ai preset's deepseek-v4-flash
155
- // (off/low/high/max; medium & xhigh alias to high server-side).
156
- id: "deepseek-v4-flash-free",
157
- name: "DeepSeek V4 Flash Free",
158
- reasoning: true,
159
- input: ["text"],
160
- contextWindow: 200_000,
161
- maxTokens: 128_000,
162
- thinkingLevelMap: levels({ off: "none", low: "low", high: "high", max: "max" }),
163
- compat: ZEN_CHAT_COMPAT,
164
- },
165
132
  {
166
133
  // Xiaomi MiMo V2.5 omni; raw model is 1M ctx but the Zen FREE tier serves 200K/32K.
134
+ // Repo metadata: inputs text/image/audio/video (pi tracks text + image),
135
+ // reasoning via separate reasoning_content stream, no reasoning_options.
167
136
  id: "mimo-v2.5-free",
168
137
  name: "MiMo V2.5 Free",
169
138
  reasoning: true,
@@ -172,19 +141,6 @@ export const PRESETS: Preset[] = [
172
141
  maxTokens: 32_000,
173
142
  compat: ZEN_CHAT_COMPAT,
174
143
  },
175
- {
176
- // Tencent Hy3; raw model is 262K ctx but the Zen FREE tier serves 190K/64K.
177
- // thinkingFormat defaults to reasoning_effort: levels map straight to Hy3's
178
- // low/medium/high effort tiers; no off toggle on the free endpoint.
179
- id: "hy3-free",
180
- name: "Hy3 Free",
181
- reasoning: true,
182
- input: ["text"],
183
- contextWindow: 190_000,
184
- maxTokens: 64_000,
185
- thinkingLevelMap: levels({ low: "low", medium: "medium", high: "high" }),
186
- compat: ZEN_CHAT_COMPAT,
187
- },
188
144
  {
189
145
  // Finance-tuned Ling 3.0 Flash; reasoning toggle only (no effort tiers).
190
146
  id: "ling-3.0-flash-fin-free",
@@ -216,26 +172,127 @@ export const PRESETS: Preset[] = [
216
172
  compat: ZEN_CHAT_COMPAT,
217
173
  },
218
174
  {
219
- // Meta Muse Spark 1.2 Contributor Free — OpenAI Responses API endpoint (not chat
220
- // completions), effort tiers minimal..xhigh, no off. compat mirrors pi's catalog.
221
- id: "muse-spark-1.2-contributor-free",
222
- name: "Muse Spark 1.2 Contributor Free",
175
+ // Meta Muse Spark 1.3 Contributor Free — OpenAI Responses API endpoint (not chat
176
+ // completions). Repo metadata: no reasoning_options (always-on reasoning, no
177
+ // effort control), reasoning bundled into content (no separate stream field);
178
+ // inputs text/image/video/pdf/audio (pi tracks text + image).
179
+ // muse-spark-1.2-contributor-free was removed: legacy variant no longer in the
180
+ // free-model list at opencode.ai/docs/zen.
181
+ id: "muse-spark-1.3-contributor-free",
182
+ name: "Muse Spark 1.3 Contributor Free",
223
183
  api: "openai-responses",
224
184
  reasoning: true,
225
185
  input: ["text", "image"],
226
186
  contextWindow: 1_048_576,
227
187
  maxTokens: 131_072,
228
- thinkingLevelMap: levels({ minimal: "minimal", low: "low", medium: "medium", high: "high", xhigh: "xhigh" }),
229
188
  compat: { sessionAffinityFormat: "openai-nosession" },
230
189
  },
231
190
  ],
232
191
  },
192
+ {
193
+ // Cline's free-model promotion: a Cline account (OAuth, no static API key)
194
+ // gets a daily per-model quota on api.cline.bot's OpenAI-compatible API.
195
+ // The lineup rotates — retired ids answer "model not found" — so the
196
+ // `_preset` sync machinery is the intended way to receive lineup updates.
197
+ // Context/output limits are best-effort (server-enforced); tune per model
198
+ // in multikey.json if a provider rejects long conversations.
199
+ id: "cline-free",
200
+ name: "Cline Free",
201
+ description:
202
+ "api.cline.bot — Cline account free tier: DeepSeek V4 Flash, Longcat 2.0, Laguna S 2.1, GLM 5.2 (daily per-model quota, lineup rotates)",
203
+ defaultPoolId: "cline",
204
+ baseUrl: "https://api.cline.bot/api/v1",
205
+ api: "openai-completions",
206
+ keyHint: "Cline account — use 'Sign in with Cline (device flow)', or paste an access token from ~/.cline/data/secrets.json",
207
+ models: [
208
+ {
209
+ id: "deepseek/deepseek-v4-flash",
210
+ name: "DeepSeek V4 Flash",
211
+ reasoning: true,
212
+ input: ["text"],
213
+ contextWindow: 1_000_000,
214
+ maxTokens: 131_072,
215
+ },
216
+ {
217
+ id: "meituan/longcat-2.0",
218
+ name: "Longcat 2.0",
219
+ reasoning: true,
220
+ input: ["text"],
221
+ contextWindow: 1_000_000,
222
+ maxTokens: 131_072,
223
+ },
224
+ {
225
+ id: "poolside/laguna-s-2.1:free",
226
+ name: "Laguna S 2.1 (Free)",
227
+ reasoning: true,
228
+ input: ["text"],
229
+ // Poolside doesn't publish the window; safe default, tune if needed.
230
+ contextWindow: 128_000,
231
+ maxTokens: 16_384,
232
+ },
233
+ {
234
+ id: "z-ai/glm-5.2:free",
235
+ name: "GLM 5.2 (Free)",
236
+ reasoning: true,
237
+ input: ["text"],
238
+ contextWindow: 200_000,
239
+ maxTokens: 131_072,
240
+ },
241
+ ],
242
+ },
233
243
  ];
234
244
 
235
245
  export function findPreset(id: string): Preset | undefined {
236
246
  return PRESETS.find((p) => p.id === id);
237
247
  }
238
248
 
249
+ /**
250
+ * Stable fingerprint of a preset's model list (sha256, first 16 hex chars).
251
+ * Both sides of the comparison come from presets.ts builds, so JSON key order
252
+ * is deterministic. Covers models only — compat/api/description changes don't
253
+ * trigger update prompts.
254
+ */
255
+ export function presetFingerprint(preset: Preset): string {
256
+ return createHash("sha256").update(JSON.stringify(preset.models)).digest("hex").slice(0, 16);
257
+ }
258
+
259
+ export interface PresetModelDiff {
260
+ /** Shipped by the preset but missing from the pool. */
261
+ added: PoolModelConfig[];
262
+ /** Still in the pool but no longer shipped by the preset. */
263
+ removed: PoolModelConfig[];
264
+ /** Same model id, different spec (per-field comparison, order-insensitive). */
265
+ changed: { id: string; fields: string[] }[];
266
+ }
267
+
268
+ /** Compare a pool's current models against the shipped preset models. */
269
+ export function diffPresetModels(poolModels: PoolModelConfig[], presetModels: PoolModelConfig[]): PresetModelDiff {
270
+ const poolById = new Map(poolModels.map((m) => [m.id, m]));
271
+ const presetById = new Map(presetModels.map((m) => [m.id, m]));
272
+ const added = presetModels.filter((m) => !poolById.has(m.id));
273
+ const removed = poolModels.filter((m) => !presetById.has(m.id));
274
+ const changed: PresetModelDiff["changed"] = [];
275
+ for (const presetModel of presetModels) {
276
+ const poolModel = poolById.get(presetModel.id);
277
+ if (!poolModel) continue;
278
+ const keys = new Set([...Object.keys(poolModel), ...Object.keys(presetModel)]);
279
+ const pool = poolModel as unknown as Record<string, unknown>;
280
+ const preset = presetModel as unknown as Record<string, unknown>;
281
+ const fields = [...keys].filter((key) => JSON.stringify(pool[key]) !== JSON.stringify(preset[key]));
282
+ if (fields.length > 0) changed.push({ id: presetModel.id, fields });
283
+ }
284
+ return { added, removed, changed };
285
+ }
286
+
287
+ /** Human-readable diff lines for prompts and menus (may be empty). */
288
+ export function describePresetDiff(diff: PresetModelDiff): string[] {
289
+ const lines: string[] = [];
290
+ if (diff.added.length > 0) lines.push(`+ added by preset: ${diff.added.map((m) => m.id).join(", ")}`);
291
+ if (diff.removed.length > 0) lines.push(`− removed from preset: ${diff.removed.map((m) => m.id).join(", ")}`);
292
+ for (const change of diff.changed) lines.push(`~ changed: ${change.id} (${change.fields.join(", ")})`);
293
+ return lines;
294
+ }
295
+
239
296
  /** Materialize a preset into a pool config with the given keys. */
240
297
  export function poolFromPreset(preset: Preset, poolId: string, keys: string[]): PoolConfig {
241
298
  return {
@@ -249,5 +306,7 @@ export function poolFromPreset(preset: Preset, poolId: string, keys: string[]):
249
306
  keys: keys.map((key, i) => ({ key, label: `key-${i + 1}`, enabled: true })),
250
307
  // Deep copy so per-pool edits never mutate the shipped preset.
251
308
  models: JSON.parse(JSON.stringify(preset.models)) as PoolModelConfig[],
309
+ // Track the preset version so future preset updates can offer a one-time align.
310
+ _preset: { id: preset.id, fingerprint: presetFingerprint(preset) },
252
311
  };
253
312
  }
package/probe.ts CHANGED
@@ -1,4 +1,4 @@
1
- import { endpointHeaders } from "./config.ts";
1
+ import { endpointHeaders, endpointIdentityHeaders } from "./config.ts";
2
2
 
3
3
  /**
4
4
  * Endpoint probing: auto-detect the auth header style and fetch the model list.
@@ -63,7 +63,7 @@ async function fetchJson(url: string, headers: Record<string, string>, timeoutMs
63
63
  try {
64
64
  const response = await fetch(url, {
65
65
  method: "GET",
66
- headers: { Accept: "application/json", ...endpointHeaders(baseUrl ?? url), ...headers },
66
+ headers: { Accept: "application/json", ...endpointHeaders(baseUrl ?? url), ...endpointIdentityHeaders(baseUrl ?? url), ...headers },
67
67
  signal: AbortSignal.timeout(timeoutMs),
68
68
  });
69
69
  let body: unknown;
@@ -144,16 +144,27 @@ export function parseModelsResponse(body: unknown): RemoteModel[] {
144
144
  * Verify a key with a minimal chat completion (a few tokens at most). Returns
145
145
  * "ok" when auth was accepted (2xx, or 4xx that clearly got past auth like a
146
146
  * bad-model/params 400/404), "rejected" on 401/403, "error" on network trouble.
147
+ * `onLog` receives the server's rejection body so the TUI can show why.
147
148
  */
148
- async function chatProbe(baseUrl: string, style: AuthStyle, key: string, modelId: string): Promise<"ok" | "rejected" | "error"> {
149
+ async function chatProbe(baseUrl: string, style: AuthStyle, key: string, modelId: string, onLog?: (line: string) => void): Promise<"ok" | "rejected" | "error"> {
149
150
  try {
150
151
  const response = await fetch(`${trimSlash(baseUrl)}/chat/completions`, {
151
152
  method: "POST",
152
- headers: { "Content-Type": "application/json", ...endpointHeaders(baseUrl), ...authHeaders(style, key) },
153
+ headers: {
154
+ "Content-Type": "application/json",
155
+ ...endpointHeaders(baseUrl),
156
+ ...endpointIdentityHeaders(baseUrl),
157
+ ...authHeaders(style, key),
158
+ },
153
159
  body: JSON.stringify({ model: modelId, max_tokens: 4, messages: [{ role: "user", content: "ping" }] }),
154
160
  signal: AbortSignal.timeout(CHAT_TIMEOUT_MS),
155
161
  });
156
- if (response.status === 401 || response.status === 403) return "rejected";
162
+ if (response.status === 401 || response.status === 403) {
163
+ const text = await response.text().catch(() => "");
164
+ const flat = text.replace(/\s+/g, " ").trim();
165
+ if (flat) onLog?.(` ↳ server said: ${flat.slice(0, 160)}`);
166
+ return "rejected";
167
+ }
157
168
  return "ok"; // 2xx, or 4xx past auth (bad model / params) — auth itself worked.
158
169
  } catch {
159
170
  return "error";
@@ -221,7 +232,7 @@ export async function probeEndpoint(
221
232
  let verified: AuthStyle | undefined;
222
233
  for (const style of styles) {
223
234
  emit(`auth check: 1-token chat on "${chatModelId}" with ${style === "bearer" ? "Bearer" : "x-api-key"}…`);
224
- const verdict = await chatProbe(baseUrl, style, key, chatModelId);
235
+ const verdict = await chatProbe(baseUrl, style, key, chatModelId, emit);
225
236
  if (verdict === "ok") {
226
237
  verified = style;
227
238
  emit(` accepted ✓ (style: ${style === "bearer" ? "Authorization: Bearer" : "x-api-key"})`);
package/stream.ts CHANGED
@@ -1,7 +1,10 @@
1
1
  /**
2
2
  * Rotating streamSimple: wraps the underlying pi-ai API implementation, picks a
3
3
  * key from the pool for every request, and transparently retries on 429/401/403
4
- * with the next key (marking a cooldown on the failed key).
4
+ * with the next key (marking a cooldown on the failed key). OAuth-backed keys
5
+ * (Cline accounts) refresh their access token pre-request and on 401, and a
6
+ * Cline daily-quota 429 cools the key down until the server-reported reset
7
+ * time instead of triggering normal rotation.
5
8
  *
6
9
  * Events are only relayed to the caller once the attempt is known to be healthy,
7
10
  * so a rotated attempt never produces duplicate/partial output.
@@ -19,9 +22,29 @@ import {
19
22
  type SimpleStreamOptions,
20
23
  } from "@earendil-works/pi-ai";
21
24
  import type { KeyOutcome, KeyPool, Lease } from "./pool.ts";
25
+ import { endpointIdentityHeaders, isClineEndpoint } from "./config.ts";
26
+ import { ensureClineAccessToken, formatClineAccessToken } from "./cline-auth.ts";
22
27
 
23
28
  const RATE_LIMIT_RE = /\b429\b|rate\s*limit|too many requests|quota\s*(exceed|limit)|requests per minute|requests per day/i;
24
29
  const INVALID_KEY_RE = /\b40[13]\b|unauthorized|forbidden|invalid\s*(api\s*)?key|incorrect\s*(api\s*)?key|authentication/i;
30
+ /**
31
+ * Cline free models enforce a daily per-account quota and answer with
32
+ * `"Daily free limit reached on model X. Try again in 23h 59m"`. This is NOT
33
+ * a rotation-worthy 429 — the cooldown is the server-reported reset time.
34
+ */
35
+ const CLINE_QUOTA_RE = /free limit reached on model/i;
36
+ const CLINE_RESET_RE = /try again in\s*(?:(\d+)\s*h)?\s*(?:(\d+)\s*m)?\s*(?:(\d+)\s*s)?/i;
37
+
38
+ /** Parse the "Try again in Xh Ym Zs" countdown from a Cline quota 429 body. */
39
+ export function parseClineQuotaResetMs(message: string): number | undefined {
40
+ const match = CLINE_RESET_RE.exec(message.toLowerCase());
41
+ if (!match) return undefined;
42
+ const hours = Number(match[1] ?? 0);
43
+ const minutes = Number(match[2] ?? 0);
44
+ const seconds = Number(match[3] ?? 0);
45
+ const ms = hours * 3_600_000 + minutes * 60_000 + seconds * 1000;
46
+ return ms > 0 ? ms : undefined;
47
+ }
25
48
 
26
49
  interface CapturedResponse {
27
50
  status: number;
@@ -30,7 +53,7 @@ interface CapturedResponse {
30
53
 
31
54
  export type Notifier = (message: string) => void;
32
55
 
33
- export function createRotatingStreamSimple(pool: KeyPool, apiName: string, notify: Notifier) {
56
+ export function createRotatingStreamSimple(pool: KeyPool, apiName: string, notify: Notifier, onConfigDirty?: () => void) {
34
57
  const impl = getApiProvider(apiName as Api);
35
58
  if (!impl) throw new Error(`multikey: no API provider registered for api: ${apiName}`);
36
59
  // Auth style: "api-key" providers want the key in x-api-key (some reject
@@ -50,19 +73,55 @@ export function createRotatingStreamSimple(pool: KeyPool, apiName: string, notif
50
73
 
51
74
  for (let attempt = 0; attempt < maxAttempts; attempt++) {
52
75
  let lease: Lease;
76
+ // One forced-token-refresh retry per acquired lease (401 → refresh → retry).
77
+ let refreshedForLease = false;
53
78
  try {
54
79
  lease = await pool.acquire(options?.signal);
55
80
  } catch {
56
81
  emitAborted(out, model);
57
82
  return;
58
83
  }
84
+ // OAuth-backed keys (Cline accounts) don't carry a static key: mint a
85
+ // fresh access token from the stored refresh token before every request.
86
+ let apiKey = lease.key;
87
+ if (lease.credential?.kind === "cline-oauth") {
88
+ try {
89
+ const fresh = await ensureClineAccessToken(lease.credential);
90
+ if (fresh.refreshed) {
91
+ pool.applyClineCredential(lease, fresh);
92
+ onConfigDirty?.();
93
+ }
94
+ // Cline's API requires the "workos:" prefix; formatClineAccessToken
95
+ // is idempotent, so this also repairs keys stored before the fix.
96
+ apiKey = formatClineAccessToken(fresh.accessToken);
97
+ } catch (error) {
98
+ // The stale token may still work; if not, the 401 path below
99
+ // force-refreshes once before giving up on this key.
100
+ notify(
101
+ `multikey[${pool.config.id}]: token refresh failed on ${lease.label} (${error instanceof Error ? error.message : String(error)}) — trying stored token`,
102
+ );
103
+ }
104
+ } else if (isClineEndpoint(pool.config.baseUrl)) {
105
+ // Static (pasted) key on the Cline endpoint: same prefix rule applies.
106
+ apiKey = formatClineAccessToken(apiKey);
107
+ }
59
108
 
60
109
  try {
61
110
  const captured: CapturedResponse = { status: 0 };
111
+ // Identity headers (session / request) are per-request, so they are
112
+ // merged here rather than baked into the provider registration.
113
+ // Providers apply options.headers last, so these win over the static
114
+ // pool headers. The message list keys the request id to the current
115
+ // turn, so retries of one turn share it like opencode's lastUser does.
116
+ const identityBaseUrl = model.baseUrl || pool.config.baseUrl;
62
117
  const attemptOptions: SimpleStreamOptions = {
63
118
  ...options,
64
- apiKey: lease.key,
65
- headers: authStyle === "api-key" ? { ...options?.headers, "x-api-key": lease.key } : options?.headers,
119
+ apiKey,
120
+ headers: {
121
+ ...options?.headers,
122
+ ...endpointIdentityHeaders(identityBaseUrl, context.messages),
123
+ ...(authStyle === "api-key" ? { "x-api-key": apiKey } : {}),
124
+ },
66
125
  onResponse: (response) => {
67
126
  captured.status = response.status;
68
127
  const ra = response.headers?.["retry-after"];
@@ -82,10 +141,31 @@ export function createRotatingStreamSimple(pool: KeyPool, apiName: string, notif
82
141
  // Attempt failed. Only rotate when nothing was relayed yet; if content
83
142
  // already streamed, the failure is surfaced as-is (mid-stream 429 is rare).
84
143
  if (verdict.kind === "rotate" && !verdict.relayedAny) {
144
+ // 401 on an OAuth key usually means an expired access token: force a
145
+ // refresh and retry the same account once, without burning a rotation.
146
+ if (
147
+ verdict.outcome === "invalid" &&
148
+ lease.credential?.kind === "cline-oauth" &&
149
+ !refreshedForLease
150
+ ) {
151
+ refreshedForLease = true;
152
+ try {
153
+ const fresh = await ensureClineAccessToken(lease.credential, { force: true });
154
+ pool.applyClineCredential(lease, fresh);
155
+ onConfigDirty?.();
156
+ notify(`multikey[${pool.config.id}]: 401 on ${lease.label} — Cline token refreshed, retrying`);
157
+ attempt--; // retry the same key; the request never got going
158
+ continue;
159
+ } catch (error) {
160
+ notify(
161
+ `multikey[${pool.config.id}]: Cline token refresh failed on ${lease.label} (${error instanceof Error ? error.message : String(error)})`,
162
+ );
163
+ }
164
+ }
85
165
  lastProblem = verdict.problem;
86
- pool.report(lease, verdict.outcome, verdict.outcome === "rate_limited" ? verdict.cooldownMs : undefined);
166
+ pool.report(lease, verdict.outcome, verdict.cooldownMs);
87
167
  notify(
88
- `multikey[${pool.config.id}]: ${verdict.outcome === "rate_limited" ? "429" : "auth error"} on ${lease.label} (${pool.mask(lease.key)}), rotating to another key`,
168
+ `multikey[${pool.config.id}]: ${describeOutcome(verdict.outcome)} on ${lease.label} (${pool.mask(lease.key)}), rotating to another key`,
89
169
  );
90
170
  continue;
91
171
  }
@@ -198,6 +278,11 @@ function classify(
198
278
  message: string,
199
279
  captured: CapturedResponse,
200
280
  ): { outcome: KeyOutcome; cooldownMs?: number } | undefined {
281
+ // Quota marker wins over the generic 429/status rules: the reset countdown
282
+ // in the body is the real cooldown, not the retry-after header.
283
+ if (CLINE_QUOTA_RE.test(message)) {
284
+ return { outcome: "quota_exhausted", cooldownMs: parseClineQuotaResetMs(message) };
285
+ }
201
286
  const fromStatus = classifyStatus(captured);
202
287
  if (fromStatus) return fromStatus;
203
288
  if (RATE_LIMIT_RE.test(message)) return { outcome: "rate_limited" };
@@ -205,6 +290,20 @@ function classify(
205
290
  return undefined;
206
291
  }
207
292
 
293
+ /** Short human label for a rotation notify line. */
294
+ function describeOutcome(outcome: KeyOutcome): string {
295
+ switch (outcome) {
296
+ case "rate_limited":
297
+ return "429";
298
+ case "quota_exhausted":
299
+ return "daily free limit reached";
300
+ case "invalid":
301
+ return "auth error";
302
+ default:
303
+ return "error";
304
+ }
305
+ }
306
+
208
307
  function baseMessage(model: Model<Api>, stopReason: "error" | "aborted", errorMessage?: string): AssistantMessage {
209
308
  return {
210
309
  role: "assistant",
package/tui.ts CHANGED
@@ -161,6 +161,69 @@ export async function showInfo(ctx: CommandContext, title: string, lines: string
161
161
  });
162
162
  }
163
163
 
164
+ export interface InfoPanelHandle {
165
+ /** Resolves when the panel closes (user enter/esc, or close()). */
166
+ closed: Promise<void>;
167
+ /** Close the panel programmatically; no-op if the user already dismissed it. */
168
+ close(): void;
169
+ }
170
+
171
+ /**
172
+ * Non-blocking variant of {@link showInfo}: returns a handle immediately so the
173
+ * caller can auto-close the panel when something finishes (e.g. an OAuth device
174
+ * flow detects browser approval) while the user can still dismiss it manually.
175
+ * Dismissing the panel never cancels whatever is running behind it.
176
+ */
177
+ export function showInfoWithHandle(ctx: CommandContext, title: string, lines: string[]): InfoPanelHandle {
178
+ let resolveClosed: (() => void) | undefined;
179
+ const closed = new Promise<void>((resolve) => {
180
+ resolveClosed = resolve;
181
+ });
182
+ let doneFn: (() => void) | undefined;
183
+ let settled = false;
184
+ const finish = () => {
185
+ if (settled) return;
186
+ settled = true;
187
+ try {
188
+ doneFn?.();
189
+ } catch {
190
+ // Panel may already be gone; the closed promise is what matters.
191
+ }
192
+ resolveClosed?.();
193
+ };
194
+
195
+ void ctx.ui.custom<void>((tui, theme, _kb, done) => {
196
+ doneFn = done;
197
+ let cachedLines: string[] | undefined;
198
+ return {
199
+ render(width: number) {
200
+ if (cachedLines) return cachedLines;
201
+ const w = Math.max(10, width);
202
+ const out: string[] = [];
203
+ const add = (line = "") => out.push(truncateToWidth(line, w));
204
+ const border = theme.fg("accent", "─".repeat(w));
205
+ add(border);
206
+ add(` ${theme.fg("accent", theme.bold(title))}`);
207
+ add();
208
+ for (const line of lines) add(` ${line}`);
209
+ add();
210
+ add(theme.fg("dim", " esc/enter dismiss (keeps running)"));
211
+ add(border);
212
+ cachedLines = out;
213
+ return out;
214
+ },
215
+ invalidate() {
216
+ cachedLines = undefined;
217
+ },
218
+ handleInput(data: string) {
219
+ if (matchesKey(data, Key.escape) || matchesKey(data, Key.enter)) finish();
220
+ },
221
+ };
222
+ });
223
+
224
+ return { closed, close: finish };
225
+ }
226
+
164
227
  /** Simple toggle list (multi-select), returns selected values or null on cancel. */
165
228
  export async function pickMany(
166
229
  ctx: CommandContext,