@selesai/code 0.13.2 → 0.13.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,15 @@
2
2
 
3
3
  All notable changes to `@selesai/code` will be documented in this file.
4
4
 
5
+ ## [0.13.3] - 2026-08-31
6
+
7
+ ### Added
8
+ - **Token-In multi-key auto-failover.** The `tokenin-onboarding` extension now registers a `tokenin` provider whose `streamSimple` automatically rotates to an alternative saved account when the active key fails with a 401, 429, or budget/quota error. Failed keys go on a 5-minute in-memory cooldown before being retried, and the newly active credential is persisted to both `tokenin-auth.json` and `auth.json`.
9
+
10
+ ### Changed
11
+ - **Cost reconciliation replaces the finalized assistant cost.** The `cost-reconcile` fetch wrapper now finishes capturing the provider-reported cost before the provider SDK finalizes its assistant message, so `message_end` replaces `usage.cost.total` with the billed amount before the message is persisted. Providers whose payloads carry no cost keep their rate-card estimate.
12
+ - **Default model catalog reasoning maps.** Added `thinkingLevelMap` entries (and `low`=low / `max` mappings) across bundled model defaults and enabled reasoning on Qwen3.8 27B and Gemini 3.7 Flash so thinking budgets map correctly.
13
+
5
14
  ## [0.13.2] - 2026-08-30
6
15
 
7
16
  ### Added
@@ -19,7 +19,7 @@
19
19
  "maxTokens": 64000,
20
20
  "thinkingLevelMap": {
21
21
  "minimal": null,
22
- "low": null,
22
+ "low": "low",
23
23
  "medium": null,
24
24
  "high": "high",
25
25
  "xhigh": null,
@@ -38,7 +38,7 @@
38
38
  "maxTokens": 64000,
39
39
  "thinkingLevelMap": {
40
40
  "minimal": null,
41
- "low": null,
41
+ "low": "low",
42
42
  "medium": null,
43
43
  "high": "high",
44
44
  "xhigh": null,
@@ -57,7 +57,7 @@
57
57
  "maxTokens": 64000,
58
58
  "thinkingLevelMap": {
59
59
  "minimal": null,
60
- "low": null,
60
+ "low": "low",
61
61
  "medium": null,
62
62
  "high": "high",
63
63
  "xhigh": null,
@@ -114,7 +114,7 @@
114
114
  "maxTokens": 64000,
115
115
  "thinkingLevelMap": {
116
116
  "minimal": null,
117
- "low": null,
117
+ "low": "low",
118
118
  "medium": null,
119
119
  "high": "high",
120
120
  "xhigh": null,
@@ -130,7 +130,15 @@
130
130
  "reasoning": true,
131
131
  "input": ["text", "image"],
132
132
  "contextWindow": 512000,
133
- "maxTokens": 128000
133
+ "maxTokens": 128000,
134
+ "thinkingLevelMap": {
135
+ "minimal": null,
136
+ "low": "low",
137
+ "medium": null,
138
+ "high": "high",
139
+ "xhigh": null,
140
+ "max": "max"
141
+ }
134
142
  },
135
143
  {
136
144
  "id": "gpt-5.6-terra",
@@ -138,7 +146,15 @@
138
146
  "reasoning": true,
139
147
  "input": ["text", "image"],
140
148
  "contextWindow": 512000,
141
- "maxTokens": 128000
149
+ "maxTokens": 128000,
150
+ "thinkingLevelMap": {
151
+ "minimal": null,
152
+ "low": "low",
153
+ "medium": null,
154
+ "high": "high",
155
+ "xhigh": null,
156
+ "max": null
157
+ }
142
158
  },
143
159
  {
144
160
  "id": "gpt-5.6-sol",
@@ -146,7 +162,15 @@
146
162
  "reasoning": true,
147
163
  "input": ["text", "image"],
148
164
  "contextWindow": 512000,
149
- "maxTokens": 128000
165
+ "maxTokens": 128000,
166
+ "thinkingLevelMap": {
167
+ "minimal": null,
168
+ "low": "low",
169
+ "medium": null,
170
+ "high": "high",
171
+ "xhigh": null,
172
+ "max": null
173
+ }
150
174
  },
151
175
  {
152
176
  "id": "kimi-k3",
@@ -209,16 +233,24 @@
209
233
  "medium": null,
210
234
  "high": "high",
211
235
  "xhigh": null,
212
- "max": null
236
+ "max": "max"
213
237
  }
214
238
  },
215
239
  {
216
240
  "id": "qwen3.8-27b",
217
241
  "name": "Qwen3.8 27B",
218
- "reasoning": false,
242
+ "reasoning": true,
219
243
  "input": ["text", "image"],
220
244
  "contextWindow": 256000,
221
245
  "maxTokens": 64000,
246
+ "thinkingLevelMap": {
247
+ "minimal": null,
248
+ "low": "low",
249
+ "medium": null,
250
+ "high": "high",
251
+ "xhigh": null,
252
+ "max": "max"
253
+ },
222
254
  "compat": {
223
255
  "thinkingFormat": "qwen"
224
256
  }
@@ -226,10 +258,18 @@
226
258
  {
227
259
  "id": "gemini-3.7-flash",
228
260
  "name": "Gemini 3.7 Flash",
229
- "reasoning": false,
230
- "input": ["text", "image" ],
261
+ "reasoning": true,
262
+ "input": ["text", "image"],
231
263
  "contextWindow": 512000,
232
- "maxTokens": 64000
264
+ "maxTokens": 64000,
265
+ "thinkingLevelMap": {
266
+ "minimal": null,
267
+ "low": "low",
268
+ "medium": null,
269
+ "high": "high",
270
+ "xhigh": null,
271
+ "max": null
272
+ }
233
273
  },
234
274
  {
235
275
  "id": "Qwen3-VL",
@@ -238,6 +278,14 @@
238
278
  "input": ["text", "image"],
239
279
  "contextWindow": 256000,
240
280
  "maxTokens": 8192,
281
+ "thinkingLevelMap": {
282
+ "minimal": null,
283
+ "low": "low",
284
+ "medium": null,
285
+ "high": "high",
286
+ "xhigh": null,
287
+ "max": null
288
+ },
241
289
  "compat": {
242
290
  "supportsDeveloperRole": false,
243
291
  "supportsReasoningEffort": false,
@@ -251,6 +299,14 @@
251
299
  "input": ["text", "image"],
252
300
  "contextWindow": 256000,
253
301
  "maxTokens": 8192,
302
+ "thinkingLevelMap": {
303
+ "minimal": null,
304
+ "low": "low",
305
+ "medium": null,
306
+ "high": "high",
307
+ "xhigh": null,
308
+ "max": null
309
+ },
254
310
  "compat": {
255
311
  "supportsDeveloperRole": false,
256
312
  "supportsReasoningEffort": false,
@@ -1,5 +1,63 @@
1
- import { describe, expect, it } from "vitest";
2
- import { extractCosts, parseCostHeader } from "./cost-reconcile.ts";
1
+ import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
2
+ import costReconcileExtension, { extractCosts, parseCostHeader } from "./cost-reconcile.ts";
3
+
4
+ type Handler = (event: any, ctx: any) => any;
5
+
6
+ const FETCH_PATCHED = Symbol.for("selesai.cost-reconcile.fetch-patched");
7
+ let originalFetch: typeof globalThis.fetch;
8
+
9
+ function createHarness() {
10
+ const handlers = new Map<string, Handler>();
11
+ const appendEntry = vi.fn();
12
+ const pi = {
13
+ on: vi.fn((event: string, handler: Handler) => handlers.set(event, handler)),
14
+ appendEntry,
15
+ };
16
+ costReconcileExtension(pi as any);
17
+ return { appendEntry, handlers };
18
+ }
19
+
20
+ async function startSession(handlers: Map<string, Handler>): Promise<void> {
21
+ await handlers.get("session_start")!({}, { sessionManager: { getEntries: () => [] } });
22
+ }
23
+
24
+ function assistantMessage(responseId: string, total = 0.75) {
25
+ return {
26
+ role: "assistant",
27
+ provider: "openrouter",
28
+ model: "openai/gpt-4.1-mini",
29
+ responseModel: "openai/gpt-4.1-mini",
30
+ responseId,
31
+ stopReason: "stop",
32
+ content: [],
33
+ usage: {
34
+ input: 10,
35
+ output: 5,
36
+ cacheRead: 2,
37
+ cacheWrite: 1,
38
+ cost: { input: 0.2, output: 0.4, cacheRead: 0.1, cacheWrite: 0.05, total },
39
+ },
40
+ };
41
+ }
42
+
43
+ function installFetch(body: string, headers: Record<string, string> = {}) {
44
+ const upstreamFetch = vi.fn(async () =>
45
+ new Response(body, { headers: { "content-type": "text/event-stream", ...headers } }),
46
+ );
47
+ globalThis.fetch = upstreamFetch as typeof globalThis.fetch;
48
+ return upstreamFetch;
49
+ }
50
+
51
+ beforeEach(() => {
52
+ originalFetch = globalThis.fetch;
53
+ delete (globalThis as typeof globalThis & { [FETCH_PATCHED]?: boolean })[FETCH_PATCHED];
54
+ });
55
+
56
+ afterEach(() => {
57
+ globalThis.fetch = originalFetch;
58
+ delete (globalThis as typeof globalThis & { [FETCH_PATCHED]?: boolean })[FETCH_PATCHED];
59
+ vi.restoreAllMocks();
60
+ });
3
61
 
4
62
  describe("parseCostHeader", () => {
5
63
  it("parses LiteLLM scientific-notation cost", () => {
@@ -10,11 +68,17 @@ describe("parseCostHeader", () => {
10
68
  expect(parseCostHeader("0.00123")).toBe(0.00123);
11
69
  });
12
70
 
13
- it("rejects null, garbage, and negatives", () => {
71
+ it("rejects null, blank, garbage, and negative values", () => {
14
72
  expect(parseCostHeader(null)).toBeUndefined();
73
+ expect(parseCostHeader("")).toBeUndefined();
74
+ expect(parseCostHeader(" \t ")).toBeUndefined();
15
75
  expect(parseCostHeader("abc")).toBeUndefined();
16
76
  expect(parseCostHeader("-1")).toBeUndefined();
17
77
  });
78
+
79
+ it("trims valid zero-valued headers", () => {
80
+ expect(parseCostHeader(" 0 ")).toBe(0);
81
+ });
18
82
  });
19
83
 
20
84
  describe("extractCosts", () => {
@@ -73,4 +137,117 @@ describe("extractCosts", () => {
73
137
  const { ids } = extractCosts(body);
74
138
  expect(ids.length).toBe(50);
75
139
  });
76
- });
140
+ });
141
+
142
+ describe("live assistant usage reconciliation", () => {
143
+ it("replaces finalized assistant usage only after the streamed body completes and keeps the custom entry", async () => {
144
+ const responseId = "live-reconcile-1";
145
+ const body = `data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\ndata: [DONE]\n\n`;
146
+ installFetch(body);
147
+ const { appendEntry, handlers } = createHarness();
148
+ await startSession(handlers);
149
+
150
+ const response = await globalThis.fetch("https://openrouter.ai/api/v1/chat/completions");
151
+ const messageEnd = handlers.get("message_end")!;
152
+ expect(await messageEnd({ message: assistantMessage(responseId) }, {})).toBeUndefined();
153
+
154
+ expect(await response.text()).toBe(body);
155
+ const result = await messageEnd({ message: assistantMessage(responseId) }, {});
156
+ expect(result.message.usage.cost).toEqual({
157
+ input: 0.2,
158
+ output: 0.4,
159
+ cacheRead: 0.1,
160
+ cacheWrite: 0.05,
161
+ total: 0.0123,
162
+ });
163
+ expect(appendEntry).toHaveBeenCalledWith(
164
+ "cost-reconcile",
165
+ expect.objectContaining({ provider: "openrouter", responseId, cost: 0.0123, source: "payload" }),
166
+ );
167
+ });
168
+
169
+ it("uses a valid zero-valued billed header over the payload cost", async () => {
170
+ const responseId = "live-header-zero-2";
171
+ installFetch(`data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\n`, {
172
+ "x-litellm-response-cost": "0",
173
+ });
174
+ const { handlers } = createHarness();
175
+ await startSession(handlers);
176
+
177
+ const response = await globalThis.fetch("https://gateway.example/v1/chat/completions");
178
+ await response.text();
179
+ const result = await handlers.get("message_end")!({ message: assistantMessage(responseId) }, {});
180
+ expect(result.message.usage.cost.total).toBe(0);
181
+ });
182
+
183
+ it("falls back when a captured body has multiple candidate response ids", async () => {
184
+ const responseId = "live-ambiguous-response-3";
185
+ installFetch(
186
+ `data: {"id":"${responseId}","usage":{"cost":0.0123},"tool_calls":[{"id":"tool-ambiguous-3"}]}\n\n`,
187
+ );
188
+ const { appendEntry, handlers } = createHarness();
189
+ await startSession(handlers);
190
+
191
+ const response = await globalThis.fetch("https://gateway.example/v1/chat/completions");
192
+ await response.text();
193
+ const message = assistantMessage(responseId);
194
+ expect(await handlers.get("message_end")!({ message }, {})).toBeUndefined();
195
+ expect(message.usage.cost.total).toBe(0.75);
196
+ expect(appendEntry).not.toHaveBeenCalled();
197
+ });
198
+
199
+ it("falls back when the process cache sees a duplicate response id", async () => {
200
+ const responseId = "live-duplicate-response-4";
201
+ installFetch(`data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\n`);
202
+ const { appendEntry, handlers } = createHarness();
203
+ await startSession(handlers);
204
+
205
+ const first = await globalThis.fetch("https://gateway.example/v1/chat/completions");
206
+ const second = await globalThis.fetch("https://gateway.example/v1/chat/completions");
207
+ await Promise.all([first.text(), second.text()]);
208
+ const message = assistantMessage(responseId);
209
+ expect(await handlers.get("message_end")!({ message }, {})).toBeUndefined();
210
+ expect(message.usage.cost.total).toBe(0.75);
211
+ expect(appendEntry).not.toHaveBeenCalled();
212
+ });
213
+
214
+ it("preserves the original message when no valid billed cost was captured", async () => {
215
+ const responseId = "live-no-cost-3";
216
+ installFetch(`data: {"id":"${responseId}","usage":{"cost":-1}}\n\n`);
217
+ const { appendEntry, handlers } = createHarness();
218
+ await startSession(handlers);
219
+
220
+ const response = await globalThis.fetch("https://gateway.example/v1/chat/completions");
221
+ await response.text();
222
+ const message = assistantMessage(responseId);
223
+ expect(await handlers.get("message_end")!({ message }, {})).toBeUndefined();
224
+ expect(message.usage.cost.total).toBe(0.75);
225
+ expect(appendEntry).not.toHaveBeenCalled();
226
+ });
227
+
228
+ it("preserves terminal error messages even when a billed cost was captured", async () => {
229
+ const responseId = "live-terminal-4";
230
+ installFetch(`data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\n`);
231
+ const { appendEntry, handlers } = createHarness();
232
+ await startSession(handlers);
233
+
234
+ const response = await globalThis.fetch("https://gateway.example/v1/chat/completions");
235
+ await response.text();
236
+ const message = { ...assistantMessage(responseId), stopReason: "error" };
237
+ expect(await handlers.get("message_end")!({ message }, {})).toBeUndefined();
238
+ expect(message.usage.cost.total).toBe(0.75);
239
+ expect(appendEntry).not.toHaveBeenCalled();
240
+ });
241
+
242
+ it("does not capture unrelated fetches", async () => {
243
+ const responseId = "unrelated-fetch-5";
244
+ const body = `data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\n`;
245
+ installFetch(body);
246
+ const { handlers } = createHarness();
247
+ await startSession(handlers);
248
+
249
+ const response = await globalThis.fetch("https://example.com/data.json");
250
+ expect(await response.text()).toBe(body);
251
+ expect(await handlers.get("message_end")!({ message: assistantMessage(responseId) }, {})).toBeUndefined();
252
+ });
253
+ });
@@ -9,19 +9,16 @@
9
9
  *
10
10
  * Extensions cannot hook the response body through the event API
11
11
  * (`after_provider_response` exposes only status/headers), but provider SDKs
12
- * fall back to `globalThis.fetch`. This extension installs a tee-wrapping
13
- * fetch once per process: LLM API responses are scanned for a
14
- * provider-reported cost (LiteLLM `x-litellm-response-cost` header,
15
- * OpenRouter `usage.cost`, `total_cost`, plus object-shaped `cost.total`)
16
- * and its response id, then recorded as a custom session entry on
17
- * `message_end`.
12
+ * fall back to `globalThis.fetch`. This extension installs a stream-wrapping
13
+ * fetch once per process: LLM API responses are scanned for a provider-reported
14
+ * cost (LiteLLM `x-litellm-response-cost` header, OpenRouter `usage.cost`,
15
+ * `total_cost`, plus object-shaped `cost.total`) and response id. The wrapper
16
+ * finishes capture before the provider SDK finalizes its assistant message, so
17
+ * `message_end` can replace `usage.cost.total` before session persistence.
18
18
  *
19
- * Providers whose payloads carry no cost (first-party OpenAI/Anthropic due to
20
- * their usage APIs) simply record nothing — their rate-card estimate matches
21
- * their bill anyway when catalog prices are configured.
22
- *
23
- * The zentui footer prefers reconciled entries over the rate-card total when
24
- * both exist for the same response id.
19
+ * Providers whose payloads carry no cost retain their rate-card estimate.
20
+ * Reconciled entries remain persisted for zentui and existing-session
21
+ * compatibility.
25
22
  */
26
23
 
27
24
  import type { ExtensionAPI } from "@selesai/code";
@@ -31,11 +28,10 @@ const ENTRY_VERSION = 1 as const;
31
28
 
32
29
  // Guard against pathological bodies; real usage chunks arrive well under 1MB.
33
30
  const MAX_CAPTURE_BYTES = 4 * 1024 * 1024;
34
- // Only LLM API paths are tee'd and scanned; unrelated fetches (web fetch tool,
35
- // model catalog refreshes) pass through untouched.
31
+ // Only LLM API paths are scanned; unrelated fetches (web fetch tool, model
32
+ // catalog refreshes) pass through untouched.
36
33
  const LLM_PATH_RE = /(\/chat\/completions|\/responses|\/messages\b|:streamGenerateContent)/;
37
34
  const MAX_CAPTURED_IDS = 50;
38
- const MAX_PENDING_LOOKUPS = 200;
39
35
  const MAX_SESSION_ENTRIES = 2_000;
40
36
 
41
37
  type ReconcileEntry = {
@@ -49,46 +45,34 @@ type ReconcileEntry = {
49
45
  source: "payload";
50
46
  };
51
47
 
52
- /**
53
- * Scan a captured SSE/JSON body for provider-reported cost and response ids.
54
- * Exported for tests.
55
- *
56
- * In streams, usage lands in the final chunk, so the LAST cost match wins.
57
- */
58
48
  type MessageInfo = {
59
49
  provider: string;
60
50
  model: string;
61
51
  responseId: string;
62
52
  };
63
53
 
64
- // Process-wide capture state. The fetch patch is installed once per process
65
- // and must not capture session-bound `pi`; per-session message_end handlers
66
- // read these maps and write entries through their own session's appendEntry.
67
- const processCaptured = new Map<string, number>();
68
- const processWaiting = new Map<
69
- string,
70
- { info: MessageInfo; write: (info: MessageInfo, cost: number) => void }
71
- >();
54
+ // Process-wide capture state. The fetch patch is installed once per process;
55
+ // per-session message_end handlers consume the matching response id. A duplicate
56
+ // capture for the same id is marked ambiguous rather than assigning either bill.
57
+ const processCaptured = new Map<string, number | null>();
72
58
 
73
- /**
74
- * Scan a captured SSE/JSON body for provider-reported cost and response ids.
75
- * Pure function; exported for tests.
76
- *
77
- * In streams, usage lands in the final chunk, so the LAST cost match wins.
78
- */
79
59
  /**
80
60
  * Parse a LiteLLM `x-litellm-response-cost` header value (a USD float,
81
61
  * possibly in scientific notation). Exported for tests.
82
62
  */
83
63
  export function parseCostHeader(value: string | null): number | undefined {
84
- if (value === null) return undefined;
85
- const cost = Number(value);
64
+ const trimmed = value?.trim();
65
+ if (!trimmed) return undefined;
66
+ const cost = Number(trimmed);
86
67
  return Number.isFinite(cost) && cost >= 0 ? cost : undefined;
87
68
  }
88
69
 
89
- export function extractCosts(
90
- text: string,
91
- ): { ids: string[]; cost?: number } {
70
+ /**
71
+ * Scan a captured SSE/JSON body for provider-reported cost and response ids.
72
+ * Exported for tests. In streams, usage lands in the final chunk, so the LAST
73
+ * valid cost match wins.
74
+ */
75
+ export function extractCosts(text: string): { ids: string[]; cost?: number } {
92
76
  const ids = new Set<string>();
93
77
  for (const match of text.matchAll(/"id"\s*:\s*"([^"]{1,200})"/g)) {
94
78
  ids.add(match[1]);
@@ -151,26 +135,22 @@ export default function costReconcileExtension(pi: ExtensionAPI): void {
151
135
  try {
152
136
  pi.appendEntry(ENTRY_TYPE, entry);
153
137
  } catch {
154
- // transcript persistence failure must not break the session
138
+ // Transcript persistence failure must not break the session.
155
139
  }
156
140
  };
157
141
 
158
142
  const finishCapture = (rawBody: string, headerCost?: number): void => {
159
- // ponytail: one naive regex pass over the whole body; a streaming
160
- // parser only matters if bodies grow past the 4MB cap.
143
+ // Ponytail: one naive regex pass over the whole body; a streaming parser
144
+ // only matters if bodies grow past the 4MB cap.
161
145
  const { ids, cost } = extractCosts(rawBody);
162
- // LiteLLM's header is the amount the gateway billed; prefer it over
163
- // any cost the upstream payload reports.
146
+ // LiteLLM's header is the amount the gateway billed; prefer it over any
147
+ // cost the upstream payload reports.
164
148
  const effectiveCost = headerCost ?? cost;
165
- if (effectiveCost === undefined) return;
166
- for (const responseId of ids) {
167
- processCaptured.set(responseId, effectiveCost);
168
- const waiting = processWaiting.get(responseId);
169
- if (waiting) {
170
- processWaiting.delete(responseId);
171
- waiting.write(waiting.info, effectiveCost);
172
- }
173
- }
149
+ // A cost must identify exactly one response. Applying it to every `id` in
150
+ // a body can mistake nested tool/message ids for the assistant response.
151
+ if (effectiveCost === undefined || ids.length !== 1) return;
152
+ const responseId = ids[0]!;
153
+ processCaptured.set(responseId, processCaptured.has(responseId) ? null : effectiveCost);
174
154
  if (processCaptured.size > 500) {
175
155
  for (const key of [...processCaptured.keys()].slice(0, processCaptured.size - 500))
176
156
  processCaptured.delete(key);
@@ -191,7 +171,7 @@ export default function costReconcileExtension(pi: ExtensionAPI): void {
191
171
  if (typeof responseId === "string") seen.add(responseId);
192
172
  }
193
173
  } catch {
194
- // best effort only
174
+ // Best effort only.
195
175
  }
196
176
 
197
177
  const globalFetch = globalThis as typeof globalThis & { [FETCH_PATCHED]?: boolean };
@@ -205,26 +185,36 @@ export default function costReconcileExtension(pi: ExtensionAPI): void {
205
185
  const contentType = response.headers.get("content-type") ?? "";
206
186
  if (!contentType.includes("json") && !contentType.includes("event-stream")) return response;
207
187
  if (!response.body) return response;
188
+
208
189
  const headerCost = parseCostHeader(response.headers.get("x-litellm-response-cost"));
209
- const [main, tee] = response.body.tee();
210
- void (async () => {
211
- let raw = "";
212
- try {
213
- const reader = tee.getReader();
214
- const decoder = new TextDecoder();
215
- for (;;) {
216
- const { done, value } = await reader.read();
217
- if (done) break;
218
- raw += decoder.decode(value, { stream: true });
219
- if (raw.length > MAX_CAPTURE_BYTES) break;
190
+ let raw = "";
191
+ let captureFailed = false;
192
+ const decoder = new TextDecoder();
193
+ const capture = new TransformStream<Uint8Array, Uint8Array>({
194
+ transform(chunk, controller) {
195
+ controller.enqueue(chunk);
196
+ if (captureFailed) return;
197
+ try {
198
+ raw += decoder.decode(chunk, { stream: true });
199
+ if (raw.length > MAX_CAPTURE_BYTES) captureFailed = true;
200
+ } catch {
201
+ captureFailed = true;
220
202
  }
221
- } catch {
222
- // capture is best-effort; the main branch is unaffected
223
- }
224
- finishCapture(raw, headerCost);
225
- })();
226
- // Preserve response identity as seen by the SDK: body swapped, rest identical.
227
- return new Response(main, {
203
+ },
204
+ flush() {
205
+ if (captureFailed) return;
206
+ try {
207
+ raw += decoder.decode();
208
+ finishCapture(raw, headerCost);
209
+ } catch {
210
+ // Capture is best-effort; the response stream is unaffected.
211
+ }
212
+ },
213
+ });
214
+ // flush() runs before the SDK observes EOF, so message_end can safely
215
+ // replace the finalized message usage without waiting on another fetch.
216
+ const body = response.body.pipeThrough(capture);
217
+ return new Response(body, {
228
218
  status: response.status,
229
219
  statusText: response.statusText,
230
220
  headers: response.headers,
@@ -236,43 +226,32 @@ export default function costReconcileExtension(pi: ExtensionAPI): void {
236
226
  });
237
227
 
238
228
  pi.on("message_end", (event) => {
239
- const message = event.message as
240
- | {
241
- role?: string;
242
- provider?: string;
243
- model?: string;
244
- responseId?: string;
245
- stopReason?: string;
246
- }
247
- | undefined;
248
- if (!message || message.role !== "assistant") return;
229
+ const message = event.message;
230
+ if (message.role !== "assistant") return;
249
231
  if (message.stopReason === "error" || message.stopReason === "aborted") return;
250
232
  const responseId = message.responseId;
251
233
  if (typeof responseId !== "string" || responseId.length === 0) return;
252
234
 
253
- const info: MessageInfo = {
254
- provider: message.provider ?? "",
255
- model: message.responseModel ?? message.model ?? "",
256
- responseId,
257
- };
258
235
  const cost = processCaptured.get(responseId);
259
- if (cost !== undefined) {
260
- writeEntry(info, cost);
236
+ if (cost === undefined || cost === null || !message.usage?.cost) {
237
+ processCaptured.delete(responseId);
261
238
  return;
262
239
  }
263
- // The tee branch may finish after message_end; remember the message so
264
- // the capture can match it on completion and write via this session.
265
- if (!seen.has(responseId) && !processWaiting.has(responseId)) {
266
- waitingLimitGuard();
267
- processWaiting.set(responseId, { info, write: writeEntry });
268
- }
240
+ processCaptured.delete(responseId);
241
+
242
+ writeEntry(
243
+ {
244
+ provider: message.provider ?? "",
245
+ model: message.responseModel ?? message.model ?? "",
246
+ responseId,
247
+ },
248
+ cost,
249
+ );
250
+ return {
251
+ message: {
252
+ ...message,
253
+ usage: { ...message.usage, cost: { ...message.usage.cost, total: cost } },
254
+ },
255
+ };
269
256
  });
270
257
  }
271
-
272
- // Trim oldest waiting entries so unclaimed captures cannot grow unbounded.
273
- function waitingLimitGuard(): void {
274
- if (processWaiting.size >= MAX_PENDING_LOOKUPS) {
275
- for (const key of [...processWaiting.keys()].slice(0, processWaiting.size - MAX_PENDING_LOOKUPS + 1))
276
- processWaiting.delete(key);
277
- }
278
- }
@@ -13,6 +13,16 @@ import { chmodSync, existsSync, mkdirSync, readFileSync, writeFileSync } from "n
13
13
  import { dirname, join } from "node:path";
14
14
  import { getAgentDir, getModelsPath } from "@selesai/code";
15
15
  import type { AuthStorage, ExtensionAPI, ExtensionCommandContext, ExtensionContext, SessionStartEvent } from "@selesai/code";
16
+ import {
17
+ createAssistantMessageEventStream,
18
+ lazyStream,
19
+ type AssistantMessageEvent,
20
+ type AssistantMessageEventStream,
21
+ type Context,
22
+ type Model,
23
+ type SimpleStreamOptions,
24
+ } from "@earendil-works/pi-ai";
25
+ import { getApiProvider } from "@earendil-works/pi-ai/compat";
16
26
 
17
27
  export const TOKEN_IN_PROVIDER = "tokenin";
18
28
  export const TOKEN_IN_DASHBOARD_URL = "https://token.selesai.in/dashboard/tokens";
@@ -216,6 +226,291 @@ function isOnboardingComplete(agentDir: string = getAgentDir()): boolean {
216
226
  return existsSync(getOnboardingMarkerPath(agentDir));
217
227
  }
218
228
 
229
+ // ---------------------------------------------------------------------------
230
+ // Multi-key load balancing / auto-failover
231
+ // ---------------------------------------------------------------------------
232
+
233
+ export const DEFAULT_TOKENIN_COOLDOWN_MS = 5 * 60 * 1000; // 5 minutes
234
+
235
+ /** In-memory cooldown tracker mapping token/accountId -> cooldown expires timestamp (ms). */
236
+ const tokenCooldowns = new Map<string, number>();
237
+
238
+ /** Clear all in-memory cooldowns. Exported for tests. */
239
+ export function clearTokenInCooldowns(): void {
240
+ tokenCooldowns.clear();
241
+ }
242
+
243
+ /** Check if a token/account is currently on cooldown. */
244
+ export function isTokenInOnCooldown(tokenId: string, now: number = Date.now()): boolean {
245
+ const expires = tokenCooldowns.get(tokenId);
246
+ if (expires === undefined) return false;
247
+ if (now >= expires) {
248
+ tokenCooldowns.delete(tokenId);
249
+ return false;
250
+ }
251
+ return true;
252
+ }
253
+
254
+ /** Put a token/account on cooldown for a given duration. */
255
+ export function setTokenInCooldown(tokenId: string, durationMs: number = DEFAULT_TOKENIN_COOLDOWN_MS, now: number = Date.now()): void {
256
+ tokenCooldowns.set(tokenId, now + durationMs);
257
+ }
258
+
259
+ /** Detect if an error message or AssistantMessage stop indicates an auth/quota failure eligible for failover. */
260
+ export function isRotateableTokenInError(error: unknown): boolean {
261
+ if (!error) return false;
262
+ let text = "";
263
+ if (typeof error === "string") {
264
+ text = error;
265
+ } else if (error instanceof Error) {
266
+ text = `${error.name}: ${error.message}`;
267
+ } else if (typeof error === "object" && error !== null) {
268
+ const obj = error as Record<string, unknown>;
269
+ if (typeof obj.errorMessage === "string") {
270
+ text = obj.errorMessage;
271
+ } else if (typeof obj.message === "string") {
272
+ text = obj.message;
273
+ } else if (Array.isArray(obj.content)) {
274
+ // AssistantMessage with content blocks containing error text or errorMessage
275
+ for (const block of obj.content) {
276
+ if (typeof block === "object" && block !== null) {
277
+ const b = block as Record<string, unknown>;
278
+ if (typeof b.errorMessage === "string") text += ` ${b.errorMessage}`;
279
+ if (typeof b.text === "string") text += ` ${b.text}`;
280
+ }
281
+ }
282
+ }
283
+ if (!text) text = JSON.stringify(error);
284
+ } else {
285
+ text = JSON.stringify(error);
286
+ }
287
+
288
+ const lower = text.toLowerCase();
289
+ // HTTP 401 / Invalid Key
290
+ if (lower.includes("401") || lower.includes("unauthorized") || lower.includes("invalid_api_key") || lower.includes("invalid api key") || lower.includes("authentication_error")) {
291
+ return true;
292
+ }
293
+ // HTTP 429 / Rate limit
294
+ if (lower.includes("429") || lower.includes("rate_limit") || lower.includes("rate limit") || lower.includes("too many requests")) {
295
+ return true;
296
+ }
297
+ // Budget / Quota / Credits
298
+ if (
299
+ lower.includes("budget has been exceeded") ||
300
+ lower.includes("budget_exceeded") ||
301
+ lower.includes("max_budget") ||
302
+ lower.includes("insufficient_quota") ||
303
+ lower.includes("quota exceeded") ||
304
+ lower.includes("exceeds max_budget") ||
305
+ lower.includes("credit limit") ||
306
+ lower.includes("out of credits")
307
+ ) {
308
+ return true;
309
+ }
310
+ return false;
311
+ }
312
+
313
+ /** Helper to persist active credential to auth.json and tokenin-auth.json without needing AuthStorage instance. */
314
+ function persistActiveTokenInAccount(account: TokenInAccount, authPath: string = getTokenInAuthPath()): void {
315
+ try {
316
+ // Update tokenin-auth.json
317
+ const auth = readTokenInAuth(authPath);
318
+ auth.activeId = account.id;
319
+ writeTokenInAuth(auth, authPath);
320
+
321
+ // Update auth.json
322
+ const agentDir = dirname(authPath);
323
+ const globalAuthPath = join(agentDir, "auth.json");
324
+ const current = existsSync(globalAuthPath) ? JSON.parse(readFileSync(globalAuthPath, "utf-8")) as Record<string, unknown> : {};
325
+ current[TOKEN_IN_PROVIDER] = { type: "api_key", key: account.apiKey };
326
+ mkdirSync(dirname(globalAuthPath), { recursive: true, mode: 0o700 });
327
+ writeFileSync(globalAuthPath, `${JSON.stringify(current, null, 2)}\n`, { encoding: "utf-8", mode: 0o600 });
328
+ try {
329
+ chmodSync(globalAuthPath, 0o600);
330
+ } catch {
331
+ // best-effort
332
+ }
333
+ } catch {
334
+ // best-effort sync
335
+ }
336
+ }
337
+
338
+ /**
339
+ * Custom streamSimple implementation for tokenin provider that wraps OpenAI completions
340
+ * and automatically rotates to alternative saved keys on 401/429/budget exceeded errors.
341
+ */
342
+ export function createTokenInStreamSimple(options?: {
343
+ authPath?: string;
344
+ getAuthStorage?: () => AuthStorage | undefined;
345
+ onRotate?: (failedAccount: TokenInAccount, nextAccount: TokenInAccount, reason: string) => void;
346
+ streamSimple?: (model: Model<any>, context: Context, options?: SimpleStreamOptions) => AssistantMessageEventStream;
347
+ }) {
348
+ const authPath = options?.authPath ?? getTokenInAuthPath();
349
+ const baseApi = options?.streamSimple ? undefined : getApiProvider("openai-completions");
350
+ const streamSimple = options?.streamSimple ?? baseApi!.streamSimple.bind(baseApi!);
351
+
352
+ const activate = (account: TokenInAccount): void => {
353
+ const authStorage = options?.getAuthStorage?.();
354
+ if (authStorage) {
355
+ applyTokenInAccountToAuth(account, authStorage, authPath);
356
+ } else {
357
+ persistActiveTokenInAccount(account, authPath);
358
+ }
359
+ };
360
+
361
+ const withAccountAuthorization = (streamOptions: SimpleStreamOptions | undefined, account: TokenInAccount): SimpleStreamOptions => {
362
+ const headers = Object.fromEntries(
363
+ Object.entries(streamOptions?.headers ?? {}).filter(([name]) => name.toLowerCase() !== "authorization"),
364
+ );
365
+ return {
366
+ ...streamOptions,
367
+ apiKey: account.apiKey,
368
+ headers: { ...headers, Authorization: `Bearer ${account.apiKey}` },
369
+ };
370
+ };
371
+
372
+ return function tokenInStreamSimple(
373
+ model: Model<any>,
374
+ context: Context,
375
+ streamOptions?: SimpleStreamOptions,
376
+ ): AssistantMessageEventStream {
377
+ return lazyStream(model, async () => {
378
+ const auth = readTokenInAuth(authPath);
379
+ const accounts = auth.accounts;
380
+
381
+ // If no accounts saved, fall back directly to base implementation
382
+ if (accounts.length === 0) {
383
+ return streamSimple(model, context, streamOptions);
384
+ }
385
+
386
+ // Order accounts: active account first, followed by remaining accounts
387
+ const activeIndex = accounts.findIndex((a) => a.id === auth.activeId || a.apiKey === streamOptions?.apiKey);
388
+ const orderedAccounts: TokenInAccount[] = [];
389
+ if (activeIndex >= 0) {
390
+ orderedAccounts.push(accounts[activeIndex]!);
391
+ for (let i = 0; i < accounts.length; i++) {
392
+ if (i !== activeIndex) orderedAccounts.push(accounts[i]!);
393
+ }
394
+ } else {
395
+ orderedAccounts.push(...accounts);
396
+ }
397
+
398
+ // Filter/order candidates: prioritize those not currently on cooldown
399
+ const availableAccounts = orderedAccounts.filter((a) => !isTokenInOnCooldown(a.id));
400
+ const candidateAccounts = (availableAccounts.length > 0 ? availableAccounts : orderedAccounts).map((account) => ({ ...account }));
401
+
402
+ let lastErrorEvent: AssistantMessageEvent | undefined;
403
+ let lastThrownError: unknown;
404
+
405
+ for (let i = 0; i < candidateAccounts.length; i++) {
406
+ const currentAccount = candidateAccounts[i]!;
407
+
408
+ const requestOptions = withAccountAuthorization(streamOptions, currentAccount);
409
+
410
+ let underlyingStream: AssistantMessageEventStream;
411
+ try {
412
+ underlyingStream = streamSimple(model, context, requestOptions);
413
+ } catch (err) {
414
+ lastThrownError = err;
415
+ if (isRotateableTokenInError(err) && i + 1 < candidateAccounts.length) {
416
+ setTokenInCooldown(currentAccount.id);
417
+ const nextAccount = candidateAccounts[i + 1]!;
418
+ activate(nextAccount);
419
+ options?.onRotate?.(currentAccount, nextAccount, String(err));
420
+ continue;
421
+ }
422
+ throw err;
423
+ }
424
+
425
+ // Hold a leading start event until the request proves it produced output. This
426
+ // permits an immediate upstream error to be retried without exposing two starts.
427
+ const iterator = underlyingStream[Symbol.asyncIterator]();
428
+ let firstResult: IteratorResult<AssistantMessageEvent>;
429
+ try {
430
+ firstResult = await iterator.next();
431
+ } catch (err) {
432
+ lastThrownError = err;
433
+ if (isRotateableTokenInError(err) && i + 1 < candidateAccounts.length) {
434
+ setTokenInCooldown(currentAccount.id);
435
+ const nextAccount = candidateAccounts[i + 1]!;
436
+ activate(nextAccount);
437
+ options?.onRotate?.(currentAccount, nextAccount, String(err));
438
+ continue;
439
+ }
440
+ throw err;
441
+ }
442
+
443
+ if (firstResult.done) {
444
+ const passthrough = createAssistantMessageEventStream();
445
+ passthrough.end();
446
+ return passthrough;
447
+ }
448
+
449
+ const bufferedEvents = [firstResult.value];
450
+ if (bufferedEvents[0].type === "start") {
451
+ try {
452
+ const next = await iterator.next();
453
+ if (!next.done) bufferedEvents.push(next.value);
454
+ } catch (err) {
455
+ lastThrownError = err;
456
+ if (isRotateableTokenInError(err) && i + 1 < candidateAccounts.length) {
457
+ setTokenInCooldown(currentAccount.id);
458
+ const nextAccount = candidateAccounts[i + 1]!;
459
+ activate(nextAccount);
460
+ options?.onRotate?.(currentAccount, nextAccount, String(err));
461
+ continue;
462
+ }
463
+ throw err;
464
+ }
465
+ }
466
+
467
+ const immediateError = bufferedEvents.find((event) => event.type === "error");
468
+ if (immediateError?.type === "error" && isRotateableTokenInError(immediateError.error)) {
469
+ lastErrorEvent = immediateError;
470
+ setTokenInCooldown(currentAccount.id);
471
+ if (i + 1 < candidateAccounts.length) {
472
+ const nextAccount = candidateAccounts[i + 1]!;
473
+ activate(nextAccount);
474
+ options?.onRotate?.(currentAccount, nextAccount, immediateError.error.errorMessage ?? "Error");
475
+ continue;
476
+ }
477
+ }
478
+
479
+ const outputStream = createAssistantMessageEventStream();
480
+ (async () => {
481
+ try {
482
+ for (const event of bufferedEvents) outputStream.push(event);
483
+ while (true) {
484
+ const next = await iterator.next();
485
+ if (next.done) break;
486
+ outputStream.push(next.value);
487
+ }
488
+ outputStream.end();
489
+ } catch {
490
+ outputStream.end();
491
+ }
492
+ })();
493
+
494
+ return outputStream;
495
+ }
496
+
497
+ // If loop exhausted with a rotateable error event, return a stream with that error
498
+ if (lastErrorEvent) {
499
+ const errorStream = createAssistantMessageEventStream();
500
+ errorStream.push(lastErrorEvent);
501
+ errorStream.end();
502
+ return errorStream;
503
+ }
504
+
505
+ if (lastThrownError) {
506
+ throw lastThrownError;
507
+ }
508
+
509
+ return streamSimple(model, context, streamOptions);
510
+ });
511
+ };
512
+ }
513
+
219
514
  async function openDashboard(pi: ExtensionAPI): Promise<void> {
220
515
  const url = TOKEN_IN_DASHBOARD_URL;
221
516
  let command: string;
@@ -537,7 +832,14 @@ async function runOnboarding(pi: ExtensionAPI, ctx: ExtensionContext): Promise<v
537
832
  }
538
833
 
539
834
  export default function tokenInOnboardingExtension(pi: ExtensionAPI): void {
835
+ let authStorage: AuthStorage | undefined;
836
+ pi.registerProvider("tokenin", {
837
+ api: "openai-completions",
838
+ streamSimple: createTokenInStreamSimple({ getAuthStorage: () => authStorage }),
839
+ });
840
+
540
841
  pi.on("session_start", async (event: SessionStartEvent, ctx: ExtensionContext) => {
842
+ authStorage = ctx.modelRegistry.authStorage;
541
843
  if (event.reason !== "startup") return;
542
844
  if (isOnboardingComplete()) return;
543
845
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@selesai/code",
3
- "version": "0.13.2",
3
+ "version": "0.13.3",
4
4
  "description": "Maintained, extension-first Pi coding agent with built-in workflows, subagents, web research, questions, skills, and an enhanced terminal UI.",
5
5
  "type": "module",
6
6
  "repository": {