@selesai/code 0.13.1 → 0.13.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/dist/defaults/models.json +68 -12
- package/dist/extensions/cost-reconcile.test.ts +181 -4
- package/dist/extensions/cost-reconcile.ts +83 -104
- package/dist/extensions/pi-rewind-hook/README.md +4 -1
- package/dist/extensions/pi-rewind-hook/index.ts +151 -21
- package/dist/extensions/pi-rewind-hook/package.json +5 -3
- package/dist/extensions/pi-subagents/src/integrations/herdr-status.ts +0 -4
- package/dist/extensions/pi-subagents/test/unit/herdr-status-bridge.test.ts +4 -6
- package/dist/extensions/tokenin-onboarding.ts +302 -0
- package/package.json +1 -1
- package/dist/extensions/ascii.txt +0 -9
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,25 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to `@selesai/code` will be documented in this file.
|
|
4
4
|
|
|
5
|
+
## [0.13.3] - 2026-08-31
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
- **Token-In multi-key auto-failover.** The `tokenin-onboarding` extension now registers a `tokenin` provider whose `streamSimple` automatically rotates to an alternative saved account when the active key fails with a 401, 429, or budget/quota error. Failed keys go on a 5-minute in-memory cooldown before being retried, and the newly active credential is persisted to both `tokenin-auth.json` and `auth.json`.
|
|
9
|
+
|
|
10
|
+
### Changed
|
|
11
|
+
- **Cost reconciliation replaces the finalized assistant cost.** The `cost-reconcile` fetch wrapper now finishes capturing the provider-reported cost before the provider SDK finalizes its assistant message, so `message_end` replaces `usage.cost.total` with the billed amount before the message is persisted. Providers whose payloads carry no cost keep their rate-card estimate.
|
|
12
|
+
- **Default model catalog reasoning maps.** Added `thinkingLevelMap` entries (and `low`=low / `max` mappings) across bundled model defaults and enabled reasoning on Qwen3.8 27B and Gemini 3.7 Flash so thinking budgets map correctly.
|
|
13
|
+
|
|
14
|
+
## [0.13.2] - 2026-08-30
|
|
15
|
+
|
|
16
|
+
### Added
|
|
17
|
+
- **Rewind submodule support.** `pi-rewind-hook` now snapshots gitlink commits and restores submodule worktrees to their target commits during exact rewind, with safety checks: submodule paths must stay unchanged, worktrees must be clean and initialized, and target commits must be available locally. Unsupported submodule states (dirty, uninitialized, added/removed paths, nested submodules) are refused with a clear error instead of being reported as exact restores.
|
|
18
|
+
- **Rewind retention sweep concurrency safety.** The retention sweep now uses a compare-and-swap on the store ref with retries, so concurrent sessions adding snapshots mid-sweep are preserved instead of being clobbered; the startup sweep is reused across session replacement instead of racing a second sweep.
|
|
19
|
+
|
|
20
|
+
### Changed
|
|
21
|
+
- **Herdr status bridge drops agent scoping.** `pi-subagents` no longer passes `--agent pi` / `--applies-to-source herdr:pi` to Herdr `report-metadata` calls; state labels are reported under the `pi-subagents:herdr` source directly.
|
|
22
|
+
- **Rewind extension packaging.** `pi-rewind-hook` declares `@selesai/code` as a peer dependency (was devDependency) and bumps to 1.8.6.
|
|
23
|
+
|
|
5
24
|
## [0.13.1] - 2026-08-30
|
|
6
25
|
|
|
7
26
|
### Fixed
|
|
@@ -19,7 +19,7 @@
|
|
|
19
19
|
"maxTokens": 64000,
|
|
20
20
|
"thinkingLevelMap": {
|
|
21
21
|
"minimal": null,
|
|
22
|
-
"low":
|
|
22
|
+
"low": "low",
|
|
23
23
|
"medium": null,
|
|
24
24
|
"high": "high",
|
|
25
25
|
"xhigh": null,
|
|
@@ -38,7 +38,7 @@
|
|
|
38
38
|
"maxTokens": 64000,
|
|
39
39
|
"thinkingLevelMap": {
|
|
40
40
|
"minimal": null,
|
|
41
|
-
"low":
|
|
41
|
+
"low": "low",
|
|
42
42
|
"medium": null,
|
|
43
43
|
"high": "high",
|
|
44
44
|
"xhigh": null,
|
|
@@ -57,7 +57,7 @@
|
|
|
57
57
|
"maxTokens": 64000,
|
|
58
58
|
"thinkingLevelMap": {
|
|
59
59
|
"minimal": null,
|
|
60
|
-
"low":
|
|
60
|
+
"low": "low",
|
|
61
61
|
"medium": null,
|
|
62
62
|
"high": "high",
|
|
63
63
|
"xhigh": null,
|
|
@@ -114,7 +114,7 @@
|
|
|
114
114
|
"maxTokens": 64000,
|
|
115
115
|
"thinkingLevelMap": {
|
|
116
116
|
"minimal": null,
|
|
117
|
-
"low":
|
|
117
|
+
"low": "low",
|
|
118
118
|
"medium": null,
|
|
119
119
|
"high": "high",
|
|
120
120
|
"xhigh": null,
|
|
@@ -130,7 +130,15 @@
|
|
|
130
130
|
"reasoning": true,
|
|
131
131
|
"input": ["text", "image"],
|
|
132
132
|
"contextWindow": 512000,
|
|
133
|
-
"maxTokens": 128000
|
|
133
|
+
"maxTokens": 128000,
|
|
134
|
+
"thinkingLevelMap": {
|
|
135
|
+
"minimal": null,
|
|
136
|
+
"low": "low",
|
|
137
|
+
"medium": null,
|
|
138
|
+
"high": "high",
|
|
139
|
+
"xhigh": null,
|
|
140
|
+
"max": "max"
|
|
141
|
+
}
|
|
134
142
|
},
|
|
135
143
|
{
|
|
136
144
|
"id": "gpt-5.6-terra",
|
|
@@ -138,7 +146,15 @@
|
|
|
138
146
|
"reasoning": true,
|
|
139
147
|
"input": ["text", "image"],
|
|
140
148
|
"contextWindow": 512000,
|
|
141
|
-
"maxTokens": 128000
|
|
149
|
+
"maxTokens": 128000,
|
|
150
|
+
"thinkingLevelMap": {
|
|
151
|
+
"minimal": null,
|
|
152
|
+
"low": "low",
|
|
153
|
+
"medium": null,
|
|
154
|
+
"high": "high",
|
|
155
|
+
"xhigh": null,
|
|
156
|
+
"max": null
|
|
157
|
+
}
|
|
142
158
|
},
|
|
143
159
|
{
|
|
144
160
|
"id": "gpt-5.6-sol",
|
|
@@ -146,7 +162,15 @@
|
|
|
146
162
|
"reasoning": true,
|
|
147
163
|
"input": ["text", "image"],
|
|
148
164
|
"contextWindow": 512000,
|
|
149
|
-
"maxTokens": 128000
|
|
165
|
+
"maxTokens": 128000,
|
|
166
|
+
"thinkingLevelMap": {
|
|
167
|
+
"minimal": null,
|
|
168
|
+
"low": "low",
|
|
169
|
+
"medium": null,
|
|
170
|
+
"high": "high",
|
|
171
|
+
"xhigh": null,
|
|
172
|
+
"max": null
|
|
173
|
+
}
|
|
150
174
|
},
|
|
151
175
|
{
|
|
152
176
|
"id": "kimi-k3",
|
|
@@ -209,16 +233,24 @@
|
|
|
209
233
|
"medium": null,
|
|
210
234
|
"high": "high",
|
|
211
235
|
"xhigh": null,
|
|
212
|
-
"max":
|
|
236
|
+
"max": "max"
|
|
213
237
|
}
|
|
214
238
|
},
|
|
215
239
|
{
|
|
216
240
|
"id": "qwen3.8-27b",
|
|
217
241
|
"name": "Qwen3.8 27B",
|
|
218
|
-
"reasoning":
|
|
242
|
+
"reasoning": true,
|
|
219
243
|
"input": ["text", "image"],
|
|
220
244
|
"contextWindow": 256000,
|
|
221
245
|
"maxTokens": 64000,
|
|
246
|
+
"thinkingLevelMap": {
|
|
247
|
+
"minimal": null,
|
|
248
|
+
"low": "low",
|
|
249
|
+
"medium": null,
|
|
250
|
+
"high": "high",
|
|
251
|
+
"xhigh": null,
|
|
252
|
+
"max": "max"
|
|
253
|
+
},
|
|
222
254
|
"compat": {
|
|
223
255
|
"thinkingFormat": "qwen"
|
|
224
256
|
}
|
|
@@ -226,10 +258,18 @@
|
|
|
226
258
|
{
|
|
227
259
|
"id": "gemini-3.7-flash",
|
|
228
260
|
"name": "Gemini 3.7 Flash",
|
|
229
|
-
"reasoning":
|
|
230
|
-
"input": ["text", "image"
|
|
261
|
+
"reasoning": true,
|
|
262
|
+
"input": ["text", "image"],
|
|
231
263
|
"contextWindow": 512000,
|
|
232
|
-
"maxTokens": 64000
|
|
264
|
+
"maxTokens": 64000,
|
|
265
|
+
"thinkingLevelMap": {
|
|
266
|
+
"minimal": null,
|
|
267
|
+
"low": "low",
|
|
268
|
+
"medium": null,
|
|
269
|
+
"high": "high",
|
|
270
|
+
"xhigh": null,
|
|
271
|
+
"max": null
|
|
272
|
+
}
|
|
233
273
|
},
|
|
234
274
|
{
|
|
235
275
|
"id": "Qwen3-VL",
|
|
@@ -238,6 +278,14 @@
|
|
|
238
278
|
"input": ["text", "image"],
|
|
239
279
|
"contextWindow": 256000,
|
|
240
280
|
"maxTokens": 8192,
|
|
281
|
+
"thinkingLevelMap": {
|
|
282
|
+
"minimal": null,
|
|
283
|
+
"low": "low",
|
|
284
|
+
"medium": null,
|
|
285
|
+
"high": "high",
|
|
286
|
+
"xhigh": null,
|
|
287
|
+
"max": null
|
|
288
|
+
},
|
|
241
289
|
"compat": {
|
|
242
290
|
"supportsDeveloperRole": false,
|
|
243
291
|
"supportsReasoningEffort": false,
|
|
@@ -251,6 +299,14 @@
|
|
|
251
299
|
"input": ["text", "image"],
|
|
252
300
|
"contextWindow": 256000,
|
|
253
301
|
"maxTokens": 8192,
|
|
302
|
+
"thinkingLevelMap": {
|
|
303
|
+
"minimal": null,
|
|
304
|
+
"low": "low",
|
|
305
|
+
"medium": null,
|
|
306
|
+
"high": "high",
|
|
307
|
+
"xhigh": null,
|
|
308
|
+
"max": null
|
|
309
|
+
},
|
|
254
310
|
"compat": {
|
|
255
311
|
"supportsDeveloperRole": false,
|
|
256
312
|
"supportsReasoningEffort": false,
|
|
@@ -1,5 +1,63 @@
|
|
|
1
|
-
import { describe, expect, it } from "vitest";
|
|
2
|
-
import { extractCosts, parseCostHeader } from "./cost-reconcile.ts";
|
|
1
|
+
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
|
|
2
|
+
import costReconcileExtension, { extractCosts, parseCostHeader } from "./cost-reconcile.ts";
|
|
3
|
+
|
|
4
|
+
type Handler = (event: any, ctx: any) => any;
|
|
5
|
+
|
|
6
|
+
const FETCH_PATCHED = Symbol.for("selesai.cost-reconcile.fetch-patched");
|
|
7
|
+
let originalFetch: typeof globalThis.fetch;
|
|
8
|
+
|
|
9
|
+
function createHarness() {
|
|
10
|
+
const handlers = new Map<string, Handler>();
|
|
11
|
+
const appendEntry = vi.fn();
|
|
12
|
+
const pi = {
|
|
13
|
+
on: vi.fn((event: string, handler: Handler) => handlers.set(event, handler)),
|
|
14
|
+
appendEntry,
|
|
15
|
+
};
|
|
16
|
+
costReconcileExtension(pi as any);
|
|
17
|
+
return { appendEntry, handlers };
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
async function startSession(handlers: Map<string, Handler>): Promise<void> {
|
|
21
|
+
await handlers.get("session_start")!({}, { sessionManager: { getEntries: () => [] } });
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
function assistantMessage(responseId: string, total = 0.75) {
|
|
25
|
+
return {
|
|
26
|
+
role: "assistant",
|
|
27
|
+
provider: "openrouter",
|
|
28
|
+
model: "openai/gpt-4.1-mini",
|
|
29
|
+
responseModel: "openai/gpt-4.1-mini",
|
|
30
|
+
responseId,
|
|
31
|
+
stopReason: "stop",
|
|
32
|
+
content: [],
|
|
33
|
+
usage: {
|
|
34
|
+
input: 10,
|
|
35
|
+
output: 5,
|
|
36
|
+
cacheRead: 2,
|
|
37
|
+
cacheWrite: 1,
|
|
38
|
+
cost: { input: 0.2, output: 0.4, cacheRead: 0.1, cacheWrite: 0.05, total },
|
|
39
|
+
},
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
function installFetch(body: string, headers: Record<string, string> = {}) {
|
|
44
|
+
const upstreamFetch = vi.fn(async () =>
|
|
45
|
+
new Response(body, { headers: { "content-type": "text/event-stream", ...headers } }),
|
|
46
|
+
);
|
|
47
|
+
globalThis.fetch = upstreamFetch as typeof globalThis.fetch;
|
|
48
|
+
return upstreamFetch;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
beforeEach(() => {
|
|
52
|
+
originalFetch = globalThis.fetch;
|
|
53
|
+
delete (globalThis as typeof globalThis & { [FETCH_PATCHED]?: boolean })[FETCH_PATCHED];
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
afterEach(() => {
|
|
57
|
+
globalThis.fetch = originalFetch;
|
|
58
|
+
delete (globalThis as typeof globalThis & { [FETCH_PATCHED]?: boolean })[FETCH_PATCHED];
|
|
59
|
+
vi.restoreAllMocks();
|
|
60
|
+
});
|
|
3
61
|
|
|
4
62
|
describe("parseCostHeader", () => {
|
|
5
63
|
it("parses LiteLLM scientific-notation cost", () => {
|
|
@@ -10,11 +68,17 @@ describe("parseCostHeader", () => {
|
|
|
10
68
|
expect(parseCostHeader("0.00123")).toBe(0.00123);
|
|
11
69
|
});
|
|
12
70
|
|
|
13
|
-
it("rejects null, garbage, and
|
|
71
|
+
it("rejects null, blank, garbage, and negative values", () => {
|
|
14
72
|
expect(parseCostHeader(null)).toBeUndefined();
|
|
73
|
+
expect(parseCostHeader("")).toBeUndefined();
|
|
74
|
+
expect(parseCostHeader(" \t ")).toBeUndefined();
|
|
15
75
|
expect(parseCostHeader("abc")).toBeUndefined();
|
|
16
76
|
expect(parseCostHeader("-1")).toBeUndefined();
|
|
17
77
|
});
|
|
78
|
+
|
|
79
|
+
it("trims valid zero-valued headers", () => {
|
|
80
|
+
expect(parseCostHeader(" 0 ")).toBe(0);
|
|
81
|
+
});
|
|
18
82
|
});
|
|
19
83
|
|
|
20
84
|
describe("extractCosts", () => {
|
|
@@ -73,4 +137,117 @@ describe("extractCosts", () => {
|
|
|
73
137
|
const { ids } = extractCosts(body);
|
|
74
138
|
expect(ids.length).toBe(50);
|
|
75
139
|
});
|
|
76
|
-
});
|
|
140
|
+
});
|
|
141
|
+
|
|
142
|
+
describe("live assistant usage reconciliation", () => {
|
|
143
|
+
it("replaces finalized assistant usage only after the streamed body completes and keeps the custom entry", async () => {
|
|
144
|
+
const responseId = "live-reconcile-1";
|
|
145
|
+
const body = `data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\ndata: [DONE]\n\n`;
|
|
146
|
+
installFetch(body);
|
|
147
|
+
const { appendEntry, handlers } = createHarness();
|
|
148
|
+
await startSession(handlers);
|
|
149
|
+
|
|
150
|
+
const response = await globalThis.fetch("https://openrouter.ai/api/v1/chat/completions");
|
|
151
|
+
const messageEnd = handlers.get("message_end")!;
|
|
152
|
+
expect(await messageEnd({ message: assistantMessage(responseId) }, {})).toBeUndefined();
|
|
153
|
+
|
|
154
|
+
expect(await response.text()).toBe(body);
|
|
155
|
+
const result = await messageEnd({ message: assistantMessage(responseId) }, {});
|
|
156
|
+
expect(result.message.usage.cost).toEqual({
|
|
157
|
+
input: 0.2,
|
|
158
|
+
output: 0.4,
|
|
159
|
+
cacheRead: 0.1,
|
|
160
|
+
cacheWrite: 0.05,
|
|
161
|
+
total: 0.0123,
|
|
162
|
+
});
|
|
163
|
+
expect(appendEntry).toHaveBeenCalledWith(
|
|
164
|
+
"cost-reconcile",
|
|
165
|
+
expect.objectContaining({ provider: "openrouter", responseId, cost: 0.0123, source: "payload" }),
|
|
166
|
+
);
|
|
167
|
+
});
|
|
168
|
+
|
|
169
|
+
it("uses a valid zero-valued billed header over the payload cost", async () => {
|
|
170
|
+
const responseId = "live-header-zero-2";
|
|
171
|
+
installFetch(`data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\n`, {
|
|
172
|
+
"x-litellm-response-cost": "0",
|
|
173
|
+
});
|
|
174
|
+
const { handlers } = createHarness();
|
|
175
|
+
await startSession(handlers);
|
|
176
|
+
|
|
177
|
+
const response = await globalThis.fetch("https://gateway.example/v1/chat/completions");
|
|
178
|
+
await response.text();
|
|
179
|
+
const result = await handlers.get("message_end")!({ message: assistantMessage(responseId) }, {});
|
|
180
|
+
expect(result.message.usage.cost.total).toBe(0);
|
|
181
|
+
});
|
|
182
|
+
|
|
183
|
+
it("falls back when a captured body has multiple candidate response ids", async () => {
|
|
184
|
+
const responseId = "live-ambiguous-response-3";
|
|
185
|
+
installFetch(
|
|
186
|
+
`data: {"id":"${responseId}","usage":{"cost":0.0123},"tool_calls":[{"id":"tool-ambiguous-3"}]}\n\n`,
|
|
187
|
+
);
|
|
188
|
+
const { appendEntry, handlers } = createHarness();
|
|
189
|
+
await startSession(handlers);
|
|
190
|
+
|
|
191
|
+
const response = await globalThis.fetch("https://gateway.example/v1/chat/completions");
|
|
192
|
+
await response.text();
|
|
193
|
+
const message = assistantMessage(responseId);
|
|
194
|
+
expect(await handlers.get("message_end")!({ message }, {})).toBeUndefined();
|
|
195
|
+
expect(message.usage.cost.total).toBe(0.75);
|
|
196
|
+
expect(appendEntry).not.toHaveBeenCalled();
|
|
197
|
+
});
|
|
198
|
+
|
|
199
|
+
it("falls back when the process cache sees a duplicate response id", async () => {
|
|
200
|
+
const responseId = "live-duplicate-response-4";
|
|
201
|
+
installFetch(`data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\n`);
|
|
202
|
+
const { appendEntry, handlers } = createHarness();
|
|
203
|
+
await startSession(handlers);
|
|
204
|
+
|
|
205
|
+
const first = await globalThis.fetch("https://gateway.example/v1/chat/completions");
|
|
206
|
+
const second = await globalThis.fetch("https://gateway.example/v1/chat/completions");
|
|
207
|
+
await Promise.all([first.text(), second.text()]);
|
|
208
|
+
const message = assistantMessage(responseId);
|
|
209
|
+
expect(await handlers.get("message_end")!({ message }, {})).toBeUndefined();
|
|
210
|
+
expect(message.usage.cost.total).toBe(0.75);
|
|
211
|
+
expect(appendEntry).not.toHaveBeenCalled();
|
|
212
|
+
});
|
|
213
|
+
|
|
214
|
+
it("preserves the original message when no valid billed cost was captured", async () => {
|
|
215
|
+
const responseId = "live-no-cost-3";
|
|
216
|
+
installFetch(`data: {"id":"${responseId}","usage":{"cost":-1}}\n\n`);
|
|
217
|
+
const { appendEntry, handlers } = createHarness();
|
|
218
|
+
await startSession(handlers);
|
|
219
|
+
|
|
220
|
+
const response = await globalThis.fetch("https://gateway.example/v1/chat/completions");
|
|
221
|
+
await response.text();
|
|
222
|
+
const message = assistantMessage(responseId);
|
|
223
|
+
expect(await handlers.get("message_end")!({ message }, {})).toBeUndefined();
|
|
224
|
+
expect(message.usage.cost.total).toBe(0.75);
|
|
225
|
+
expect(appendEntry).not.toHaveBeenCalled();
|
|
226
|
+
});
|
|
227
|
+
|
|
228
|
+
it("preserves terminal error messages even when a billed cost was captured", async () => {
|
|
229
|
+
const responseId = "live-terminal-4";
|
|
230
|
+
installFetch(`data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\n`);
|
|
231
|
+
const { appendEntry, handlers } = createHarness();
|
|
232
|
+
await startSession(handlers);
|
|
233
|
+
|
|
234
|
+
const response = await globalThis.fetch("https://gateway.example/v1/chat/completions");
|
|
235
|
+
await response.text();
|
|
236
|
+
const message = { ...assistantMessage(responseId), stopReason: "error" };
|
|
237
|
+
expect(await handlers.get("message_end")!({ message }, {})).toBeUndefined();
|
|
238
|
+
expect(message.usage.cost.total).toBe(0.75);
|
|
239
|
+
expect(appendEntry).not.toHaveBeenCalled();
|
|
240
|
+
});
|
|
241
|
+
|
|
242
|
+
it("does not capture unrelated fetches", async () => {
|
|
243
|
+
const responseId = "unrelated-fetch-5";
|
|
244
|
+
const body = `data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\n`;
|
|
245
|
+
installFetch(body);
|
|
246
|
+
const { handlers } = createHarness();
|
|
247
|
+
await startSession(handlers);
|
|
248
|
+
|
|
249
|
+
const response = await globalThis.fetch("https://example.com/data.json");
|
|
250
|
+
expect(await response.text()).toBe(body);
|
|
251
|
+
expect(await handlers.get("message_end")!({ message: assistantMessage(responseId) }, {})).toBeUndefined();
|
|
252
|
+
});
|
|
253
|
+
});
|
|
@@ -9,19 +9,16 @@
|
|
|
9
9
|
*
|
|
10
10
|
* Extensions cannot hook the response body through the event API
|
|
11
11
|
* (`after_provider_response` exposes only status/headers), but provider SDKs
|
|
12
|
-
* fall back to `globalThis.fetch`. This extension installs a
|
|
13
|
-
* fetch once per process: LLM API responses are scanned for a
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
* `message_end
|
|
12
|
+
* fall back to `globalThis.fetch`. This extension installs a stream-wrapping
|
|
13
|
+
* fetch once per process: LLM API responses are scanned for a provider-reported
|
|
14
|
+
* cost (LiteLLM `x-litellm-response-cost` header, OpenRouter `usage.cost`,
|
|
15
|
+
* `total_cost`, plus object-shaped `cost.total`) and response id. The wrapper
|
|
16
|
+
* finishes capture before the provider SDK finalizes its assistant message, so
|
|
17
|
+
* `message_end` can replace `usage.cost.total` before session persistence.
|
|
18
18
|
*
|
|
19
|
-
* Providers whose payloads carry no cost
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
* The zentui footer prefers reconciled entries over the rate-card total when
|
|
24
|
-
* both exist for the same response id.
|
|
19
|
+
* Providers whose payloads carry no cost retain their rate-card estimate.
|
|
20
|
+
* Reconciled entries remain persisted for zentui and existing-session
|
|
21
|
+
* compatibility.
|
|
25
22
|
*/
|
|
26
23
|
|
|
27
24
|
import type { ExtensionAPI } from "@selesai/code";
|
|
@@ -31,11 +28,10 @@ const ENTRY_VERSION = 1 as const;
|
|
|
31
28
|
|
|
32
29
|
// Guard against pathological bodies; real usage chunks arrive well under 1MB.
|
|
33
30
|
const MAX_CAPTURE_BYTES = 4 * 1024 * 1024;
|
|
34
|
-
// Only LLM API paths are
|
|
35
|
-
//
|
|
31
|
+
// Only LLM API paths are scanned; unrelated fetches (web fetch tool, model
|
|
32
|
+
// catalog refreshes) pass through untouched.
|
|
36
33
|
const LLM_PATH_RE = /(\/chat\/completions|\/responses|\/messages\b|:streamGenerateContent)/;
|
|
37
34
|
const MAX_CAPTURED_IDS = 50;
|
|
38
|
-
const MAX_PENDING_LOOKUPS = 200;
|
|
39
35
|
const MAX_SESSION_ENTRIES = 2_000;
|
|
40
36
|
|
|
41
37
|
type ReconcileEntry = {
|
|
@@ -49,46 +45,34 @@ type ReconcileEntry = {
|
|
|
49
45
|
source: "payload";
|
|
50
46
|
};
|
|
51
47
|
|
|
52
|
-
/**
|
|
53
|
-
* Scan a captured SSE/JSON body for provider-reported cost and response ids.
|
|
54
|
-
* Exported for tests.
|
|
55
|
-
*
|
|
56
|
-
* In streams, usage lands in the final chunk, so the LAST cost match wins.
|
|
57
|
-
*/
|
|
58
48
|
type MessageInfo = {
|
|
59
49
|
provider: string;
|
|
60
50
|
model: string;
|
|
61
51
|
responseId: string;
|
|
62
52
|
};
|
|
63
53
|
|
|
64
|
-
// Process-wide capture state. The fetch patch is installed once per process
|
|
65
|
-
//
|
|
66
|
-
//
|
|
67
|
-
const processCaptured = new Map<string, number>();
|
|
68
|
-
const processWaiting = new Map<
|
|
69
|
-
string,
|
|
70
|
-
{ info: MessageInfo; write: (info: MessageInfo, cost: number) => void }
|
|
71
|
-
>();
|
|
54
|
+
// Process-wide capture state. The fetch patch is installed once per process;
|
|
55
|
+
// per-session message_end handlers consume the matching response id. A duplicate
|
|
56
|
+
// capture for the same id is marked ambiguous rather than assigning either bill.
|
|
57
|
+
const processCaptured = new Map<string, number | null>();
|
|
72
58
|
|
|
73
|
-
/**
|
|
74
|
-
* Scan a captured SSE/JSON body for provider-reported cost and response ids.
|
|
75
|
-
* Pure function; exported for tests.
|
|
76
|
-
*
|
|
77
|
-
* In streams, usage lands in the final chunk, so the LAST cost match wins.
|
|
78
|
-
*/
|
|
79
59
|
/**
|
|
80
60
|
* Parse a LiteLLM `x-litellm-response-cost` header value (a USD float,
|
|
81
61
|
* possibly in scientific notation). Exported for tests.
|
|
82
62
|
*/
|
|
83
63
|
export function parseCostHeader(value: string | null): number | undefined {
|
|
84
|
-
|
|
85
|
-
|
|
64
|
+
const trimmed = value?.trim();
|
|
65
|
+
if (!trimmed) return undefined;
|
|
66
|
+
const cost = Number(trimmed);
|
|
86
67
|
return Number.isFinite(cost) && cost >= 0 ? cost : undefined;
|
|
87
68
|
}
|
|
88
69
|
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
70
|
+
/**
|
|
71
|
+
* Scan a captured SSE/JSON body for provider-reported cost and response ids.
|
|
72
|
+
* Exported for tests. In streams, usage lands in the final chunk, so the LAST
|
|
73
|
+
* valid cost match wins.
|
|
74
|
+
*/
|
|
75
|
+
export function extractCosts(text: string): { ids: string[]; cost?: number } {
|
|
92
76
|
const ids = new Set<string>();
|
|
93
77
|
for (const match of text.matchAll(/"id"\s*:\s*"([^"]{1,200})"/g)) {
|
|
94
78
|
ids.add(match[1]);
|
|
@@ -151,26 +135,22 @@ export default function costReconcileExtension(pi: ExtensionAPI): void {
|
|
|
151
135
|
try {
|
|
152
136
|
pi.appendEntry(ENTRY_TYPE, entry);
|
|
153
137
|
} catch {
|
|
154
|
-
//
|
|
138
|
+
// Transcript persistence failure must not break the session.
|
|
155
139
|
}
|
|
156
140
|
};
|
|
157
141
|
|
|
158
142
|
const finishCapture = (rawBody: string, headerCost?: number): void => {
|
|
159
|
-
//
|
|
160
|
-
//
|
|
143
|
+
// Ponytail: one naive regex pass over the whole body; a streaming parser
|
|
144
|
+
// only matters if bodies grow past the 4MB cap.
|
|
161
145
|
const { ids, cost } = extractCosts(rawBody);
|
|
162
|
-
// LiteLLM's header is the amount the gateway billed; prefer it over
|
|
163
|
-
//
|
|
146
|
+
// LiteLLM's header is the amount the gateway billed; prefer it over any
|
|
147
|
+
// cost the upstream payload reports.
|
|
164
148
|
const effectiveCost = headerCost ?? cost;
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
processWaiting.delete(responseId);
|
|
171
|
-
waiting.write(waiting.info, effectiveCost);
|
|
172
|
-
}
|
|
173
|
-
}
|
|
149
|
+
// A cost must identify exactly one response. Applying it to every `id` in
|
|
150
|
+
// a body can mistake nested tool/message ids for the assistant response.
|
|
151
|
+
if (effectiveCost === undefined || ids.length !== 1) return;
|
|
152
|
+
const responseId = ids[0]!;
|
|
153
|
+
processCaptured.set(responseId, processCaptured.has(responseId) ? null : effectiveCost);
|
|
174
154
|
if (processCaptured.size > 500) {
|
|
175
155
|
for (const key of [...processCaptured.keys()].slice(0, processCaptured.size - 500))
|
|
176
156
|
processCaptured.delete(key);
|
|
@@ -191,7 +171,7 @@ export default function costReconcileExtension(pi: ExtensionAPI): void {
|
|
|
191
171
|
if (typeof responseId === "string") seen.add(responseId);
|
|
192
172
|
}
|
|
193
173
|
} catch {
|
|
194
|
-
//
|
|
174
|
+
// Best effort only.
|
|
195
175
|
}
|
|
196
176
|
|
|
197
177
|
const globalFetch = globalThis as typeof globalThis & { [FETCH_PATCHED]?: boolean };
|
|
@@ -205,26 +185,36 @@ export default function costReconcileExtension(pi: ExtensionAPI): void {
|
|
|
205
185
|
const contentType = response.headers.get("content-type") ?? "";
|
|
206
186
|
if (!contentType.includes("json") && !contentType.includes("event-stream")) return response;
|
|
207
187
|
if (!response.body) return response;
|
|
188
|
+
|
|
208
189
|
const headerCost = parseCostHeader(response.headers.get("x-litellm-response-cost"));
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
raw
|
|
219
|
-
|
|
190
|
+
let raw = "";
|
|
191
|
+
let captureFailed = false;
|
|
192
|
+
const decoder = new TextDecoder();
|
|
193
|
+
const capture = new TransformStream<Uint8Array, Uint8Array>({
|
|
194
|
+
transform(chunk, controller) {
|
|
195
|
+
controller.enqueue(chunk);
|
|
196
|
+
if (captureFailed) return;
|
|
197
|
+
try {
|
|
198
|
+
raw += decoder.decode(chunk, { stream: true });
|
|
199
|
+
if (raw.length > MAX_CAPTURE_BYTES) captureFailed = true;
|
|
200
|
+
} catch {
|
|
201
|
+
captureFailed = true;
|
|
220
202
|
}
|
|
221
|
-
}
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
203
|
+
},
|
|
204
|
+
flush() {
|
|
205
|
+
if (captureFailed) return;
|
|
206
|
+
try {
|
|
207
|
+
raw += decoder.decode();
|
|
208
|
+
finishCapture(raw, headerCost);
|
|
209
|
+
} catch {
|
|
210
|
+
// Capture is best-effort; the response stream is unaffected.
|
|
211
|
+
}
|
|
212
|
+
},
|
|
213
|
+
});
|
|
214
|
+
// flush() runs before the SDK observes EOF, so message_end can safely
|
|
215
|
+
// replace the finalized message usage without waiting on another fetch.
|
|
216
|
+
const body = response.body.pipeThrough(capture);
|
|
217
|
+
return new Response(body, {
|
|
228
218
|
status: response.status,
|
|
229
219
|
statusText: response.statusText,
|
|
230
220
|
headers: response.headers,
|
|
@@ -236,43 +226,32 @@ export default function costReconcileExtension(pi: ExtensionAPI): void {
|
|
|
236
226
|
});
|
|
237
227
|
|
|
238
228
|
pi.on("message_end", (event) => {
|
|
239
|
-
const message = event.message
|
|
240
|
-
|
|
241
|
-
role?: string;
|
|
242
|
-
provider?: string;
|
|
243
|
-
model?: string;
|
|
244
|
-
responseId?: string;
|
|
245
|
-
stopReason?: string;
|
|
246
|
-
}
|
|
247
|
-
| undefined;
|
|
248
|
-
if (!message || message.role !== "assistant") return;
|
|
229
|
+
const message = event.message;
|
|
230
|
+
if (message.role !== "assistant") return;
|
|
249
231
|
if (message.stopReason === "error" || message.stopReason === "aborted") return;
|
|
250
232
|
const responseId = message.responseId;
|
|
251
233
|
if (typeof responseId !== "string" || responseId.length === 0) return;
|
|
252
234
|
|
|
253
|
-
const info: MessageInfo = {
|
|
254
|
-
provider: message.provider ?? "",
|
|
255
|
-
model: message.responseModel ?? message.model ?? "",
|
|
256
|
-
responseId,
|
|
257
|
-
};
|
|
258
235
|
const cost = processCaptured.get(responseId);
|
|
259
|
-
if (cost
|
|
260
|
-
|
|
236
|
+
if (cost === undefined || cost === null || !message.usage?.cost) {
|
|
237
|
+
processCaptured.delete(responseId);
|
|
261
238
|
return;
|
|
262
239
|
}
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
240
|
+
processCaptured.delete(responseId);
|
|
241
|
+
|
|
242
|
+
writeEntry(
|
|
243
|
+
{
|
|
244
|
+
provider: message.provider ?? "",
|
|
245
|
+
model: message.responseModel ?? message.model ?? "",
|
|
246
|
+
responseId,
|
|
247
|
+
},
|
|
248
|
+
cost,
|
|
249
|
+
);
|
|
250
|
+
return {
|
|
251
|
+
message: {
|
|
252
|
+
...message,
|
|
253
|
+
usage: { ...message.usage, cost: { ...message.usage.cost, total: cost } },
|
|
254
|
+
},
|
|
255
|
+
};
|
|
269
256
|
});
|
|
270
257
|
}
|
|
271
|
-
|
|
272
|
-
// Trim oldest waiting entries so unclaimed captures cannot grow unbounded.
|
|
273
|
-
function waitingLimitGuard(): void {
|
|
274
|
-
if (processWaiting.size >= MAX_PENDING_LOOKUPS) {
|
|
275
|
-
for (const key of [...processWaiting.keys()].slice(0, processWaiting.size - MAX_PENDING_LOOKUPS + 1))
|
|
276
|
-
processWaiting.delete(key);
|
|
277
|
-
}
|
|
278
|
-
}
|