@selesai/code 0.13.2 → 0.13.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. package/CHANGELOG.md +20 -0
  2. package/dist/defaults/models.json +85 -13
  3. package/dist/extensions/cost-reconcile.test.ts +200 -4
  4. package/dist/extensions/cost-reconcile.ts +88 -102
  5. package/dist/extensions/pi-intercom/CHANGELOG.md +13 -0
  6. package/dist/extensions/pi-intercom/README.md +4 -5
  7. package/dist/extensions/pi-intercom/config.test.ts +3 -31
  8. package/dist/extensions/pi-intercom/config.ts +0 -15
  9. package/dist/extensions/pi-intercom/index.ts +9 -46
  10. package/dist/extensions/pi-intercom/intercom.integration.test.ts +49 -57
  11. package/dist/extensions/pi-intercom/package.json +1 -1
  12. package/dist/extensions/pi-intercom/reply-tracker.test.ts +20 -0
  13. package/dist/extensions/pi-intercom/reply-tracker.ts +8 -0
  14. package/dist/extensions/pi-subagents/CHANGELOG.md +27 -0
  15. package/dist/extensions/pi-subagents/docs/tool-reference.md +4 -1
  16. package/dist/extensions/pi-subagents/docs/workflows.md +2 -2
  17. package/dist/extensions/pi-subagents/package-lock.json +2 -2
  18. package/dist/extensions/pi-subagents/package.json +1 -1
  19. package/dist/extensions/pi-subagents/skills/council-mode/SKILL.md +48 -243
  20. package/dist/extensions/pi-subagents/skills/council-mode/references/pass-contracts.md +150 -0
  21. package/dist/extensions/pi-subagents/skills/pi-subagents/SKILL.md +87 -37
  22. package/dist/extensions/pi-subagents/skills/pi-subagents/references/constraints-and-recipes.md +29 -233
  23. package/dist/extensions/pi-subagents/skills/pi-subagents/references/execution-controls.md +49 -8
  24. package/dist/extensions/pi-subagents/skills/pi-subagents/references/management-authoring-rpc.md +2 -2
  25. package/dist/extensions/pi-subagents/skills/pi-subagents/references/multi-lane-orchestration.md +13 -1
  26. package/dist/extensions/pi-subagents/skills/pi-subagents/references/prompting-and-roles.md +34 -27
  27. package/dist/extensions/pi-subagents/skills/pi-subagents/references/review-and-validation.md +73 -0
  28. package/dist/extensions/pi-subagents/src/agents/agent-management.ts +157 -28
  29. package/dist/extensions/pi-subagents/src/api/shared-types.ts +2 -0
  30. package/dist/extensions/pi-subagents/src/extension/public-execution.ts +1 -0
  31. package/dist/extensions/pi-subagents/src/extension/schemas.ts +1 -0
  32. package/dist/extensions/pi-subagents/src/extension/tool-description.ts +4 -1
  33. package/dist/extensions/pi-subagents/src/runs/background/async-execution.ts +2 -2
  34. package/dist/extensions/pi-subagents/src/runs/background/async-job-tracker.ts +3 -0
  35. package/dist/extensions/pi-subagents/src/runs/background/async-status.ts +45 -2
  36. package/dist/extensions/pi-subagents/src/runs/background/control-channel.ts +3 -2
  37. package/dist/extensions/pi-subagents/src/runs/background/run-status.ts +13 -2
  38. package/dist/extensions/pi-subagents/src/runs/background/subagent-runner.ts +5 -1
  39. package/dist/extensions/pi-subagents/src/runs/background/subagent-wait.ts +10 -2
  40. package/dist/extensions/pi-subagents/src/runs/background/wait-completions.ts +3 -0
  41. package/dist/extensions/pi-subagents/src/runs/foreground/execution.ts +11 -2
  42. package/dist/extensions/pi-subagents/src/runs/foreground/subagent-executor.ts +98 -1
  43. package/dist/extensions/pi-subagents/src/runs/shared/async-status-projection.ts +138 -4
  44. package/dist/extensions/pi-subagents/src/runs/shared/background-process-options.ts +9 -0
  45. package/dist/extensions/pi-subagents/src/runs/shared/mcp-direct-tool-grant.ts +2 -5
  46. package/dist/extensions/pi-subagents/src/runs/shared/mutation-evidence.ts +52 -3
  47. package/dist/extensions/pi-subagents/src/runs/shared/pi-args.ts +47 -1
  48. package/dist/extensions/pi-subagents/src/runs/shared/single-output.ts +45 -18
  49. package/dist/extensions/pi-subagents/src/runs/shared/subagent-prompt-runtime.ts +21 -2
  50. package/dist/extensions/pi-subagents/src/runs/shared/workflow-graph.ts +15 -0
  51. package/dist/extensions/pi-subagents/src/shared/types.ts +34 -1
  52. package/dist/extensions/pi-subagents/src/tui/fleet-status.ts +11 -3
  53. package/dist/extensions/pi-subagents/src/tui/render-helpers.ts +31 -0
  54. package/dist/extensions/pi-subagents/src/tui/render.ts +597 -112
  55. package/dist/extensions/pi-subagents/src/watchdog/change-signature.ts +40 -1
  56. package/dist/extensions/pi-subagents/src/workflows/host-command.ts +6 -1
  57. package/dist/extensions/pi-subagents/src/workflows/scripted-workflow.ts +53 -2
  58. package/dist/extensions/pi-subagents/test/integration/async-execution.test.ts +55 -6
  59. package/dist/extensions/pi-subagents/test/integration/async-status.test.ts +111 -1
  60. package/dist/extensions/pi-subagents/test/integration/render-fork-badge.test.ts +206 -38
  61. package/dist/extensions/pi-subagents/test/integration/render-widget.test.ts +522 -31
  62. package/dist/extensions/pi-subagents/test/integration/single-execution.test.ts +123 -0
  63. package/dist/extensions/pi-subagents/test/unit/agent-management.test.ts +48 -0
  64. package/dist/extensions/pi-subagents/test/unit/async-status-projection.test.ts +57 -1
  65. package/dist/extensions/pi-subagents/test/unit/background-process-options.test.ts +17 -0
  66. package/dist/extensions/pi-subagents/test/unit/external-cli-runner.test.ts +1 -1
  67. package/dist/extensions/pi-subagents/test/unit/fleet-status.test.ts +44 -4
  68. package/dist/extensions/pi-subagents/test/unit/fork-cache-key.test.ts +91 -0
  69. package/dist/extensions/pi-subagents/test/unit/host-command.test.ts +1 -0
  70. package/dist/extensions/pi-subagents/test/unit/index-child-registration.test.ts +0 -1
  71. package/dist/extensions/pi-subagents/test/unit/mcp-direct-tool-grant.test.ts +20 -3
  72. package/dist/extensions/pi-subagents/test/unit/mutation-evidence.test.ts +27 -0
  73. package/dist/extensions/pi-subagents/test/unit/pi-args.test.ts +93 -17
  74. package/dist/extensions/pi-subagents/test/unit/public-execution.test.ts +1 -0
  75. package/dist/extensions/pi-subagents/test/unit/render-helpers.test.ts +103 -26
  76. package/dist/extensions/pi-subagents/test/unit/run-status.test.ts +58 -0
  77. package/dist/extensions/pi-subagents/test/unit/schemas.test.ts +22 -2
  78. package/dist/extensions/pi-subagents/test/unit/scripted-workflow.test.ts +21 -0
  79. package/dist/extensions/pi-subagents/test/unit/single-output.test.ts +13 -0
  80. package/dist/extensions/pi-subagents/test/unit/subagent-wait.test.ts +54 -0
  81. package/dist/extensions/pi-subagents/test/unit/tool-description.test.ts +2 -0
  82. package/dist/extensions/pi-subagents/test/unit/wait-completions.test.ts +32 -0
  83. package/dist/extensions/pi-subagents/test/unit/watchdog-change-signature.test.ts +48 -1
  84. package/dist/extensions/pi-subagents/test/unit/widget-nested-render.test.ts +21 -6
  85. package/dist/extensions/pi-subagents/test/unit/windows-hide-spawn.test.ts +11 -0
  86. package/dist/extensions/pi-web-agent/CHANGELOG.md +410 -0
  87. package/dist/extensions/pi-web-agent/README.md +131 -0
  88. package/dist/extensions/pi-web-agent/package.json +4 -2
  89. package/dist/extensions/pi-web-agent/src/backends/config.ts +62 -5
  90. package/dist/extensions/pi-web-agent/src/backends/doctor.ts +136 -0
  91. package/dist/extensions/pi-web-agent/src/backends/factory.ts +96 -6
  92. package/dist/extensions/pi-web-agent/src/commands/web-agent-config.ts +183 -47
  93. package/dist/extensions/pi-web-agent/src/extension.ts +62 -25
  94. package/dist/extensions/pi-web-agent/src/extract/readability.ts +19 -11
  95. package/dist/extensions/pi-web-agent/src/orchestration/candidate-selector.ts +5 -4
  96. package/dist/extensions/pi-web-agent/src/orchestration/direct-url.ts +2 -25
  97. package/dist/extensions/pi-web-agent/src/orchestration/evidence-quality.ts +5 -2
  98. package/dist/extensions/pi-web-agent/src/orchestration/evidence-ranker.ts +2 -0
  99. package/dist/extensions/pi-web-agent/src/orchestration/research-orchestrator.ts +72 -7
  100. package/dist/extensions/pi-web-agent/src/orchestration/research-types.ts +8 -1
  101. package/dist/extensions/pi-web-agent/src/orchestration/research-worker.ts +28 -3
  102. package/dist/extensions/pi-web-agent/src/orchestration/source-profile.ts +4 -0
  103. package/dist/extensions/pi-web-agent/src/orchestration/url.ts +35 -0
  104. package/dist/extensions/pi-web-agent/src/presentation/explore-presentation.ts +18 -6
  105. package/dist/extensions/pi-web-agent/src/presentation/search-presentation.ts +14 -2
  106. package/dist/extensions/pi-web-agent/src/readers/github-reader.ts +150 -0
  107. package/dist/extensions/pi-web-agent/src/readers/limits.ts +3 -0
  108. package/dist/extensions/pi-web-agent/src/readers/pdf-reader.ts +87 -0
  109. package/dist/extensions/pi-web-agent/src/readers/resolver.ts +25 -0
  110. package/dist/extensions/pi-web-agent/src/readers/types.ts +11 -0
  111. package/dist/extensions/pi-web-agent/src/readers/youtube-reader.ts +79 -0
  112. package/dist/extensions/pi-web-agent/src/search/duckduckgo.ts +32 -5
  113. package/dist/extensions/pi-web-agent/src/search/exa.ts +109 -0
  114. package/dist/extensions/pi-web-agent/src/search/fanout.ts +154 -0
  115. package/dist/extensions/pi-web-agent/src/search/tavily.ts +113 -0
  116. package/dist/extensions/pi-web-agent/src/search/youcom.ts +109 -0
  117. package/dist/extensions/pi-web-agent/src/tools/web-search.ts +22 -9
  118. package/dist/extensions/pi-web-agent/src/types.ts +19 -4
  119. package/dist/extensions/tokenin-onboarding.ts +302 -0
  120. package/package.json +1 -1
package/CHANGELOG.md CHANGED
@@ -2,6 +2,26 @@
2
2
 
3
3
  All notable changes to `@selesai/code` will be documented in this file.
4
4
 
5
+ ## [0.13.4] - 2026-08-31
6
+
7
+ ### Added
8
+ - **Bundled `pi-subagents` 0.60.0.** `subagent({ action: "list", capabilities: true })` returns a compact prompt-free capability catalog (`details.catalog`: agent name, description, source, runner, tools/model/execution/output/extensions snapshot, executable and restriction status), and `list` results stay human text. Forked children now stamp an OpenAI-compatible `prompt_cache_key` (`pi-fork:<sha256>`) on provider requests for prompt-cache reuse; timeout recovery evidence (report status, changed tracked files, dirty-worktree classification) is projected into async status and `subagent_wait` completions; `runs.lanes` lane plans (stages, phase, label, structured output) are emitted and rendered as workflow-graph stages; `runs.host` rejects per-step `cwd` with a hint; watchdog repo signatures skip untracked scans for home-root and entry-less repos; single-output snapshots propagate non-ENOENT stat errors; background processes use `windowsHide` and detached-only-on-unix.
9
+ - **Bundled `pi-web-agent` 1.10.0.** Search backends: You.com (`YDC_API_KEY`), Exa (`EXA_API_KEY`), Tavily (`TAVILY_API_KEY`) in addition to DuckDuckGo, SearXNG and Brave; search fanout across configured providers with dedupe/agreement ranking (`backends.search.fanout` with `off`/`on`/`auto` modes and provider selection); keyless Tavily fallback when DuckDuckGo is bot-blocked (disable with `PI_WEB_AGENT_DISABLE_KEYLESS_FALLBACK=1`); DuckDuckGo hardened against bot-walls (browser headers + one retry); GitHub/PDF/YouTube direct readers wired into the fetch path; canonical URL normalization (tracking-param stripping); the model tool result now carries the full synthesized findings/sources/caveats regardless of terminal presentation.
10
+ - **GLM-5.3 models in the default catalog.** Added `glm-5.3` (GLM-5.3) to the bundled `tokenin` provider and upgraded `glm-5.3-flash` to text+image input with a full `thinkingLevelMap`.
11
+
12
+ ### Changed
13
+ - **Cost reconciliation keys every streamed response id.** Streamed bodies repeat the response id across chunks and carry tool-call ids (`call_*`), so the fetch wrapper now keys the billed cost by every id found and `message_end` consumes only the one matching the finalized message's `responseId`; a duplicate capture (retry) is marked ambiguous instead of assigning either bill.
14
+ - **Bundled `pi-intercom` 0.12.1.** During a turn triggered by an inbound ask, a non-`reply` `send` to a different target is refused so guessed parent/root CWD cannot receive an accidental reply; the `toolVisibility` setting and `after-first-use` reveal path were removed (the generic `intercom` tool stays in the active tool set for prompt-cache stability; obsolete `toolVisibility` config keys are ignored).
15
+
16
+ ## [0.13.3] - 2026-08-31
17
+
18
+ ### Added
19
+ - **Token-In multi-key auto-failover.** The `tokenin-onboarding` extension now registers a `tokenin` provider whose `streamSimple` automatically rotates to an alternative saved account when the active key fails with a 401, 429, or budget/quota error. Failed keys go on a 5-minute in-memory cooldown before being retried, and the newly active credential is persisted to both `tokenin-auth.json` and `auth.json`.
20
+
21
+ ### Changed
22
+ - **Cost reconciliation replaces the finalized assistant cost.** The `cost-reconcile` fetch wrapper now finishes capturing the provider-reported cost before the provider SDK finalizes its assistant message, so `message_end` replaces `usage.cost.total` with the billed amount before the message is persisted. Providers whose payloads carry no cost keep their rate-card estimate.
23
+ - **Default model catalog reasoning maps.** Added `thinkingLevelMap` entries (and `low`=low / `max` mappings) across bundled model defaults and enabled reasoning on Qwen3.8 27B and Gemini 3.7 Flash so thinking budgets map correctly.
24
+
5
25
  ## [0.13.2] - 2026-08-30
6
26
 
7
27
  ### Added
@@ -19,7 +19,7 @@
19
19
  "maxTokens": 64000,
20
20
  "thinkingLevelMap": {
21
21
  "minimal": null,
22
- "low": null,
22
+ "low": "low",
23
23
  "medium": null,
24
24
  "high": "high",
25
25
  "xhigh": null,
@@ -38,7 +38,7 @@
38
38
  "maxTokens": 64000,
39
39
  "thinkingLevelMap": {
40
40
  "minimal": null,
41
- "low": null,
41
+ "low": "low",
42
42
  "medium": null,
43
43
  "high": "high",
44
44
  "xhigh": null,
@@ -57,7 +57,7 @@
57
57
  "maxTokens": 64000,
58
58
  "thinkingLevelMap": {
59
59
  "minimal": null,
60
- "low": null,
60
+ "low": "low",
61
61
  "medium": null,
62
62
  "high": "high",
63
63
  "xhigh": null,
@@ -114,7 +114,7 @@
114
114
  "maxTokens": 64000,
115
115
  "thinkingLevelMap": {
116
116
  "minimal": null,
117
- "low": null,
117
+ "low": "low",
118
118
  "medium": null,
119
119
  "high": "high",
120
120
  "xhigh": null,
@@ -130,7 +130,15 @@
130
130
  "reasoning": true,
131
131
  "input": ["text", "image"],
132
132
  "contextWindow": 512000,
133
- "maxTokens": 128000
133
+ "maxTokens": 128000,
134
+ "thinkingLevelMap": {
135
+ "minimal": null,
136
+ "low": "low",
137
+ "medium": null,
138
+ "high": "high",
139
+ "xhigh": null,
140
+ "max": "max"
141
+ }
134
142
  },
135
143
  {
136
144
  "id": "gpt-5.6-terra",
@@ -138,7 +146,15 @@
138
146
  "reasoning": true,
139
147
  "input": ["text", "image"],
140
148
  "contextWindow": 512000,
141
- "maxTokens": 128000
149
+ "maxTokens": 128000,
150
+ "thinkingLevelMap": {
151
+ "minimal": null,
152
+ "low": "low",
153
+ "medium": null,
154
+ "high": "high",
155
+ "xhigh": null,
156
+ "max": null
157
+ }
142
158
  },
143
159
  {
144
160
  "id": "gpt-5.6-sol",
@@ -146,7 +162,15 @@
146
162
  "reasoning": true,
147
163
  "input": ["text", "image"],
148
164
  "contextWindow": 512000,
149
- "maxTokens": 128000
165
+ "maxTokens": 128000,
166
+ "thinkingLevelMap": {
167
+ "minimal": null,
168
+ "low": "low",
169
+ "medium": null,
170
+ "high": "high",
171
+ "xhigh": null,
172
+ "max": null
173
+ }
150
174
  },
151
175
  {
152
176
  "id": "kimi-k3",
@@ -200,7 +224,7 @@
200
224
  "id": "glm-5.3-flash",
201
225
  "name": "GLM-5.3 Flash",
202
226
  "reasoning": true,
203
- "input": ["text"],
227
+ "input": ["text", "image"],
204
228
  "contextWindow": 1000000,
205
229
  "maxTokens": 64000,
206
230
  "thinkingLevelMap": {
@@ -209,16 +233,40 @@
209
233
  "medium": null,
210
234
  "high": "high",
211
235
  "xhigh": null,
212
- "max": null
236
+ "max": "max"
237
+ }
238
+ },
239
+ {
240
+ "id": "glm-5.3",
241
+ "name": "GLM-5.3",
242
+ "reasoning": true,
243
+ "input": ["text", "image"],
244
+ "contextWindow": 1000000,
245
+ "maxTokens": 64000,
246
+ "thinkingLevelMap": {
247
+ "minimal": null,
248
+ "low": "low",
249
+ "medium": null,
250
+ "high": "high",
251
+ "xhigh": null,
252
+ "max": "max"
213
253
  }
214
254
  },
215
255
  {
216
256
  "id": "qwen3.8-27b",
217
257
  "name": "Qwen3.8 27B",
218
- "reasoning": false,
258
+ "reasoning": true,
219
259
  "input": ["text", "image"],
220
260
  "contextWindow": 256000,
221
261
  "maxTokens": 64000,
262
+ "thinkingLevelMap": {
263
+ "minimal": null,
264
+ "low": "low",
265
+ "medium": null,
266
+ "high": "high",
267
+ "xhigh": null,
268
+ "max": "max"
269
+ },
222
270
  "compat": {
223
271
  "thinkingFormat": "qwen"
224
272
  }
@@ -226,10 +274,18 @@
226
274
  {
227
275
  "id": "gemini-3.7-flash",
228
276
  "name": "Gemini 3.7 Flash",
229
- "reasoning": false,
230
- "input": ["text", "image" ],
277
+ "reasoning": true,
278
+ "input": ["text", "image"],
231
279
  "contextWindow": 512000,
232
- "maxTokens": 64000
280
+ "maxTokens": 64000,
281
+ "thinkingLevelMap": {
282
+ "minimal": null,
283
+ "low": "low",
284
+ "medium": null,
285
+ "high": "high",
286
+ "xhigh": null,
287
+ "max": null
288
+ }
233
289
  },
234
290
  {
235
291
  "id": "Qwen3-VL",
@@ -238,6 +294,14 @@
238
294
  "input": ["text", "image"],
239
295
  "contextWindow": 256000,
240
296
  "maxTokens": 8192,
297
+ "thinkingLevelMap": {
298
+ "minimal": null,
299
+ "low": "low",
300
+ "medium": null,
301
+ "high": "high",
302
+ "xhigh": null,
303
+ "max": null
304
+ },
241
305
  "compat": {
242
306
  "supportsDeveloperRole": false,
243
307
  "supportsReasoningEffort": false,
@@ -251,6 +315,14 @@
251
315
  "input": ["text", "image"],
252
316
  "contextWindow": 256000,
253
317
  "maxTokens": 8192,
318
+ "thinkingLevelMap": {
319
+ "minimal": null,
320
+ "low": "low",
321
+ "medium": null,
322
+ "high": "high",
323
+ "xhigh": null,
324
+ "max": null
325
+ },
254
326
  "compat": {
255
327
  "supportsDeveloperRole": false,
256
328
  "supportsReasoningEffort": false,
@@ -1,5 +1,63 @@
1
- import { describe, expect, it } from "vitest";
2
- import { extractCosts, parseCostHeader } from "./cost-reconcile.ts";
1
+ import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
2
+ import costReconcileExtension, { extractCosts, parseCostHeader } from "./cost-reconcile.ts";
3
+
4
+ type Handler = (event: any, ctx: any) => any;
5
+
6
+ const FETCH_PATCHED = Symbol.for("selesai.cost-reconcile.fetch-patched");
7
+ let originalFetch: typeof globalThis.fetch;
8
+
9
+ function createHarness() {
10
+ const handlers = new Map<string, Handler>();
11
+ const appendEntry = vi.fn();
12
+ const pi = {
13
+ on: vi.fn((event: string, handler: Handler) => handlers.set(event, handler)),
14
+ appendEntry,
15
+ };
16
+ costReconcileExtension(pi as any);
17
+ return { appendEntry, handlers };
18
+ }
19
+
20
+ async function startSession(handlers: Map<string, Handler>): Promise<void> {
21
+ await handlers.get("session_start")!({}, { sessionManager: { getEntries: () => [] } });
22
+ }
23
+
24
+ function assistantMessage(responseId: string, total = 0.75) {
25
+ return {
26
+ role: "assistant",
27
+ provider: "openrouter",
28
+ model: "openai/gpt-4.1-mini",
29
+ responseModel: "openai/gpt-4.1-mini",
30
+ responseId,
31
+ stopReason: "stop",
32
+ content: [],
33
+ usage: {
34
+ input: 10,
35
+ output: 5,
36
+ cacheRead: 2,
37
+ cacheWrite: 1,
38
+ cost: { input: 0.2, output: 0.4, cacheRead: 0.1, cacheWrite: 0.05, total },
39
+ },
40
+ };
41
+ }
42
+
43
+ function installFetch(body: string, headers: Record<string, string> = {}) {
44
+ const upstreamFetch = vi.fn(async () =>
45
+ new Response(body, { headers: { "content-type": "text/event-stream", ...headers } }),
46
+ );
47
+ globalThis.fetch = upstreamFetch as typeof globalThis.fetch;
48
+ return upstreamFetch;
49
+ }
50
+
51
+ beforeEach(() => {
52
+ originalFetch = globalThis.fetch;
53
+ delete (globalThis as typeof globalThis & { [FETCH_PATCHED]?: boolean })[FETCH_PATCHED];
54
+ });
55
+
56
+ afterEach(() => {
57
+ globalThis.fetch = originalFetch;
58
+ delete (globalThis as typeof globalThis & { [FETCH_PATCHED]?: boolean })[FETCH_PATCHED];
59
+ vi.restoreAllMocks();
60
+ });
3
61
 
4
62
  describe("parseCostHeader", () => {
5
63
  it("parses LiteLLM scientific-notation cost", () => {
@@ -10,11 +68,17 @@ describe("parseCostHeader", () => {
10
68
  expect(parseCostHeader("0.00123")).toBe(0.00123);
11
69
  });
12
70
 
13
- it("rejects null, garbage, and negatives", () => {
71
+ it("rejects null, blank, garbage, and negative values", () => {
14
72
  expect(parseCostHeader(null)).toBeUndefined();
73
+ expect(parseCostHeader("")).toBeUndefined();
74
+ expect(parseCostHeader(" \t ")).toBeUndefined();
15
75
  expect(parseCostHeader("abc")).toBeUndefined();
16
76
  expect(parseCostHeader("-1")).toBeUndefined();
17
77
  });
78
+
79
+ it("trims valid zero-valued headers", () => {
80
+ expect(parseCostHeader(" 0 ")).toBe(0);
81
+ });
18
82
  });
19
83
 
20
84
  describe("extractCosts", () => {
@@ -73,4 +137,136 @@ describe("extractCosts", () => {
73
137
  const { ids } = extractCosts(body);
74
138
  expect(ids.length).toBe(50);
75
139
  });
76
- });
140
+ });
141
+
142
+ describe("live assistant usage reconciliation", () => {
143
+ it("replaces finalized assistant usage only after the streamed body completes and keeps the custom entry", async () => {
144
+ const responseId = "live-reconcile-1";
145
+ const body = `data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\ndata: [DONE]\n\n`;
146
+ installFetch(body);
147
+ const { appendEntry, handlers } = createHarness();
148
+ await startSession(handlers);
149
+
150
+ const response = await globalThis.fetch("https://openrouter.ai/api/v1/chat/completions");
151
+ const messageEnd = handlers.get("message_end")!;
152
+ expect(await messageEnd({ message: assistantMessage(responseId) }, {})).toBeUndefined();
153
+
154
+ expect(await response.text()).toBe(body);
155
+ const result = await messageEnd({ message: assistantMessage(responseId) }, {});
156
+ expect(result.message.usage.cost).toEqual({
157
+ input: 0.2,
158
+ output: 0.4,
159
+ cacheRead: 0.1,
160
+ cacheWrite: 0.05,
161
+ total: 0.0123,
162
+ });
163
+ expect(appendEntry).toHaveBeenCalledWith(
164
+ "cost-reconcile",
165
+ expect.objectContaining({ provider: "openrouter", responseId, cost: 0.0123, source: "payload" }),
166
+ );
167
+ });
168
+
169
+ it("uses a valid zero-valued billed header over the payload cost", async () => {
170
+ const responseId = "live-header-zero-2";
171
+ installFetch(`data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\n`, {
172
+ "x-litellm-response-cost": "0",
173
+ });
174
+ const { handlers } = createHarness();
175
+ await startSession(handlers);
176
+
177
+ const response = await globalThis.fetch("https://gateway.example/v1/chat/completions");
178
+ await response.text();
179
+ const result = await handlers.get("message_end")!({ message: assistantMessage(responseId) }, {});
180
+ expect(result.message.usage.cost.total).toBe(0);
181
+ });
182
+
183
+ it("reconciles a tool-call stream: keys cost by every id, message_end consumes the matching one", async () => {
184
+ const responseId = "live-toolcall-3";
185
+ const body =
186
+ `data: {"id":"${responseId}","choices":[{"delta":{"tool_calls":[{"id":"call_abc123","function":{"name":"bash","arguments":""}}]}}]}\n\n` +
187
+ `data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\n` +
188
+ "data: [DONE]\n\n";
189
+ installFetch(body);
190
+ const { appendEntry, handlers } = createHarness();
191
+ await startSession(handlers);
192
+
193
+ const response = await globalThis.fetch("https://gateway.example/v1/chat/completions");
194
+ await response.text();
195
+ const message = assistantMessage(responseId);
196
+ const result = await handlers.get("message_end")!({ message }, {});
197
+ expect(result.message.usage.cost.total).toBe(0.0123);
198
+ expect(appendEntry).toHaveBeenCalledWith(
199
+ "cost-reconcile",
200
+ expect.objectContaining({ provider: "openrouter", responseId, cost: 0.0123, source: "payload" }),
201
+ );
202
+ });
203
+
204
+ it("does not apply a captured cost to a message whose id was not in the body", async () => {
205
+ const body = `data: {"id":"other-response","usage":{"cost":0.0123}}\n\ndata: [DONE]\n\n`;
206
+ installFetch(body);
207
+ const { appendEntry, handlers } = createHarness();
208
+ await startSession(handlers);
209
+
210
+ const response = await globalThis.fetch("https://gateway.example/v1/chat/completions");
211
+ await response.text();
212
+ const message = assistantMessage("unrelated-message");
213
+ expect(await handlers.get("message_end")!({ message }, {})).toBeUndefined();
214
+ expect(message.usage.cost.total).toBe(0.75);
215
+ expect(appendEntry).not.toHaveBeenCalled();
216
+ });
217
+
218
+ it("falls back when the process cache sees a duplicate response id", async () => {
219
+ const responseId = "live-duplicate-response-4";
220
+ installFetch(`data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\n`);
221
+ const { appendEntry, handlers } = createHarness();
222
+ await startSession(handlers);
223
+
224
+ const first = await globalThis.fetch("https://gateway.example/v1/chat/completions");
225
+ const second = await globalThis.fetch("https://gateway.example/v1/chat/completions");
226
+ await Promise.all([first.text(), second.text()]);
227
+ const message = assistantMessage(responseId);
228
+ expect(await handlers.get("message_end")!({ message }, {})).toBeUndefined();
229
+ expect(message.usage.cost.total).toBe(0.75);
230
+ expect(appendEntry).not.toHaveBeenCalled();
231
+ });
232
+
233
+ it("preserves the original message when no valid billed cost was captured", async () => {
234
+ const responseId = "live-no-cost-3";
235
+ installFetch(`data: {"id":"${responseId}","usage":{"cost":-1}}\n\n`);
236
+ const { appendEntry, handlers } = createHarness();
237
+ await startSession(handlers);
238
+
239
+ const response = await globalThis.fetch("https://gateway.example/v1/chat/completions");
240
+ await response.text();
241
+ const message = assistantMessage(responseId);
242
+ expect(await handlers.get("message_end")!({ message }, {})).toBeUndefined();
243
+ expect(message.usage.cost.total).toBe(0.75);
244
+ expect(appendEntry).not.toHaveBeenCalled();
245
+ });
246
+
247
+ it("preserves terminal error messages even when a billed cost was captured", async () => {
248
+ const responseId = "live-terminal-4";
249
+ installFetch(`data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\n`);
250
+ const { appendEntry, handlers } = createHarness();
251
+ await startSession(handlers);
252
+
253
+ const response = await globalThis.fetch("https://gateway.example/v1/chat/completions");
254
+ await response.text();
255
+ const message = { ...assistantMessage(responseId), stopReason: "error" };
256
+ expect(await handlers.get("message_end")!({ message }, {})).toBeUndefined();
257
+ expect(message.usage.cost.total).toBe(0.75);
258
+ expect(appendEntry).not.toHaveBeenCalled();
259
+ });
260
+
261
+ it("does not capture unrelated fetches", async () => {
262
+ const responseId = "unrelated-fetch-5";
263
+ const body = `data: {"id":"${responseId}","usage":{"cost":0.0123}}\n\n`;
264
+ installFetch(body);
265
+ const { handlers } = createHarness();
266
+ await startSession(handlers);
267
+
268
+ const response = await globalThis.fetch("https://example.com/data.json");
269
+ expect(await response.text()).toBe(body);
270
+ expect(await handlers.get("message_end")!({ message: assistantMessage(responseId) }, {})).toBeUndefined();
271
+ });
272
+ });