gentle-pi 3.1.1 → 3.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/assets/orchestrator-delegation.md +18 -11
  2. package/assets/orchestrator.md +2 -2
  3. package/docs/gentle-shell.md +15 -5
  4. package/docs/readme-reference.md +43 -18
  5. package/docs/review-integration.md +15 -6
  6. package/extensions/gentle-agents.ts +22 -1
  7. package/extensions/gentle-ai.ts +103 -143
  8. package/extensions/gentle-shell.ts +29 -7
  9. package/lib/background-subagents-policy.ts +148 -0
  10. package/lib/model-routing-authority.ts +1 -1
  11. package/lib/native-review-cli.ts +18 -0
  12. package/lib/opaque-pi-reviewer-adapter.ts +130 -10
  13. package/lib/review-candidate-view.ts +27 -6
  14. package/lib/review-host-relay.ts +95 -2
  15. package/lib/shell-bar.ts +63 -9
  16. package/lib/shell-usage.ts +226 -10
  17. package/package.json +2 -1
  18. package/runtime/native-review-cli.mjs +18 -0
  19. package/scripts/gentle-ai-installer.mjs +10 -10
  20. package/scripts/mirror-odd-routing.mjs +242 -0
  21. package/scripts/verify-package-files.mjs +3 -3
  22. package/skills/chained-pr/SKILL.md +2 -1
  23. package/skills/work-unit-commits/SKILL.md +9 -0
  24. package/tests/background-subagents-default-mode.test.ts +105 -0
  25. package/tests/gentle-agents.test.ts +14 -1
  26. package/tests/gentle-ai-binary.test.ts +1 -1
  27. package/tests/gentle-ai-installer.test.ts +47 -47
  28. package/tests/gentle-shell.test.ts +109 -3
  29. package/tests/native-review-capability-contract.test.ts +27 -1
  30. package/tests/odd-routing-canonical-ratchet.test.ts +293 -0
  31. package/tests/odd-routing-contract.test.ts +101 -2
  32. package/tests/opaque-pi-reviewer-adapter.test.ts +153 -9
  33. package/tests/package-manifest.test.ts +6 -6
  34. package/tests/review-base-ref-hint.test.ts +39 -0
  35. package/tests/review-candidate-view.test.ts +55 -0
  36. package/tests/review-controller-native-routing.test.ts +60 -1
  37. package/tests/review-host-relay.test.ts +83 -14
  38. package/tests/review-relay-transport-agent.test.ts +86 -1
  39. package/tests/shell-bar.test.ts +153 -3
  40. package/tests/shell-usage-view.test.ts +3 -2
  41. package/tests/shell-usage.test.ts +254 -6
@@ -2,9 +2,10 @@ import assert from "node:assert/strict";
2
2
  import { execFileSync } from "node:child_process";
3
3
  import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
4
4
  import { tmpdir } from "node:os";
5
- import { join } from "node:path";
5
+ import { delimiter as pathDelimiter, join } from "node:path";
6
6
  import test from "node:test";
7
7
  import { __testing } from "../extensions/gentle-ai.ts";
8
+ import { REVIEW_HOST_RELAY_FAILURE, ReviewHostRelayError } from "../lib/review-host-relay.ts";
8
9
  import { NativeReviewIntegrationError, type NativeReviewCli } from "../lib/native-review-cli.ts";
9
10
  import { CandidateViewRegistry } from "../lib/review-candidate-view.ts";
10
11
  import type { ReviewCollectInputV3, ReviewStatusV3 } from "../lib/review-integration-v2.ts";
@@ -186,6 +187,90 @@ test("the negotiated status asks for the pi agent so the provider offers its mat
186
187
  assert.equal(hostRelay.transport, "pi_host_relay");
187
188
  });
188
189
 
190
+ // gentle-shell#1136 / #1158: the lens's user-owned reviewer selection (agent
191
+ // model routing config) and the extension allowlist environment ride the relay
192
+ // request; without them the child runs the ambient default with no auth
193
+ // adapters. With neither configured the launch stays selection-free.
194
+ test("capture forwards the lens's user-owned reviewer selection and extension allowlist to the relay request", async (t) => {
195
+ t.after(() => __testing.setReviewHostRelayRunnerForTesting());
196
+ const configHome = mkdtempSync(join(tmpdir(), "gentle-pi-relay-config-"));
197
+ const cwd = repository(t);
198
+ t.after(() => rmSync(configHome, { recursive: true, force: true }));
199
+ writeFileSync(join(configHome, "models.json"), JSON.stringify({ "review-reliability": { model: "minimax/MiniMax-M3" } }), "utf8");
200
+ const adapter = join(configHome, "auth-adapter.ts");
201
+ const second = join(configHome, "second-adapter.ts");
202
+ writeFileSync(adapter, "export default () => {};");
203
+ writeFileSync(second, "export default () => {};");
204
+ const previousConfigHome = process.env.GENTLE_PI_CONFIG_HOME;
205
+ const previousExtensions = process.env.GENTLE_PI_REVIEW_RELAY_EXTENSIONS;
206
+ process.env.GENTLE_PI_CONFIG_HOME = configHome;
207
+ process.env.GENTLE_PI_REVIEW_RELAY_EXTENSIONS = [adapter, second].join(pathDelimiter);
208
+ t.after(() => {
209
+ if (previousConfigHome === undefined) delete process.env.GENTLE_PI_CONFIG_HOME;
210
+ else process.env.GENTLE_PI_CONFIG_HOME = previousConfigHome;
211
+ if (previousExtensions === undefined) delete process.env.GENTLE_PI_REVIEW_RELAY_EXTENSIONS;
212
+ else process.env.GENTLE_PI_REVIEW_RELAY_EXTENSIONS = previousExtensions;
213
+ });
214
+ const { native } = transportAwareNative();
215
+ const relayed: ReviewHostRelayRequest[] = [];
216
+ __testing.setReviewHostRelayRunnerForTesting(async (request: ReviewHostRelayRequest) => {
217
+ relayed.push(request);
218
+ return { promptByteLength: 128, resultByteLength: 64, submission: '{"admission_decision":"completed"}' };
219
+ });
220
+
221
+ await runCapture(cwd, native, "selection-lineage");
222
+ assert.equal(relayed.length, 1);
223
+ assert.equal(relayed[0]!.reviewerModel, "minimax/MiniMax-M3", "the lens's routing entry must name the child's selection");
224
+ assert.deepEqual(relayed[0]!.reviewerExtensionPaths, [adapter, second]);
225
+ });
226
+
227
+ test("capture keeps the relay launch selection-free when the user configured neither a selection nor extensions", async (t) => {
228
+ t.after(() => __testing.setReviewHostRelayRunnerForTesting());
229
+ const configHome = mkdtempSync(join(tmpdir(), "gentle-pi-relay-config-empty-"));
230
+ const cwd = repository(t);
231
+ t.after(() => rmSync(configHome, { recursive: true, force: true }));
232
+ const previousConfigHome = process.env.GENTLE_PI_CONFIG_HOME;
233
+ const previousExtensions = process.env.GENTLE_PI_REVIEW_RELAY_EXTENSIONS;
234
+ process.env.GENTLE_PI_CONFIG_HOME = configHome;
235
+ delete process.env.GENTLE_PI_REVIEW_RELAY_EXTENSIONS;
236
+ t.after(() => {
237
+ if (previousConfigHome === undefined) delete process.env.GENTLE_PI_CONFIG_HOME;
238
+ else process.env.GENTLE_PI_CONFIG_HOME = previousConfigHome;
239
+ if (previousExtensions === undefined) delete process.env.GENTLE_PI_REVIEW_RELAY_EXTENSIONS;
240
+ else process.env.GENTLE_PI_REVIEW_RELAY_EXTENSIONS = previousExtensions;
241
+ });
242
+ const { native } = transportAwareNative();
243
+ const relayed: ReviewHostRelayRequest[] = [];
244
+ __testing.setReviewHostRelayRunnerForTesting(async (request: ReviewHostRelayRequest) => {
245
+ relayed.push(request);
246
+ return { promptByteLength: 128, resultByteLength: 64, submission: '{"admission_decision":"completed"}' };
247
+ });
248
+
249
+ await runCapture(cwd, native, "default-lineage");
250
+ assert.equal(relayed.length, 1);
251
+ assert.equal(relayed[0]!.reviewerModel, undefined);
252
+ assert.equal(relayed[0]!.reviewerExtensionPaths, undefined);
253
+ });
254
+
255
+ test("a relayed empty-output failure carries the child's own evidence in the failure report", async (t) => {
256
+ t.after(() => __testing.setReviewHostRelayRunnerForTesting());
257
+ const cwd = repository(t);
258
+ const { native } = transportAwareNative();
259
+ __testing.setReviewHostRelayRunnerForTesting(async () => {
260
+ throw new ReviewHostRelayError(REVIEW_HOST_RELAY_FAILURE.PI_EMPTY_OUTPUT, "pi", "pi subprocess produced no assistant text (stdout kind: no-assistant-text; a tool call was attempted)", {
261
+ reviewerEvidence: { stdoutKind: "no-assistant-text", reviewerModel: "nan/deepseek-v4-flash", toolCallAttempted: true },
262
+ });
263
+ });
264
+
265
+ const result = await runCapture(cwd, native, "evidence-lineage");
266
+ assert.equal(result.outcome, "pi-host-relay-transport-failure");
267
+ const failure = result.failure as { reviewer?: { stdoutKind?: string; reviewerModel?: string; toolCallAttempted?: boolean } } | undefined;
268
+ assert.ok(failure !== undefined, "the envelope carries the failure report");
269
+ assert.equal(failure.reviewer?.stdoutKind, "no-assistant-text");
270
+ assert.equal(failure.reviewer?.reviewerModel, "nan/deepseek-v4-flash");
271
+ assert.equal(failure.reviewer?.toolCallAttempted, true);
272
+ });
273
+
189
274
  test("capture forecasts the reviewer model run once and spends nothing until it is acknowledged", async (t) => {
190
275
  t.after(() => __testing.setReviewHostRelayRunnerForTesting());
191
276
  const cwd = repository(t);
@@ -12,6 +12,7 @@ import {
12
12
  type ShellBarModel,
13
13
  type ShellBarTheme,
14
14
  } from "../lib/shell-bar.ts";
15
+ import { parseNanQuota } from "../lib/shell-usage.ts";
15
16
 
16
17
  // The Gentle Shell bar replaces pi's three-line footer with one line of
17
18
  // segments. Rendering is pure so it can be verified without a TUI.
@@ -52,6 +53,28 @@ function model(overrides: Partial<ShellBarModel> = {}): ShellBarModel {
52
53
  };
53
54
  }
54
55
 
56
+ // The grouped NaN fixture the panel test uses, so both surfaces are asserted
57
+ // against the same payload, the same order and the same percentages.
58
+ const GROUPED_NAN_QUOTA = {
59
+ periodEnd: "2026-10-01T00:00:00.000Z",
60
+ models: [
61
+ { model: "glm5.3-flash", cap: 2_000_000_000, tokensUsed: 200_000_000 },
62
+ { model: "glm5.3", cap: 3_000_000_000, tokensUsed: 0, periodEnd: "2026-10-17T05:53:20.000Z" },
63
+ { model: "glm5.2", cap: 3_000_000_000, tokensUsed: 0, periodEnd: "2026-10-17T05:53:20.000Z" },
64
+ { model: "deepseek-v4-flash", cap: 3_000_000_000, tokensUsed: 300_000_000 },
65
+ ],
66
+ };
67
+
68
+ // The sidebar prints the card body between its borders; the Usage rows are the
69
+ // metered block between the Usage heading and Integrations, minus the context
70
+ // meter, which is not a subscription allowance.
71
+ function sidebarUsageRows(lines: string[]): string[] {
72
+ const body = lines.filter((line) => line.startsWith("│ ")).map((line) => line.slice(2, -2).trim());
73
+ const start = body.indexOf("Usage");
74
+ const end = body.indexOf("Integrations");
75
+ return body.slice(start + 1, end).filter((line) => /[▰▱]/.test(line) && !line.startsWith("Context"));
76
+ }
77
+
55
78
  test("renderGauge fills cells proportionally to the percentage", () => {
56
79
  assert.equal(renderGauge(45, 8), "▰▰▰▰▱▱▱▱");
57
80
  assert.equal(renderGauge(0, 8), "▱▱▱▱▱▱▱▱");
@@ -81,13 +104,13 @@ test("renderShellBar renders one line with the segments in order", () => {
81
104
  assert.equal(rest.length, 0);
82
105
  assert.equal(
83
106
  line,
84
- "✿ gentle-pi ⟡ ~/work/gentle-pi main ⟡ gpt-5.5 · medium ⟡ ctx ▰▰▰▰▱▱▱▱ 45% ⟡ $9.49 sub",
107
+ "✿ gentle shell ⟡ ~/work/gentle-pi main ⟡ gpt-5.5 · medium ⟡ ctx ▰▰▰▰▱▱▱▱ 45% ⟡ $9.49 sub",
85
108
  );
86
109
  });
87
110
 
88
111
  test("renderShellBar colors the brand, model, effort, and gauge by role", () => {
89
112
  const [line] = renderShellBar(model(), taggedTheme, 400);
90
- assert.match(line, /<accent>✿ gentle-pi<\/accent>/);
113
+ assert.match(line, /<accent>✿ gentle shell<\/accent>/);
91
114
  assert.match(line, /<text>gpt-5\.5<\/text>/);
92
115
  assert.match(line, /<syntaxFunction>medium<\/syntaxFunction>/);
93
116
  assert.match(line, /<accent>▰▰▰▰<\/accent><border>▱▱▱▱<\/border>/);
@@ -120,6 +143,133 @@ test("renderShellBar adds the subscription windows after the cost when usage is
120
143
  assert.match(line, /\$9\.49 sub ⟡ codex 5h ▰▰▰▰▰▱▱▱ 62% · week 31%$/);
121
144
  });
122
145
 
146
+ test("renderShellBar meters the model the session is using inside a multi-model provider", () => {
147
+ const usage = {
148
+ provider: "nan",
149
+ plan: undefined,
150
+ fetchedAt: 0,
151
+ limits: [
152
+ { name: "deepseek-v4-flash", limitReached: false, windows: [{ label: "", usedPercent: 18, windowSeconds: 0, resetAt: null, used: 545_000_000, budget: 3_000_000_000 }] },
153
+ { name: "glm5.3-flash", limitReached: false, windows: [{ label: "", usedPercent: 10, windowSeconds: 0, resetAt: null, used: 200_000_000, budget: 2_000_000_000 }] },
154
+ ],
155
+ };
156
+ const [glm] = renderShellBar(model({ modelId: "glm5.3-flash", usage }), plainTheme, 200);
157
+ assert.match(glm, /glm5\.3-flash ▰▱▱▱▱▱▱▱ 10%$/);
158
+ assert.doesNotMatch(glm, /deepseek-v4-flash ▰/);
159
+ const [other] = renderShellBar(model({ modelId: "deepseek-v4-flash", usage }), plainTheme, 200);
160
+ assert.match(other, /deepseek-v4-flash ▰▱▱▱▱▱▱▱ 18%$/);
161
+ const sidebar = renderShellSidebarBar(model({ modelId: "glm5.3-flash", usage }), plainTheme, 60).join("\n");
162
+ assert.match(sidebar, /glm5\.3-flash +▰▱▱▱▱▱▱▱ +10%/, "the sidebar lists the session model's own allowance");
163
+ });
164
+
165
+ // The sidebar is the surface that never needs opening, so a provider with
166
+ // per-model allowances shows the panel's model rows there too — most consumed
167
+ // family first, its models inside it — and leaves the aggregate totals and the
168
+ // reset dates to the bar and the panel.
169
+ test("sidebar groups the NaN allowances by subscription without totals or resets", () => {
170
+ const usage = parseNanQuota(GROUPED_NAN_QUOTA, 0);
171
+ const lines = renderShellSidebarBar(model({ modelId: "glm5.3-flash", usage }), plainTheme, 60);
172
+ const rows = sidebarUsageRows(lines);
173
+ assert.deepEqual(rows.map((row) => row.replace(/\s*[▰▱].*$/, "")), ["deepseek-v4-flash", "glm5.3-flash"]);
174
+ assert.deepEqual(rows.map((row) => Number.parseInt(row.match(/(\d+)%$/)![1] ?? "", 10)), [10, 10]);
175
+ assert.equal(rows.some((row) => / 0%$/.test(row)), false, "a window that consumed nothing only spends space");
176
+ assert.equal(lines.join("\n").includes("total"), false, "an aggregate row only costs space");
177
+ assert.equal(lines.join("\n").includes("resets in"), false, "the reset dates belong to the panel");
178
+ for (const width of [24, 46, 60]) {
179
+ assert.ok(renderShellSidebarBar(model({ usage }), plainTheme, width).every((line) => visibleWidth(line) <= width));
180
+ }
181
+ });
182
+
183
+ test("sidebar treats one metered NaN allowance as a per-model provider", () => {
184
+ const usage = parseNanQuota({
185
+ periodEnd: "2026-10-01T00:00:00.000Z",
186
+ models: [{ model: "glm5.3-flash", cap: 2_000_000_000, tokensUsed: 400_000_000, windowHours: 4, windowTokens: 400_000_000, windowTokensUsed: 120_000_000, windowResetsAt: 1_788_620_161 }],
187
+ }, 0);
188
+ const lines = renderShellSidebarBar(model({ modelId: "glm5.3-flash", usage }), plainTheme, 60);
189
+ const rows = sidebarUsageRows(lines).map((row) => row.replace(/\s+/g, " "));
190
+ // A single allowance is still the panel's rows, not the bar's one-line meter:
191
+ // the rolling window it reports is a row of its own here too, and the reset
192
+ // stays in the panel.
193
+ assert.deepEqual(rows, ["glm5.3-flash ▰▰▱▱▱▱▱▱ 20%", "glm5.3-flash 4h ▰▰▱▱▱▱▱▱ 30%"]);
194
+ assert.equal(lines.join("\n").includes("resets in"), false);
195
+ });
196
+
197
+ test("sidebar keeps one aggregate line for a provider without raw allowances", () => {
198
+ const usage = {
199
+ provider: "openai-codex",
200
+ plan: "pro",
201
+ fetchedAt: 0,
202
+ limits: [{ name: "codex", limitReached: false, windows: [
203
+ { label: "5h", usedPercent: 62, windowSeconds: 18_000, resetAt: null },
204
+ { label: "week", usedPercent: 31, windowSeconds: 604_800, resetAt: null },
205
+ ] }],
206
+ };
207
+ const lines = renderShellSidebarBar(model({ usage }), plainTheme, 60);
208
+ assert.deepEqual(sidebarUsageRows(lines), ["codex 5h ▰▰▰▰▰▱▱▱ 62% · week 31%"]);
209
+ });
210
+
211
+ test("sidebar drops an aggregate allowance that consumed nothing", () => {
212
+ const usage = {
213
+ provider: "openai-codex",
214
+ plan: "pro",
215
+ fetchedAt: 0,
216
+ limits: [{ name: "codex", limitReached: false, windows: [
217
+ { label: "5h", usedPercent: 0, windowSeconds: 18_000, resetAt: null },
218
+ { label: "week", usedPercent: 0, windowSeconds: 604_800, resetAt: null },
219
+ ] }],
220
+ };
221
+ const lines = renderShellSidebarBar(model({ usage }), plainTheme, 60);
222
+ assert.deepEqual(sidebarUsageRows(lines), []);
223
+ const text = lines.join("\n");
224
+ assert.match(text, /Usage/);
225
+ assert.match(text, /Context/);
226
+ assert.match(text, /Cost/);
227
+ // The bar keeps its own contract: only the sidebar gives zero rows the boot.
228
+ assert.match(renderShellBar(model({ usage }), plainTheme, 200)[0], /codex 5h ▱▱▱▱▱▱▱▱ 0% · week 0%$/);
229
+ });
230
+
231
+ test("sidebar keeps the Usage group when nothing was consumed", () => {
232
+ const usage = parseNanQuota({
233
+ periodEnd: "2026-10-01T00:00:00.000Z",
234
+ models: [
235
+ { model: "glm5.3", cap: 3_000_000_000, tokensUsed: 0 },
236
+ { model: "glm5.2", cap: 3_000_000_000, tokensUsed: 0 },
237
+ ],
238
+ }, 0);
239
+ const lines = renderShellSidebarBar(model({ usage }), plainTheme, 60);
240
+ const text = lines.join("\n");
241
+ assert.deepEqual(sidebarUsageRows(lines), []);
242
+ assert.match(text, /Usage/);
243
+ assert.match(text, /Context/);
244
+ assert.match(text, /Cost/);
245
+ assert.match(text, /Integrations/);
246
+ });
247
+
248
+ test("sidebar hides exactly the rows that would print 0%, rounding included", () => {
249
+ // The row's own rounding is the only threshold: a fraction below half a
250
+ // percent is a 0% row and leaves; half a percent keeps its row and prints 1%.
251
+ const usage = (usedPercent: number) => ({
252
+ provider: "openai-codex",
253
+ plan: "pro",
254
+ fetchedAt: 0,
255
+ limits: [{ name: "codex", limitReached: false, windows: [{ label: "5h", usedPercent, windowSeconds: 18_000, resetAt: null }] }],
256
+ });
257
+ assert.deepEqual(sidebarUsageRows(renderShellSidebarBar(model({ usage: usage(0.4) }), plainTheme, 60)), []);
258
+ assert.deepEqual(sidebarUsageRows(renderShellSidebarBar(model({ usage: usage(0.5) }), plainTheme, 60)), ["codex 5h ▱▱▱▱▱▱▱▱ 1%"]);
259
+ });
260
+
261
+ test("sidebar reads a limit without windows as nothing to draw, not as zero consumption", () => {
262
+ const usage = {
263
+ provider: "openai-codex",
264
+ plan: "pro",
265
+ fetchedAt: 0,
266
+ limits: [{ name: "codex", limitReached: false, windows: [] }],
267
+ };
268
+ const lines = renderShellSidebarBar(model({ usage }), plainTheme, 60);
269
+ assert.deepEqual(sidebarUsageRows(lines), []);
270
+ assert.match(lines.join("\n"), /Cost/);
271
+ });
272
+
123
273
  test("renderShellBar shows an unknown context as a question mark after compaction", () => {
124
274
  const [line] = renderShellBar(model({ contextPercent: null }), plainTheme, 160);
125
275
  assert.match(line, /ctx ▱▱▱▱▱▱▱▱ \?%/);
@@ -162,7 +312,7 @@ test("renderShellBar drops the session name, then trailing segments, before trun
162
312
 
163
313
  const [atFifty] = renderShellBar(wide, plainTheme, 50);
164
314
  assert.ok(visibleWidth(atFifty) <= 50, `line overflowed: ${visibleWidth(atFifty)}`);
165
- assert.match(atFifty, /^✿ gentle-pi/);
315
+ assert.match(atFifty, /^✿ gentle shell/);
166
316
  });
167
317
 
168
318
  test("shellEnabled stays off inside a Gentle Agents child", () => {
@@ -34,7 +34,8 @@ test("UsageView frames the panel, keeps every line at width, and shows the empty
34
34
  for (const line of lines) assert.equal(visibleWidth(line), 90, `"${stripAnsi(line)}" is not 90 wide`);
35
35
  const plain = lines.map(stripAnsi);
36
36
  assert.match(plain[1], /^│ openai-codex · pro · updated just now +│$/);
37
- assert.match(plain[3], /^│ {5}week +▰+▱+ +40% +resets in 2h 0m +│$/);
37
+ assert.match(plain[2], /^│ {3}codex week +[▰▱]{16} +40% · resets in 2h 0m +│$/);
38
+ assert.match(plain[3], /r refresh .* esc close/);
38
39
  });
39
40
 
40
41
  test("UsageView refetches on r and closes on escape or q", async () => {
@@ -54,7 +55,7 @@ test("UsageView refetches on r and closes on escape or q", async () => {
54
55
  view.handleInput("r");
55
56
  await new Promise((resolve) => setTimeout(resolve, 0));
56
57
  assert.deepEqual(events, ["render", "refresh", "render"]);
57
- assert.match(stripAnsi(view.render(90)[3]), /55%/);
58
+ assert.match(stripAnsi(view.render(90)[2]), /55%/);
58
59
  assert.match(stripAnsi(view.render(90)[1]), /^│ ✿ openai-codex · pro · updated just now/);
59
60
  view.handleInput("\x1b");
60
61
  view.handleInput("q");
@@ -6,13 +6,17 @@ import {
6
6
  formatReset,
7
7
  parseAnthropicHeaders,
8
8
  parseCodexHeaders,
9
+ parseNanQuota,
9
10
  parseUsageHeaders,
10
11
  parseCodexUsage,
12
+ providerNote,
11
13
  renderUsageBar,
12
14
  renderUsagePanel,
15
+ SUPPORTED_USAGE_PROVIDERS,
13
16
  UsageStore,
14
17
  windowLabel,
15
18
  type ProviderUsage,
19
+ type UsageWindow,
16
20
  } from "../lib/shell-usage.ts";
17
21
 
18
22
  // Subscription usage: what each connected provider says about its windows.
@@ -134,11 +138,11 @@ test("renderUsagePanel lists each provider with meters, resets, and a stale mark
134
138
  const lines = renderUsagePanel([usage], plainTheme, 70, NOW + 3 * 60_000);
135
139
  for (const line of lines) assert.ok(visibleWidth(line) <= 70, `too wide: ${line}`);
136
140
  assert.match(lines[0], /^openai-codex · pro · updated 3m ago$/);
137
- assert.match(lines[1], /^ {2}codex$/);
138
- assert.match(lines[2], /^ {4}week +▰+▱+ +40% +resets in 2d 1h$/);
139
- assert.match(lines[3], /^ {2}codex_spark$/);
140
- assert.match(lines[4], /^ {4}5h /);
141
- assert.match(lines[5], /^ {4}week /);
141
+ // One row per window: the name, its meter, its percentage and, when the window
142
+ // reports one, its reset, all on the same line.
143
+ assert.match(lines[1], /^ {2}codex week +[▰▱]{16} +40% · resets in 2d 1h$/);
144
+ assert.match(lines[2], /^ {2}codex_spark 5h +[▰▱]{16} +12% · resets in \d+h \d+m$/);
145
+ assert.match(lines[3], /^ {2}codex_spark week +[▰▱]{16} +3% · resets in \d+d \d+h$/);
142
146
  assert.deepEqual(renderUsagePanel([], plainTheme, 120, NOW), ["No subscription usage yet. Usage arrives with the next response, or press r to fetch it."]);
143
147
  });
144
148
 
@@ -148,7 +152,7 @@ test("renderUsagePanel puts the active provider first and explains missing data"
148
152
  assert.ok(claude);
149
153
  const both = renderUsagePanel([codex, claude], plainTheme, 100, NOW, { provider: "anthropic" });
150
154
  assert.match(both[0], /^✿ anthropic · updated just now$/);
151
- assert.match(both[1], /^ {2}claude$/);
155
+ assert.match(both[1], /^ {2}claude 5h +[▰▱]{16} +20%$/);
152
156
  assert.match(both.find((line) => line.startsWith("openai-codex")) ?? "", /^openai-codex · pro/);
153
157
 
154
158
  const apiKey = renderUsagePanel([codex], plainTheme, 100, NOW, { provider: "openai" });
@@ -160,6 +164,250 @@ test("renderUsagePanel puts the active provider first and explains missing data"
160
164
  assert.deepEqual(renderUsagePanel([], plainTheme, 100, NOW, { provider: "openai-codex" }), ["✿ openai-codex · no usage yet · r to fetch"]);
161
165
  });
162
166
 
167
+ // NaN Cloud quota: per-model allowances for the billing period, plus the
168
+ // rolling window the model reports. Percentages are tokensUsed over cap, the
169
+ // same ratio the dashboard draws. The payload shape was read off the official
170
+ // dashboard bundle, not a published schema, so every field stays optional.
171
+ const NAN_QUOTA = {
172
+ periodEnd: "2026-10-01T00:00:00.000Z",
173
+ models: [
174
+ {
175
+ model: "glm5.3",
176
+ cap: 3_000_000_000,
177
+ fullCap: 3_000_000_000,
178
+ tokensUsed: 820_000_000,
179
+ windowHours: 4,
180
+ windowTokens: 400_000_000,
181
+ windowTokensUsed: 120_000_000,
182
+ windowResetsAt: 1_788_620_161,
183
+ email: "someone@example.com",
184
+ },
185
+ { model: "deepseek-v4-flash", cap: 1_500_000_000, tokensUsed: 150_000_000 },
186
+ { model: "qwen3.8-flash", cap: 0, tokensUsed: 10 },
187
+ ],
188
+ };
189
+
190
+ // The billing-period window carries no label: the model id in front of it names
191
+ // the allowance, and the reset text says what the window is. Only a sub-window
192
+ // on top (a rolling `4h`) needs a name, so `windowText` spells the unlabeled one
193
+ // out for the assertions below.
194
+ function windowText(window: UsageWindow): string {
195
+ return `${window.label === "" ? "period" : window.label}:${window.usedPercent}`;
196
+ }
197
+
198
+ test("parseNanQuota maps each model allowance and its rolling window", () => {
199
+ const usage = parseNanQuota(NAN_QUOTA, NOW);
200
+ assert.equal(usage.provider, "nan");
201
+ assert.equal(usage.plan, undefined);
202
+ assert.equal(usage.fetchedAt, NOW);
203
+ assert.deepEqual(usage.limits.map((limit) => limit.name), ["glm5.3", "deepseek-v4-flash"]);
204
+
205
+ const [glm, deepseek] = usage.limits;
206
+ assert.deepEqual(glm.windows.map((window) => window.label), ["", "4h"]);
207
+ assert.equal(glm.windows[0].usedPercent, (820_000_000 / 3_000_000_000) * 100);
208
+ assert.equal(glm.windows[0].windowSeconds, 2_212_800);
209
+ assert.equal(glm.windows[0].resetAt, 1_790_812_800_000);
210
+ assert.equal(glm.windows[1].usedPercent, 30);
211
+ assert.equal(glm.windows[1].windowSeconds, 14_400);
212
+ assert.equal(glm.windows[1].resetAt, 1_788_620_161_000);
213
+ assert.equal(glm.limitReached, false);
214
+ assert.deepEqual(deepseek.windows.map(windowText), ["period:10"]);
215
+ assert.equal(JSON.stringify(usage).includes("example.com"), false, "the quota parser must not keep unrelated account fields");
216
+ });
217
+
218
+ test("parseNanQuota falls back to the top-level period end and defaults the window budget", () => {
219
+ const topLevel = parseNanQuota({ periodEnd: 1_790_812_800, models: [{ model: "glm5.3", cap: 3_000_000_000, tokensUsed: 0 }] }, NOW);
220
+ assert.equal(topLevel.limits[0].windows[0].resetAt, 1_790_812_800_000);
221
+
222
+ const defaulted = parseNanQuota({ models: [{ model: "glm5.3", cap: 3_000_000_000, tokensUsed: 0, windowTokensUsed: 100_000_000 }] }, NOW);
223
+ assert.deepEqual(defaulted.limits[0].windows.map(windowText), ["period:0", "4h:25"]);
224
+
225
+ const overCap = parseNanQuota({ models: [{ model: "glm5.3", cap: 3_000_000_000, tokensUsed: 3_000_000_000, windowHours: 12, windowTokensUsed: 60_000_000 }] }, NOW);
226
+ assert.deepEqual(overCap.limits[0].windows.map(windowText), ["period:100", "12h:15"]);
227
+ assert.equal(overCap.limits[0].limitReached, true);
228
+ });
229
+
230
+ test("parseNanQuota degrades to no data instead of throwing", () => {
231
+ assert.deepEqual(parseNanQuota({}, NOW).limits, []);
232
+ assert.deepEqual(parseNanQuota(undefined, NOW).limits, []);
233
+ assert.deepEqual(parseNanQuota({ models: "nope" }, NOW).limits, []);
234
+ assert.deepEqual(parseNanQuota({ models: [null, "glm5.3", 7] }, NOW).limits, []);
235
+ assert.deepEqual(parseNanQuota({ models: [{ model: "glm5.3", cap: "3000000000", tokensUsed: 1 }] }, NOW).limits, []);
236
+ assert.deepEqual(parseNanQuota({ models: [{ model: "glm5.3", cap: 3_000_000_000 }] }, NOW).limits, []);
237
+ assert.deepEqual(parseNanQuota({ models: [{ model: "", cap: 3_000_000_000, tokensUsed: 1 }] }, NOW).limits, []);
238
+ });
239
+
240
+ test("nan is a supported usage provider with its own pending note", () => {
241
+ assert.ok(SUPPORTED_USAGE_PROVIDERS.includes("nan"));
242
+ assert.equal(providerNote("nan"), "no usage yet · r to fetch");
243
+ assert.deepEqual(renderUsagePanel([], plainTheme, 100, NOW, { provider: "nan" }), ["✿ nan · no usage yet · r to fetch"]);
244
+ });
245
+
246
+ test("renderUsagePanel lists each NaN model allowance with its reset on the same row", () => {
247
+ const usage = parseNanQuota(NAN_QUOTA, NOW);
248
+ const lines = renderUsagePanel([usage], plainTheme, 80, NOW, { provider: "nan" });
249
+ for (const line of lines) assert.ok(visibleWidth(line) <= 80, `too wide: ${line}`);
250
+ assert.match(lines[0], /^✿ nan · updated just now$/);
251
+ assert.match(lines[1], /^ {2}glm5\.3 +[▰▱]{16} +27% · resets in \d+d \d+h$/);
252
+ // The rolling window a model reports is a row of its own, with its own reset.
253
+ assert.match(lines[2], /^ {2}glm5\.3 4h +[▰▱]{16} +30% · resets in \d+h \d+m$/);
254
+ assert.match(lines[3], /^ {2}deepseek-v4-flash +[▰▱]{16} +10% · resets in \d+d \d+h$/);
255
+ assert.equal(lines.some((line) => line.includes("total")), false, "an aggregate row only costs space");
256
+ });
257
+
258
+ // The server picks the order of the per-model allowances, so drawing the first
259
+ // one showed DeepSeek's meter inside a GLM session. The bar follows the model
260
+ // the session actually uses, and falls back to the account total when that
261
+ // model holds no allowance of its own.
262
+ test("renderUsageBar prefers the active model allowance over the payload order", () => {
263
+ const usage = parseNanQuota(NAN_QUOTA, NOW);
264
+ assert.equal(renderUsageBar(usage, plainTheme, "deepseek-v4-flash"), "deepseek-v4-flash ▰▱▱▱▱▱▱▱ 10%");
265
+ assert.equal(renderUsageBar(usage, plainTheme, "glm5.3"), "glm5.3 ▰▰▱▱▱▱▱▱ 27% · 4h 30%");
266
+ assert.equal(renderUsageBar(usage, plainTheme), "glm5.3 ▰▰▱▱▱▱▱▱ 27% · 4h 30%", "without an active model the first limit still wins");
267
+ assert.equal(renderUsageBar(usage, plainTheme, "gemma4"), "nan total ▰▰▱▱▱▱▱▱ 22%", "an unmetered model reports the account, never another model");
268
+ assert.equal(renderUsageBar(usage, plainTheme, "qwen3.8-flash"), "nan total ▰▰▱▱▱▱▱▱ 22%", "a model the payload skips holds no allowance either");
269
+ });
270
+
271
+ test("renderUsageBar prefers the active model's family before the account total", () => {
272
+ const usage = parseNanQuota({
273
+ periodEnd: "2026-10-01T00:00:00.000Z",
274
+ models: [
275
+ { model: "glm5.3-flash", cap: 2_000_000_000, tokensUsed: 200_000_000 },
276
+ { model: "deepseek-v4-flash", cap: 4_000_000_000, tokensUsed: 200_000_000 },
277
+ ],
278
+ }, NOW);
279
+ // The session model holds no allowance of its own and its family has exactly
280
+ // one member: the family is still a closer name for the meter than the whole
281
+ // account, so the ladder visits it before the account rung.
282
+ assert.match(renderUsageBar(usage, plainTheme, "glm5.3-turbo") ?? "", /^glm total ▰▱▱▱▱▱▱▱ 10%$/);
283
+ assert.match(renderUsageBar(usage, plainTheme, "gemma4") ?? "", /^nan total ▰▱▱▱▱▱▱▱ 7%$/, "no member of the family means the account is the only honest name");
284
+ });
285
+
286
+ test("a single metered allowance still takes the family and account names", () => {
287
+ const usage = parseNanQuota({
288
+ periodEnd: "2026-10-01T00:00:00.000Z",
289
+ models: [
290
+ { model: "glm5.3-flash", cap: 2_000_000_000, tokensUsed: 400_000_000 },
291
+ { model: "gemma4", cap: 0 },
292
+ ],
293
+ }, NOW);
294
+ // One metered model is still a payload that carries raw allowances, so the bar
295
+ // names the allowance the session draws from instead of listing whichever
296
+ // model the payload happened to report.
297
+ assert.deepEqual(usage.limits.map((limit) => limit.name), ["glm5.3-flash"]);
298
+ assert.equal(renderUsageBar(usage, plainTheme, "glm5.3-flash"), "glm5.3-flash ▰▰▱▱▱▱▱▱ 20%");
299
+ assert.match(renderUsageBar(usage, plainTheme, "glm5.4") ?? "", /^glm total ▰▰▱▱▱▱▱▱ 20%$/);
300
+ assert.match(renderUsageBar(usage, plainTheme, "gemma4") ?? "", /^nan total ▰▰▱▱▱▱▱▱ 20%$/);
301
+ });
302
+
303
+ test("renderUsageBar leaves providers without raw allowances on their first limit", () => {
304
+ assert.equal(renderUsageBar(parseCodexUsage(CODEX_PAYLOAD, NOW), plainTheme, "gpt-5.2-codex"), "codex week ▰▰▰▱▱▱▱▱ 40%");
305
+ });
306
+
307
+ test("parseNanQuota weights the period window by the effective allowance", () => {
308
+ const usage = parseNanQuota({
309
+ periodEnd: "2026-10-01T00:00:00.000Z",
310
+ models: [
311
+ { model: "glm5.3-flash", cap: 1_500_000_000, fullCap: 2_000_000_000, tokensUsed: 200_000_000 },
312
+ { model: "glm5.3", cap: 0, fullCap: 3_000_000_000, tokensUsed: 300_000_000 },
313
+ ],
314
+ }, NOW);
315
+ // The dashboard divides by the full-period allowance, not by the prorated cap
316
+ // the current period reports, so the percentages keep matching it.
317
+ assert.deepEqual(usage.limits.map((limit) => limit.name), ["glm5.3-flash", "glm5.3"]);
318
+ const [prorated, noPeriodCap] = usage.limits;
319
+ assert.equal(prorated.windows[0].budget, 2_000_000_000, "fullCap is the denominator when it is positive");
320
+ assert.equal(prorated.windows[0].usedPercent, 10);
321
+ assert.equal(noPeriodCap.windows[0].budget, 3_000_000_000, "a period cap prorated to zero still reports a metered model");
322
+ assert.equal(noPeriodCap.windows[0].usedPercent, 10);
323
+ assert.equal(noPeriodCap.limitReached, false);
324
+ });
325
+
326
+ test("parseNanQuota refuses a payload that hides a metered model's usage", () => {
327
+ // A sibling with a metered allowance whose usage cannot be read is drift, not
328
+ // a model to skip: a partial snapshot would understate every aggregate it
329
+ // feeds, so the read fails whole and the last valid snapshot survives.
330
+ const drifted = parseNanQuota({
331
+ periodEnd: "2026-10-01T00:00:00.000Z",
332
+ models: [
333
+ { model: "glm5.3", cap: 3_000_000_000, tokensUsed: 820_000_000 },
334
+ { model: "deepseek-v4-flash", cap: 2_000_000_000 },
335
+ ],
336
+ }, NOW);
337
+ assert.deepEqual(drifted.limits, []);
338
+ // No allowance at all is not drift: the dashboard draws nothing for these
339
+ // models either, and today's real payload carries them.
340
+ const unmetered = parseNanQuota({
341
+ models: [
342
+ { model: "glm5.3", cap: 3_000_000_000, tokensUsed: 820_000_000 },
343
+ { model: "gemma4", cap: 0 },
344
+ { model: "qwen3.8-flash", tokensUsed: 10 },
345
+ ],
346
+ }, NOW);
347
+ assert.deepEqual(unmetered.limits.map((limit) => limit.name), ["glm5.3"]);
348
+ assert.deepEqual(parseNanQuota({ models: "none" }, NOW).limits, []);
349
+ });
350
+
351
+ test("parseNanQuota keeps the raw numbers the aggregates are weighted by", () => {
352
+ const [glm] = parseNanQuota(NAN_QUOTA, NOW).limits;
353
+ assert.equal(glm.windows[0].used, 820_000_000);
354
+ assert.equal(glm.windows[0].budget, 3_000_000_000);
355
+ const [codex] = parseCodexUsage(CODEX_PAYLOAD, NOW).limits[0].windows;
356
+ assert.equal(codex.used, undefined, "only NaN reports raw allowance numbers, which is what gates the aggregates");
357
+ assert.equal(codex.budget, undefined);
358
+ });
359
+
360
+ // Grouping is a presentation decision: the panel adds the account total and one
361
+ // row per family with more than one metered model, and each of those rows is an
362
+ // ordinary limit block, the shape Codex already uses for its extra limits.
363
+ const GROUPED_NAN_QUOTA = {
364
+ periodEnd: "2026-10-01T00:00:00.000Z",
365
+ models: [
366
+ { model: "glm5.3-flash", cap: 2_000_000_000, tokensUsed: 200_000_000 },
367
+ { model: "glm5.3", cap: 3_000_000_000, tokensUsed: 0, periodEnd: "2026-10-17T05:53:20.000Z" },
368
+ { model: "glm5.2", cap: 3_000_000_000, tokensUsed: 0, periodEnd: "2026-10-17T05:53:20.000Z" },
369
+ { model: "deepseek-v4-flash", cap: 3_000_000_000, tokensUsed: 300_000_000 },
370
+ ],
371
+ };
372
+
373
+ function panelNames(lines: string[]): string[] {
374
+ return lines.filter(isMeterRow).map((line) => line.trim().replace(/\s*[▰▱].*$/, ""));
375
+ }
376
+
377
+ // A meter row carries the row name and its gauge on one line; the reset, when the
378
+ // window reports one, is the line underneath.
379
+ function isMeterRow(line: string): boolean {
380
+ return /[▰▱]/.test(line);
381
+ }
382
+
383
+ // A row carries its own reset on the same line when the window reports one.
384
+ function panelResets(lines: string[]): string[] {
385
+ return lines.filter(isMeterRow).map((line) => /resets in .*$/.exec(line)?.[0] ?? "").filter((reset) => reset.length > 0);
386
+ }
387
+
388
+ function panelPercents(lines: string[]): number[] {
389
+ return lines.filter(isMeterRow).map((line) => Number.parseInt(line.trim().match(/(\d+)%/)![1] ?? "", 10));
390
+ }
391
+
392
+ test("renderUsagePanel orders the NaN allowances by family and prints no totals", () => {
393
+ const lines = renderUsagePanel([parseNanQuota(GROUPED_NAN_QUOTA, NOW)], plainTheme, 80, NOW, { provider: "nan" }).map((line) => line.trimEnd());
394
+ assert.deepEqual(panelNames(lines), ["deepseek-v4-flash", "glm5.3-flash", "glm5.3", "glm5.2"]);
395
+ // 300M of 3B for DeepSeek first, then the GLM family by its own allowance.
396
+ assert.deepEqual(panelPercents(lines), [10, 10, 0, 0]);
397
+ // Every row keeps its own reset, inline: one line per window, never two.
398
+ const resets = panelResets(lines);
399
+ assert.equal(resets.length, 4);
400
+ assert.equal(resets[0], resets[1], "the two models on the 2026-10-01 period share their date");
401
+ assert.equal(resets[2], resets[3], "the two models on the 2026-10-17 period share theirs");
402
+ assert.notEqual(resets[1], resets[2], "each row carries its own window's reset, not its family's");
403
+ });
404
+
405
+ test("renderUsagePanel leaves providers without raw allowances ungrouped", () => {
406
+ const lines = renderUsagePanel([parseCodexUsage(CODEX_PAYLOAD, NOW)], plainTheme, 70, NOW);
407
+ assert.equal(lines.some((line) => line.includes("total")), false);
408
+ assert.match(lines[1], /^ {2}codex week /);
409
+ });
410
+
163
411
  test("UsageStore keeps the latest snapshot per provider and lists them in order", () => {
164
412
  const store = new UsageStore();
165
413
  const first: ProviderUsage = { provider: "openai-codex", plan: "pro", limits: [], fetchedAt: 1 };