aegis-desktop 0.7.1 → 0.7.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "aegis-desktop",
3
3
  "productName": "AEGIS Desktop",
4
- "version": "0.7.1",
4
+ "version": "0.7.3",
5
5
  "description": "Thin Electron host for AEGIS — a local chat UI over the shared client/aegis.js transport. Ships transport + UI only; engine logic stays server-side.",
6
6
  "author": {
7
7
  "name": "AEGIS Code",
package/renderer/app.js CHANGED
@@ -14,12 +14,13 @@
14
14
  * wrong.
15
15
  *
16
16
  * `budgetFor`/`maxTokensCeiling`/`FLAT_CEILING`/`EFFORT_TOKEN_BUDGET` come from
17
- * budget.js and `usageTokens` from usage.js, sibling classic scripts loaded
18
- * before this one (see index.html). There is no max-tokens control on this
19
- * surface at all: the ceiling is display-only (what a model says its own output
20
- * limit is, reported in the Model hint) and `budgetFor` answers — from the
21
- * Effort rung — what a request actually travels with. The token-usage →
22
- * displayed-number mapping stays unit-testable without window.aegis/models.
17
+ * budget.js and `turnAccounting`/`fmtCost` from usage.js, sibling classic
18
+ * scripts loaded before this one (see index.html). There is no
19
+ * max-tokens control on this surface at all: the ceiling is display-only (what a
20
+ * model says its own output limit is, reported in the Model hint) and
21
+ * `budgetFor` answers — from the Effort rung — what a request actually travels
22
+ * with. The token-usage → displayed-number mapping stays unit-testable without
23
+ * window.aegis/models.
23
24
  */
24
25
 
25
26
  // Everything below runs inside an IIFE. preload.js's contextBridge.exposeInMainWorld
@@ -105,6 +106,7 @@ const ELEMENT_IDS = {
105
106
  sessionsHint: 'sessions-hint',
106
107
  syncNow: 'sync-now',
107
108
  syncStatus: 'sync-status',
109
+ sessionMeter: 'session-meter',
108
110
  newChat: 'new-chat',
109
111
  memorySearchForm: 'memory-search-form',
110
112
  memoryQuery: 'memory-query',
@@ -229,6 +231,83 @@ let threadMessages = [];
229
231
  let classOptions = [];
230
232
  let modelMeta = new Map(); // model id -> raw model object from listModels() (P2 §6.3 ceiling)
231
233
 
234
+ // ------------------------------------------------------- rolling token meter
235
+ //
236
+ // Tokens are accounted the way the CLI accounts them: as a RUNNING SESSION
237
+ // TOTAL, folded turn by turn, not as a per-turn number that resets at the next
238
+ // call. The CLI's `recordTurn` (cli/src/app.js) folds every finished turn into
239
+ // one `session` object and prints that total in the status bar and in `ctrl+t`;
240
+ // this surface printed each turn's count and nothing else, so the only way to
241
+ // answer "what has this session spent" was to add the rows up by eye across the
242
+ // scrollback. That is the accounting difference between two surfaces running
243
+ // the same engine on the same prompt.
244
+ //
245
+ // Keyed by sessionId rather than held in one global, because the window holds
246
+ // several sessions across its life: switching threads must not carry one
247
+ // thread's spend into another's meter, and resuming a thread must not start its
248
+ // total at zero. `rollMessages` rebuilds a resumed session's total from the
249
+ // ledger rows the shared store already keeps.
250
+ const rollsBySession = new Map();
251
+
252
+ /** The rolling tallies for a session — empty, never undefined, when unseen. */
253
+ function rollFor(sessionId) {
254
+ const id = sessionId || '';
255
+ if (!rollsBySession.has(id)) rollsBySession.set(id, emptyRoll());
256
+ return rollsBySession.get(id);
257
+ }
258
+
259
+ /**
260
+ * Fold one finished dispatch into its session's rolling total and refresh the
261
+ * meter. Returns the turn's own accounting so the caller can still print the
262
+ * per-turn figure beside the session total (the CLI shows both: the meta row's
263
+ * `1.5k tok` and the status bar's rolling total).
264
+ */
265
+ function foldRoll(sessionId, usage, opts = {}) {
266
+ const id = sessionId || '';
267
+ const next = rollTurn(rollFor(id), usage, opts);
268
+ rollsBySession.set(id, next);
269
+ renderRollMeter(id);
270
+ return next;
271
+ }
272
+
273
+ /**
274
+ * Paint the topbar meter. Hidden while a session has accounted for nothing —
275
+ * an unused thread must not display a `0 tok` it never measured — and shown
276
+ * the moment a turn is folded in.
277
+ */
278
+ function renderRollMeter(sessionId) {
279
+ const el = els.sessionMeter;
280
+ if (!el) return;
281
+ const roll = rollsBySession.get(sessionId || '');
282
+ const line = roll && currentSessionId === sessionId ? fmtRoll(roll) : '';
283
+ el.textContent = line;
284
+ el.hidden = !line;
285
+ if (line) el.title = 'This session, counted the way the CLI counts it — every turn rolled into one running total';
286
+ }
287
+
288
+ /**
289
+ * The ledger fields one finished turn must carry into the session store, so
290
+ * the rolling total can be REBUILT when the thread is reopened.
291
+ *
292
+ * Without this the rolling meter was a one-window illusion: the store kept
293
+ * `{role, content}` only, so `rollMessages` found no `tokens` on any row this
294
+ * window had written and a reopened thread came back as a stack of
295
+ * unaccounted turns while the CLI — whose `recordExchange` does write `tokens`
296
+ * and `costUsd` into the very same file — came back with its full total. That
297
+ * asymmetry is the accounting difference, not the rendering of it.
298
+ *
299
+ * Nothing is written for a turn that reported no usage: a fabricated
300
+ * `{input: 0, output: 0}` row would read as a measured zero forever after,
301
+ * which is the one lie the token meter was built to avoid.
302
+ */
303
+ function ledgerFields(usage, model, turn) {
304
+ const fields = {};
305
+ if (turn && turn.tokens != null) fields.tokens = usageBuckets(usage);
306
+ if (turn && turn.cost != null && turn.real) fields.costUsd = turn.cost;
307
+ if (model) fields.model = model;
308
+ return fields;
309
+ }
310
+
232
311
  // ---------------------------------------------------------- discovery lane
233
312
  //
234
313
  // The chat flow reads vertically: one prompt, one answer, forever. That makes
@@ -1977,6 +2056,15 @@ async function spawnPath(card, spec) {
1977
2056
  // text nor a tool call would trigger the empty-turn recovery — an
1978
2057
  // extra dispatch this lane has no gathered context to justify.
1979
2058
  tools: false,
2059
+ // `singlePass` is what makes that promise true on the pooled class.
2060
+ // The engine derives the pooled brain flag as
2061
+ // `singlePass ? false : autonomous ? true : undefined`, and this lane
2062
+ // set neither: on a `nexus-brain*` id the flag was `undefined`, so the
2063
+ // model id's own default decided — which is the fan-out. A card that
2064
+ // was meant to be one cheap 1024-token pass could therefore bill a
2065
+ // workers + synthesis dispatch, twice per turn. Explicit `false` =
2066
+ // one provider call, always.
2067
+ singlePass: true,
1980
2068
  sessionId: id,
1981
2069
  },
1982
2070
  onDelta
@@ -1993,8 +2081,29 @@ async function spawnPath(card, spec) {
1993
2081
  const bits = [spec.path.title];
1994
2082
  if (data && data.model) bits.push(data.model);
1995
2083
  else if (spec.model) bits.push(spec.model);
1996
- const flowTokens = usageTokens(data && data.usage);
1997
- if (flowTokens != null) bits.push(`${flowTokens} tokens`);
2084
+ // A lane card is a dispatch the same way a turn is, and it was the
2085
+ // un-costed half of the 3x story: it reported tokens with no charge beside
2086
+ // them, so the extra calls were the least visible thing on the screen.
2087
+ const flow = turnAccounting(data && data.usage, spec.model, {
2088
+ costUsd: data && typeof data.costUsd === 'number' ? data.costUsd : undefined,
2089
+ });
2090
+ if (flow.tokens != null) bits.push(`${flow.tokens} tokens`);
2091
+ if (flow.cost != null) bits.push(fmtCost(flow.cost, flow.real));
2092
+ // …and roll it into the session, which is the half that was missing. The
2093
+ // card shows this ONE dispatch; a session total that skipped it would be
2094
+ // the lane's calls — the extra ones this feature's cost story is made of —
2095
+ // being the only calls on screen that never get counted. `turns: 0`
2096
+ // because a discovery path is not a turn the user asked for: it
2097
+ // contributes tokens and calls, and leaves the turn count to real
2098
+ // exchanges.
2099
+ const roll = foldRoll(spec.parentSessionId, data && data.usage, {
2100
+ model: spec.model,
2101
+ costUsd: data && typeof data.costUsd === 'number' ? data.costUsd : undefined,
2102
+ calls: data && data.calls,
2103
+ turns: 0,
2104
+ });
2105
+ const rollLine = fmtRoll(roll);
2106
+ if (rollLine) bits.push(`session: ${rollLine}`);
1998
2107
  meta.textContent = bits.join(' · ');
1999
2108
  } catch (err) {
2000
2109
  const message = err && err.message ? err.message : String(err);
@@ -2845,6 +2954,15 @@ function openSession(id) {
2845
2954
  // its on-screen transcript — continuing it as sessionId reuses the same
2846
2955
  // id and threadMessages carries the prior turns into the next send().
2847
2956
  currentSessionId = s.id;
2957
+ // A resumed thread resumes its spend too. The shared store's ledger rows
2958
+ // carry `tokens`/`costUsd` (client/session-store.js recordExchange writes
2959
+ // exactly the shape rollMessages reads), so the rolling total is rebuilt
2960
+ // from what was really recorded rather than restarting at zero — the
2961
+ // CLI's aggregateSessionUsage, which sums history.jsonl for the same
2962
+ // reason. A row this window wrote before ledgerFields existed carries no
2963
+ // `tokens` and folds as unaccounted, which is stated rather than guessed.
2964
+ rollsBySession.set(s.id, rollMessages(msgs));
2965
+ renderRollMeter(s.id);
2848
2966
  threadMessages = msgs
2849
2967
  .filter((m) => m.role === 'user' || m.role === 'assistant')
2850
2968
  .map((m) => ({ role: m.role, content: m.content || m.text || '' }));
@@ -2899,6 +3017,12 @@ function newChat() {
2899
3017
  currentSessionId = null;
2900
3018
  threadMessages = [];
2901
3019
  flowCount = 0;
3020
+ // A fresh thread opens with an empty meter. The outgoing session's roll is
3021
+ // left in the map (reopening it rebuilds from the store anyway), but the
3022
+ // topbar must not keep showing the thread the user just left — `sessionId`
3023
+ // nulls out here and the new id is minted on the first send, so nothing is
3024
+ // hidden that will not reappear with this thread's own number.
3025
+ renderRollMeter(null);
2902
3026
  }
2903
3027
 
2904
3028
  /**
@@ -3071,12 +3195,41 @@ async function send() {
3071
3195
  // class it is what sized the call, and a user cannot tell a 16k turn from a
3072
3196
  // 64k one by looking at the answer.
3073
3197
  else if (effort) bits.push(`effort: ${effort}`);
3074
- const turnTokens = usageTokens(data && data.usage);
3075
- if (turnTokens != null) bits.push(`tokens: ${turnTokens}`);
3198
+ // Tokens AND what they cost, on the CLI's rule: a pooled turn's settled
3199
+ // charge (`costUsd`) is reported verbatim, and only a turn without one is
3200
+ // priced from the local rate table and marked an estimate. Printing tokens
3201
+ // alone left this surface with no comparable meter at all — the desktop
3202
+ // and the CLI run the same engine, so a gap between them had to be
3203
+ // measured on the same prompt, and one side was not measuring.
3204
+ const turn = turnAccounting(data && data.usage, model, {
3205
+ costUsd: data && typeof data.costUsd === 'number' ? data.costUsd : undefined,
3206
+ });
3207
+ if (turn.tokens != null) bits.push(`tokens: ${turn.tokens}`);
3208
+ if (turn.cost != null) bits.push(fmtCost(turn.cost, turn.real));
3209
+ // …AND the running session total beside it, which is the number the CLI
3210
+ // prints. The per-turn figure answers "what did that call cost"; only the
3211
+ // rolling one answers "what has this conversation cost", and it was the
3212
+ // missing half. Folded here rather than only on the meter so a turn that
3213
+ // reported no usage still counts as a turn (rollTurn counts before it
3214
+ // tests the count) instead of vanishing from the session.
3215
+ const roll = foldRoll(sessionId, data && data.usage, {
3216
+ model,
3217
+ costUsd: data && typeof data.costUsd === 'number' ? data.costUsd : undefined,
3218
+ calls: data && data.calls,
3219
+ });
3220
+ const rollLine = fmtRoll(roll);
3221
+ if (rollLine) bits.push(`session: ${rollLine}`);
3076
3222
  addMessage('assistant', text, bits.join(' · ') || undefined, sessionId, toolLog);
3077
3223
 
3078
3224
  try {
3079
- await sync.append(sessionId, { role: 'assistant', content: text });
3225
+ await sync.append(sessionId, {
3226
+ role: 'assistant',
3227
+ content: text,
3228
+ // The turn's ledger fields, so this window's spend survives the window
3229
+ // — see ledgerFields. This is what makes the rolling meter the same
3230
+ // quantity after a reopen as it was before one.
3231
+ ...ledgerFields(data && data.usage, model, turn),
3232
+ });
3080
3233
  await sync.save({ id: sessionId, title: prompt.slice(0, 60) });
3081
3234
  } catch {
3082
3235
  /* persistence is non-fatal */
@@ -3206,9 +3359,17 @@ async function init() {
3206
3359
  });
3207
3360
  updateAutonomousControlsVisibility();
3208
3361
 
3209
- // The discovery lane is opt-in per machine, remembered across restarts.
3362
+ // The discovery lane is opt-in, remembered across restarts.
3363
+ //
3364
+ // This used to read `savedExplore === 'off'` against a checkbox that shipped
3365
+ // `checked` in index.html, i.e. the opposite of the comment above it: a fresh
3366
+ // install (no `aegis.explore` key, and nothing had ever written one) landed
3367
+ // with the lane ON and billed two extra model calls after every single turn —
3368
+ // ~3× the tokens of a plain reply, silently, because the lane is
3369
+ // fire-and-forget and never looks slow. Only an explicit 'on' turns it on
3370
+ // now; every other state, including "never asked", means off.
3210
3371
  const savedExplore = localStorage.getItem(EXPLORE_KEY);
3211
- if (savedExplore === 'off') els.exploreToggle.checked = false;
3372
+ els.exploreToggle.checked = savedExplore === 'on';
3212
3373
  els.exploreToggle.addEventListener('change', () => {
3213
3374
  localStorage.setItem(EXPLORE_KEY, els.exploreToggle.checked ? 'on' : 'off');
3214
3375
  if (!els.exploreToggle.checked) abortBranches();
@@ -22,6 +22,11 @@
22
22
  <span class="dot" id="conn-dot"></span>
23
23
  <span id="conn-text">connecting…</span>
24
24
  </div>
25
+ <!-- The rolling session token total, counted the way the CLI counts it:
26
+ every finished turn folded into one running number rather than a
27
+ per-turn count that resets. Hidden until a turn has been accounted
28
+ for, so an untouched thread never shows a `0 tok` it never measured. -->
29
+ <div class="session-meter" id="session-meter" hidden></div>
25
30
  <div class="spacer"></div>
26
31
  <button type="button" id="new-chat" class="ghost-btn">New chat</button>
27
32
  </header>
@@ -353,9 +358,12 @@
353
358
  <label
354
359
  class="flow-toggle"
355
360
  for="explore-toggle"
356
- title="After every answer, send the AI down two extra paths — an alternative angle and an unexpected discovery — shown as a horizontal discovery lane."
361
+ title="Optional, off by default. After every answer, send the AI down two extra paths — an alternative angle and an unexpected discovery — shown as a horizontal discovery lane. This costs two extra model calls per turn (roughly 3× the tokens of a plain reply)."
357
362
  >
358
- <input type="checkbox" id="explore-toggle" checked />
363
+ <!-- Unchecked in the markup on purpose: the lane bills two extra
364
+ model calls per turn, so it is opt-in. app.js only ticks this
365
+ when the user has explicitly saved 'on' (aegis.explore). -->
366
+ <input type="checkbox" id="explore-toggle" />
359
367
  <span>discovery lane</span>
360
368
  </label>
361
369
  <button type="submit" id="send">Send</button>
@@ -114,6 +114,28 @@ body {
114
114
  box-shadow: 0 0 6px var(--teal);
115
115
  }
116
116
 
117
+ /* The rolling session token total — the CLI's status-bar figure, in the
118
+ topbar. Muted and monospaced so it reads as a meter rather than a heading,
119
+ and it never wraps: a number that reflows the topbar every turn is worse
120
+ than no number at all. Hidden via the `hidden` attribute, so the rule below
121
+ has to win against the flex display the base class would otherwise get. */
122
+ .session-meter {
123
+ font-size: 12px;
124
+ font-family: var(--font-mono);
125
+ color: var(--text2);
126
+ white-space: nowrap;
127
+ overflow: hidden;
128
+ text-overflow: ellipsis;
129
+ max-width: 40ch;
130
+ padding: 2px 8px;
131
+ border: 1px solid var(--border);
132
+ border-radius: 999px;
133
+ }
134
+
135
+ .session-meter[hidden] {
136
+ display: none;
137
+ }
138
+
117
139
  .spacer {
118
140
  flex: 1;
119
141
  }
package/renderer/usage.js CHANGED
@@ -1,7 +1,7 @@
1
1
  'use strict';
2
2
 
3
3
  /**
4
- * Pure token-usage → display-number mapping.
4
+ * Pure token-usage → display-number mapping, and the cost half of it.
5
5
  *
6
6
  * Standalone from app.js (same reason as budget.js/stream-policy.js): it is
7
7
  * requireable from a plain Node test without window.aegis. app.js only calls
@@ -18,6 +18,27 @@
18
18
  * the call was silently being billed. Accepting both spellings, and deriving
19
19
  * the total when the provider doesn't state one, is what makes the spend
20
20
  * visible regardless of which endpoint answered.
21
+ *
22
+ * ── The cost half, on the CLI's principle ──────────────────────────────────
23
+ *
24
+ * The renderer printed a token count and nothing else, so the desktop had no
25
+ * meter to compare against `aegiscodex` on the same engine and the same
26
+ * prompt — the comparison that started this whole line of work. The CLI's
27
+ * rule (cli/src/tokens.js, ported here) has three parts, and all three matter:
28
+ *
29
+ * 1. A pooled turn's bill is settled SERVER-side and carries a margin and a
30
+ * prompt-cache discount the client cannot see. When the response states
31
+ * `costUsd`, that figure wins verbatim — it is the truth about the
32
+ * charge, not an estimate of it.
33
+ * 2. Only in the absence of a settled charge does the local rate table
34
+ * apply, and the result is labeled an estimate so nobody reads a guess as
35
+ * a bill.
36
+ * 3. The rate table is resolved by exact id, then longest prefix. The
37
+ * DeepSeek row is load-bearing: without it every direct DeepSeek turn
38
+ * fell through to Sonnet's $3.00/$15.00 per M against a real
39
+ * $0.14/$0.28 — 21x the input rate and 54x the output rate — which made
40
+ * the *meter* the largest single contributor to the apparent cost gap
41
+ * between two surfaces running identical code.
21
42
  */
22
43
 
23
44
  /**
@@ -25,18 +46,352 @@
25
46
  * reported none (an unknown count must render as nothing, never as `0`).
26
47
  *
27
48
  * @param {{total_tokens?: number, prompt_tokens?: number, completion_tokens?: number,
28
- * input_tokens?: number, output_tokens?: number}|null|undefined} usage
49
+ * input_tokens?: number, output_tokens?: number,
50
+ * input?: number, output?: number}|null|undefined} usage
29
51
  * @returns {number|null}
30
52
  */
31
53
  function usageTokens(usage) {
32
54
  if (!usage || typeof usage !== 'object') return null;
33
55
  if (typeof usage.total_tokens === 'number') return usage.total_tokens;
34
- const input = usage.input_tokens ?? usage.prompt_tokens;
35
- const output = usage.output_tokens ?? usage.completion_tokens;
56
+ // Three spellings of the same quantity: OpenAI's, Anthropic's, and this
57
+ // file's own bucket shape — which is also the shape a LEDGER row carries
58
+ // (cli/src/history.js writes `{input, output, cacheRead, cacheWrite}` into
59
+ // sessions.json, and the CLI's demo path writes `{input, output}`). Reading
60
+ // only the two wire spellings left every stored row uncountable, so a
61
+ // resumed session's rolling total started at zero even though its ledger
62
+ // said otherwise.
63
+ const input = usage.input_tokens ?? usage.prompt_tokens ?? usage.input;
64
+ const output = usage.output_tokens ?? usage.completion_tokens ?? usage.output;
36
65
  if (typeof input !== 'number' && typeof output !== 'number') return null;
37
66
  return (input || 0) + (output || 0);
38
67
  }
39
68
 
69
+ // Per-million-token USD rates. Cache-read/write matter for long sessions.
70
+ //
71
+ // These are the PROVIDER's rates, for the fallback path where a turn has no
72
+ // server-settled charge (a direct provider, ollama, a custom endpoint). A
73
+ // pooled turn reports the ledger figure instead — see turnAccounting's
74
+ // `costUsd` — because the pool's bill carries a margin and a prompt-cache
75
+ // discount this table cannot see.
76
+ //
77
+ // Keys are matched by exact id first, then by prefix (ratesFor below), so a
78
+ // family row covers every dated variant of it.
79
+ const RATES = {
80
+ sonnet: { input: 3.00, output: 15.00, cacheRead: 0.30, cacheWrite: 3.75 },
81
+ default: { input: 3.00, output: 15.00, cacheRead: 0.30, cacheWrite: 3.75 },
82
+ fable: { input: 5.00, output: 25.00, cacheRead: 0.50, cacheWrite: 6.25 },
83
+ opus: { input: 5.00, output: 25.00, cacheRead: 0.50, cacheWrite: 6.25 },
84
+ // DeepSeek V4 — the provider's published rate, independently corroborated by
85
+ // aegis1 services/nexus_provider/catalog.py (cost_per_1k_input=0.00014,
86
+ // cost_per_1k_output=0.00028) and referenced in services/pricing.py.
87
+ //
88
+ // This row was MISSING upstream, and usageCost fell through to RATES.sonnet
89
+ // for every DeepSeek turn: $3.00/$15.00 per M against a real $0.14/$0.28 —
90
+ // 21x the input rate and 54x the output rate, on every direct DeepSeek call.
91
+ //
92
+ // cacheRead is 0.1x input (aegis1 services/pricing.py
93
+ // CACHE_READ_FRACTION_BY_COMPANY["deepseek"] = 0.1 -> $0.014/M).
94
+ // cacheWrite is the input rate: DeepSeek bills a cache write as ordinary
95
+ // input tokens and charges no separate write premium, unlike Anthropic's
96
+ // 1.25x. A zero here would understate a session that writes cache.
97
+ deepseek: { input: 0.14, output: 0.28, cacheRead: 0.014, cacheWrite: 0.14 },
98
+ };
99
+
100
+ /**
101
+ * The rate row for a model id, provider, or alias. Exact id wins, then the
102
+ * longest matching prefix, then the Sonnet-class default.
103
+ */
104
+ function ratesFor(model) {
105
+ const id = String(model || '').toLowerCase();
106
+ if (!id) return RATES.default;
107
+ if (RATES[id]) return RATES[id];
108
+ let best = null;
109
+ let bestLen = 0;
110
+ for (const [prefix, rates] of Object.entries(RATES)) {
111
+ if (id.startsWith(prefix) && prefix.length > bestLen) {
112
+ best = rates;
113
+ bestLen = prefix.length;
114
+ }
115
+ }
116
+ return best || RATES.default;
117
+ }
118
+
119
+ /**
120
+ * The four billable buckets from a wire usage object, accepting both provider
121
+ * spellings. Cache fields have no Anthropic-compatible short form here because
122
+ * the desktop's transport normalises them (desktop/lib/local/providers.js).
123
+ *
124
+ * @param {object|null|undefined} usage
125
+ * @returns {{input: number, output: number, cacheRead: number, cacheWrite: number}}
126
+ */
127
+ function usageBuckets(usage) {
128
+ const u = usage && typeof usage === 'object' ? usage : {};
129
+ const num = (...candidates) => {
130
+ for (const c of candidates) if (typeof c === 'number') return c;
131
+ return 0;
132
+ };
133
+ const input = num(u.input_tokens, u.prompt_tokens, u.input);
134
+ const output = num(u.output_tokens, u.completion_tokens, u.output);
135
+ // A stated total larger than the split means the provider counted tokens the
136
+ // split does not name (thinking, cached reads). Attribute the remainder to
137
+ // input rather than dropping it: dropping it would understate the bill.
138
+ const total = num(u.total_tokens);
139
+ const cacheRead = num(u.cache_read_input_tokens, u.cacheRead);
140
+ const cacheWrite = num(u.cache_creation_input_tokens, u.cacheWrite);
141
+ const accounted = input + output + cacheRead + cacheWrite;
142
+ return {
143
+ input: total > accounted ? input + (total - accounted) : input,
144
+ output,
145
+ cacheRead,
146
+ cacheWrite,
147
+ };
148
+ }
149
+
150
+ /** Dollar cost of a usage record at the given model's rates (USD, estimate). */
151
+ function usageCost(usage, model = 'sonnet') {
152
+ const r = ratesFor(model);
153
+ const u = usageBuckets(usage);
154
+ const toD = (n, rate) => (n / 1_000_000) * rate;
155
+ return toD(u.input, r.input)
156
+ + toD(u.output, r.output)
157
+ + toD(u.cacheRead, r.cacheRead)
158
+ + toD(u.cacheWrite, r.cacheWrite);
159
+ }
160
+
161
+ /**
162
+ * What one turn cost, as the CLI reports it: the server's settled charge when
163
+ * the response carries one, otherwise the rate table's estimate, always with
164
+ * the distinction preserved so it can be labeled.
165
+ *
166
+ * @param {object|null|undefined} usage the response's `usage` object
167
+ * @param {string} model the model id that answered
168
+ * @param {{costUsd?: number}} [opts] the server-settled charge, if any
169
+ * @returns {{tokens: number|null, cost: number|null, real: boolean, estimated: boolean}}
170
+ * `real` is true only when `cost` is the settled charge. `cost` is
171
+ * `null` when nothing was reported and no model was named — an
172
+ * unpriced turn must render as nothing, never as $0.0000.
173
+ */
174
+ function turnAccounting(usage, model, opts = {}) {
175
+ const tokens = usageTokens(usage);
176
+ const u = usage && typeof usage === 'object' ? usage : {};
177
+ // The settled charge can arrive either on the response or folded into the
178
+ // usage object — aegiscodex-dev/src/main.js does the latter
179
+ // (`{ ...result.usage, costUsd: result.costUsd }`), so both are accepted.
180
+ const settled = typeof opts.costUsd === 'number'
181
+ ? opts.costUsd
182
+ : (typeof u.costUsd === 'number' ? u.costUsd : undefined);
183
+ if (typeof settled === 'number') {
184
+ return { tokens, cost: settled, real: true, estimated: false };
185
+ }
186
+ if (tokens == null) return { tokens, cost: null, real: false, estimated: false };
187
+ const priced = usageCost(u, model);
188
+ // A model the table cannot place is still priced at the Sonnet-class default
189
+ // (ratesFor never returns nothing), so this is always an estimate.
190
+ return { tokens, cost: priced, real: false, estimated: true };
191
+ }
192
+
193
+ /**
194
+ * ── The rolling session tallies, on the CLI's rule ─────────────────────────
195
+ *
196
+ * The two surfaces did not merely print different numbers — they counted
197
+ * differently. The CLI never shows a turn's tokens in isolation: `recordTurn`
198
+ * (cli/src/app.js) folds each finished turn's usage into one `session` object
199
+ * — `tokens`, `inputTokens`, `outputTokens`, `calls` — and what the user reads
200
+ * back is that RUNNING TOTAL. The status bar prints `state.tokens`
201
+ * (renderStatus), and `ctrl+t` prints the tallies in one line
202
+ * (`tokenSummary`: `12,400 tok (10,100 in / 2,300 out) · 4 calls · €0.03`).
203
+ *
204
+ * The desktop counted per turn only. Every meta row was a fresh count that
205
+ * reset at the next call, so "what has this session spent" was answerable only
206
+ * by adding the rows up by eye across a scrollback — which is most of why the
207
+ * desktop looked like it accounted differently from the CLI on identical
208
+ * engine code and an identical prompt.
209
+ *
210
+ * Three properties of the CLI's fold are load-bearing and are reproduced here
211
+ * exactly, because dropping any one of them reintroduces a specific lie:
212
+ *
213
+ * 1. `turns` and `calls` are incremented BEFORE the "did usage come back?"
214
+ * gate (recordTurn counts first, then tests `tokens != null`). A turn
215
+ * that reported nothing still happened; a tally that skipped it would
216
+ * report the session as shorter and cheaper than it was.
217
+ * 2. `tokens`, `input` and `output` accumulate — never reset. A rolling
218
+ * total that resets per turn is the per-turn count it replaced.
219
+ * 3. A `null` count contributes nothing and is counted in `unknown`, so
220
+ * `tokens: 0` is only ever read as a real zero. Nothing is ever added as
221
+ * a fabricated 0 to make the arithmetic look complete.
222
+ *
223
+ * The money split is the CLI's too: a settled charge (the pool's ledger
224
+ * figure) rolls into `cost`, and a locally priced turn rolls into `estimate`.
225
+ * They are kept apart rather than summed so a `~`-estimate can never be read
226
+ * as part of the bill — the distinction `fmtCost` marks on a single turn, held
227
+ * across the session.
228
+ */
229
+
230
+ /** A session with nothing accounted for yet. */
231
+ function emptyRoll() {
232
+ return {
233
+ turns: 0,
234
+ calls: 0,
235
+ tokens: 0,
236
+ input: 0,
237
+ output: 0,
238
+ cacheRead: 0,
239
+ cacheWrite: 0,
240
+ /** Dispatches that reported no usage at all — the honest gap in `tokens`. */
241
+ unknown: 0,
242
+ /** Settled charges (server-settled `costUsd`), rolled. */
243
+ cost: 0,
244
+ /** Locally priced turns, rolled. Never mixed into `cost`. */
245
+ estimate: 0,
246
+ };
247
+ }
248
+
249
+ /**
250
+ * Fold one completed dispatch into a session's rolling tallies. Pure: it
251
+ * returns a NEW roll and never mutates the one it was handed, so a half-applied
252
+ * fold cannot exist.
253
+ *
254
+ * @param {object} [roll] the roll so far (emptyRoll() when omitted)
255
+ * @param {object} [usage] the response's `usage` object
256
+ * @param {{model?: string, costUsd?: number, calls?: number, turns?: number}} [opts]
257
+ * `calls` defaults to 1; a pooled turn may report how many provider
258
+ * calls it actually made. `turns: 0` folds a dispatch that is not a
259
+ * turn of its own — the discovery-lane card, which bills like any other
260
+ * call but is not something the user asked for.
261
+ * @returns {object} the new roll
262
+ */
263
+ function rollTurn(roll, usage, opts = {}) {
264
+ const next = Object.assign(emptyRoll(), roll || {});
265
+ next.turns += opts.turns === undefined ? 1 : Number(opts.turns) || 0;
266
+ next.calls += opts.calls === undefined ? 1 : Number(opts.calls) || 0;
267
+ const turn = turnAccounting(usage, opts.model, { costUsd: opts.costUsd });
268
+ if (turn.tokens == null) {
269
+ next.unknown += 1;
270
+ return next;
271
+ }
272
+ const b = usageBuckets(usage);
273
+ next.tokens += turn.tokens;
274
+ next.input += b.input;
275
+ next.output += b.output;
276
+ next.cacheRead += b.cacheRead;
277
+ next.cacheWrite += b.cacheWrite;
278
+ if (turn.real) next.cost += turn.cost;
279
+ else if (turn.cost != null) next.estimate += turn.cost;
280
+ return next;
281
+ }
282
+
283
+ /**
284
+ * Rebuild a session's rolling total from stored exchanges — the desktop's
285
+ * counterpart of the CLI's `aggregateSessionUsage`, which sums history.jsonl so
286
+ * a resumed session (and a compacted one) still reports everything it spent.
287
+ *
288
+ * Reads the shapes the shared store writes: an assistant message carrying
289
+ * `tokens: {input, output, cacheRead, cacheWrite}` and, when the pool settled
290
+ * the turn, `costUsd` (cli/src/history.js → session-store.recordExchange).
291
+ * Messages the desktop itself appended carry no `tokens` and fold as `unknown`
292
+ * — a resumed thread states what is known and does not invent the rest.
293
+ *
294
+ * @param {Array<{role?: string, tokens?: object, costUsd?: number, model?: string}>} [messages]
295
+ * @returns {object} the roll
296
+ */
297
+ function rollMessages(messages) {
298
+ let roll = emptyRoll();
299
+ for (const m of Array.isArray(messages) ? messages : []) {
300
+ if (!m || m.role !== 'assistant') continue;
301
+ roll = rollTurn(roll, m.tokens, {
302
+ model: m.model,
303
+ costUsd: typeof m.costUsd === 'number' ? m.costUsd : undefined,
304
+ calls: m.calls,
305
+ });
306
+ }
307
+ return roll;
308
+ }
309
+
310
+ /**
311
+ * The CLI's rendering of a session tally — `cli/src/format.js fmtTokens`, which
312
+ * is the one `tokenSummary` actually imports (`cli/src/app.js:50`). NOT the
313
+ * `1.5k`/`12.3k` form in `cli/src/tokens.js`: that one belongs to the /cost
314
+ * panels, and using it here would print `12.4k` on the very total the CLI
315
+ * prints as `12,400` — a rendering difference stacked on top of the accounting
316
+ * difference this change exists to remove.
317
+ *
318
+ * Comma-grouped integer. Anything not finite and positive renders `0`, which is
319
+ * the CLI's rule and keeps a stray NaN from reaching the topbar.
320
+ */
321
+ function fmtTokens(n) {
322
+ const v = Number(n);
323
+ if (!Number.isFinite(v) || v <= 0) return '0';
324
+ return Math.round(v)
325
+ .toString()
326
+ .replace(/\B(?=(\d{3})+(?!\d))/g, ',');
327
+ }
328
+
329
+ /**
330
+ * The rolling session total as one line — the desktop's counterpart of the
331
+ * CLI's `tokenSummary`. Empty string when nothing has been accounted for, so
332
+ * an untouched session adds no noise to a turn's meta line.
333
+ *
334
+ * @param {object} [roll]
335
+ * @returns {string} e.g. `12,400 tok (10,100 in / 2,300 out) · 4 calls · $0.0310`
336
+ */
337
+ function fmtRoll(roll) {
338
+ const r = roll || emptyRoll();
339
+ if (!r.turns && !r.calls) return '';
340
+ // The first two fields are `tokenSummary` (cli/src/app.js:1413) verbatim,
341
+ // separator and all: `12,400 tok (10,100 in / 2,300 out) · 4 calls`. There the
342
+ // parenthetical is unconditional and the call count is singular at one; both
343
+ // are kept, because a line that is only *sometimes* shaped like the CLI's is a
344
+ // lookalike rather than the same quantity. The call count is also the number
345
+ // that reveals a fan-out, which is why it is not hidden at 1.
346
+ const bits = [
347
+ `${fmtTokens(r.tokens)} tok (${fmtTokens(r.input)} in / ${fmtTokens(r.output)} out)`,
348
+ `${r.calls} call${r.calls === 1 ? '' : 's'}`,
349
+ ];
350
+ // Money is where this line departs from `tokenSummary`, deliberately: that
351
+ // one sums a single `session.cost` in EUR via fmtEur, while this surface keeps
352
+ // a settled charge and a local estimate apart so a `~`-estimate can never be
353
+ // read as part of the bill. Desktop's pre-existing fmtCost renders both, and
354
+ // its `$` convention is left exactly as it was.
355
+ if (r.cost > 0 || r.estimate > 0) {
356
+ const money = [];
357
+ if (r.cost > 0) money.push(fmtCost(r.cost, true));
358
+ if (r.estimate > 0) money.push(fmtCost(r.estimate, false));
359
+ bits.push(money.join(' + '));
360
+ }
361
+ // No counterpart in `tokenSummary`, which folds a usage-less turn silently and
362
+ // so reports a total short of the truth without saying so. Named here instead:
363
+ // the count appears only when something went unreported, and it never changes
364
+ // a number — it only says the number is not the whole story.
365
+ if (r.unknown) bits.push(`${r.unknown} unrpt`);
366
+ return bits.join(' · ');
367
+ }
368
+
369
+ /**
370
+ * A cost for display. Estimates are marked with `~` so an estimate is never
371
+ * mistaken for a settled charge.
372
+ *
373
+ * @param {number} cost
374
+ * @param {boolean} [real]
375
+ * @returns {string}
376
+ */
377
+ function fmtCost(cost, real) {
378
+ if (typeof cost !== 'number' || !isFinite(cost)) return '';
379
+ return `${real ? '' : '~'}$${cost.toFixed(4)}`;
380
+ }
381
+
40
382
  if (typeof module !== 'undefined' && module.exports) {
41
- module.exports = { usageTokens };
383
+ module.exports = {
384
+ RATES,
385
+ usageTokens,
386
+ ratesFor,
387
+ usageBuckets,
388
+ usageCost,
389
+ turnAccounting,
390
+ emptyRoll,
391
+ rollTurn,
392
+ rollMessages,
393
+ fmtTokens,
394
+ fmtRoll,
395
+ fmtCost,
396
+ };
42
397
  }