aegis-desktop 0.7.1 → 0.7.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/renderer/app.js +174 -13
- package/renderer/index.html +10 -2
- package/renderer/style.css +22 -0
- package/renderer/usage.js +360 -5
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "aegis-desktop",
|
|
3
3
|
"productName": "AEGIS Desktop",
|
|
4
|
-
"version": "0.7.
|
|
4
|
+
"version": "0.7.3",
|
|
5
5
|
"description": "Thin Electron host for AEGIS — a local chat UI over the shared client/aegis.js transport. Ships transport + UI only; engine logic stays server-side.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "AEGIS Code",
|
package/renderer/app.js
CHANGED
|
@@ -14,12 +14,13 @@
|
|
|
14
14
|
* wrong.
|
|
15
15
|
*
|
|
16
16
|
* `budgetFor`/`maxTokensCeiling`/`FLAT_CEILING`/`EFFORT_TOKEN_BUDGET` come from
|
|
17
|
-
* budget.js and `
|
|
18
|
-
* before this one (see index.html). There is no
|
|
19
|
-
* surface at all: the ceiling is display-only (what a
|
|
20
|
-
* limit is, reported in the Model hint) and
|
|
21
|
-
* Effort rung — what a request actually travels
|
|
22
|
-
* displayed-number mapping stays unit-testable without
|
|
17
|
+
* budget.js and `turnAccounting`/`fmtCost` from usage.js, sibling classic
|
|
18
|
+
* scripts loaded before this one (see index.html). There is no
|
|
19
|
+
* max-tokens control on this surface at all: the ceiling is display-only (what a
|
|
20
|
+
* model says its own output limit is, reported in the Model hint) and
|
|
21
|
+
* `budgetFor` answers — from the Effort rung — what a request actually travels
|
|
22
|
+
* with. The token-usage → displayed-number mapping stays unit-testable without
|
|
23
|
+
* window.aegis/models.
|
|
23
24
|
*/
|
|
24
25
|
|
|
25
26
|
// Everything below runs inside an IIFE. preload.js's contextBridge.exposeInMainWorld
|
|
@@ -105,6 +106,7 @@ const ELEMENT_IDS = {
|
|
|
105
106
|
sessionsHint: 'sessions-hint',
|
|
106
107
|
syncNow: 'sync-now',
|
|
107
108
|
syncStatus: 'sync-status',
|
|
109
|
+
sessionMeter: 'session-meter',
|
|
108
110
|
newChat: 'new-chat',
|
|
109
111
|
memorySearchForm: 'memory-search-form',
|
|
110
112
|
memoryQuery: 'memory-query',
|
|
@@ -229,6 +231,83 @@ let threadMessages = [];
|
|
|
229
231
|
let classOptions = [];
|
|
230
232
|
let modelMeta = new Map(); // model id -> raw model object from listModels() (P2 §6.3 ceiling)
|
|
231
233
|
|
|
234
|
+
// ------------------------------------------------------- rolling token meter
|
|
235
|
+
//
|
|
236
|
+
// Tokens are accounted the way the CLI accounts them: as a RUNNING SESSION
|
|
237
|
+
// TOTAL, folded turn by turn, not as a per-turn number that resets at the next
|
|
238
|
+
// call. The CLI's `recordTurn` (cli/src/app.js) folds every finished turn into
|
|
239
|
+
// one `session` object and prints that total in the status bar and in `ctrl+t`;
|
|
240
|
+
// this surface printed each turn's count and nothing else, so the only way to
|
|
241
|
+
// answer "what has this session spent" was to add the rows up by eye across the
|
|
242
|
+
// scrollback. That is the accounting difference between two surfaces running
|
|
243
|
+
// the same engine on the same prompt.
|
|
244
|
+
//
|
|
245
|
+
// Keyed by sessionId rather than held in one global, because the window holds
|
|
246
|
+
// several sessions across its life: switching threads must not carry one
|
|
247
|
+
// thread's spend into another's meter, and resuming a thread must not start its
|
|
248
|
+
// total at zero. `rollMessages` rebuilds a resumed session's total from the
|
|
249
|
+
// ledger rows the shared store already keeps.
|
|
250
|
+
const rollsBySession = new Map();
|
|
251
|
+
|
|
252
|
+
/** The rolling tallies for a session — empty, never undefined, when unseen. */
|
|
253
|
+
function rollFor(sessionId) {
|
|
254
|
+
const id = sessionId || '';
|
|
255
|
+
if (!rollsBySession.has(id)) rollsBySession.set(id, emptyRoll());
|
|
256
|
+
return rollsBySession.get(id);
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
/**
|
|
260
|
+
* Fold one finished dispatch into its session's rolling total and refresh the
|
|
261
|
+
* meter. Returns the turn's own accounting so the caller can still print the
|
|
262
|
+
* per-turn figure beside the session total (the CLI shows both: the meta row's
|
|
263
|
+
* `1.5k tok` and the status bar's rolling total).
|
|
264
|
+
*/
|
|
265
|
+
function foldRoll(sessionId, usage, opts = {}) {
|
|
266
|
+
const id = sessionId || '';
|
|
267
|
+
const next = rollTurn(rollFor(id), usage, opts);
|
|
268
|
+
rollsBySession.set(id, next);
|
|
269
|
+
renderRollMeter(id);
|
|
270
|
+
return next;
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
/**
|
|
274
|
+
* Paint the topbar meter. Hidden while a session has accounted for nothing —
|
|
275
|
+
* an unused thread must not display a `0 tok` it never measured — and shown
|
|
276
|
+
* the moment a turn is folded in.
|
|
277
|
+
*/
|
|
278
|
+
function renderRollMeter(sessionId) {
|
|
279
|
+
const el = els.sessionMeter;
|
|
280
|
+
if (!el) return;
|
|
281
|
+
const roll = rollsBySession.get(sessionId || '');
|
|
282
|
+
const line = roll && currentSessionId === sessionId ? fmtRoll(roll) : '';
|
|
283
|
+
el.textContent = line;
|
|
284
|
+
el.hidden = !line;
|
|
285
|
+
if (line) el.title = 'This session, counted the way the CLI counts it — every turn rolled into one running total';
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/**
|
|
289
|
+
* The ledger fields one finished turn must carry into the session store, so
|
|
290
|
+
* the rolling total can be REBUILT when the thread is reopened.
|
|
291
|
+
*
|
|
292
|
+
* Without this the rolling meter was a one-window illusion: the store kept
|
|
293
|
+
* `{role, content}` only, so `rollMessages` found no `tokens` on any row this
|
|
294
|
+
* window had written and a reopened thread came back as a stack of
|
|
295
|
+
* unaccounted turns while the CLI — whose `recordExchange` does write `tokens`
|
|
296
|
+
* and `costUsd` into the very same file — came back with its full total. That
|
|
297
|
+
* asymmetry is the accounting difference, not the rendering of it.
|
|
298
|
+
*
|
|
299
|
+
* Nothing is written for a turn that reported no usage: a fabricated
|
|
300
|
+
* `{input: 0, output: 0}` row would read as a measured zero forever after,
|
|
301
|
+
* which is the one lie the token meter was built to avoid.
|
|
302
|
+
*/
|
|
303
|
+
function ledgerFields(usage, model, turn) {
|
|
304
|
+
const fields = {};
|
|
305
|
+
if (turn && turn.tokens != null) fields.tokens = usageBuckets(usage);
|
|
306
|
+
if (turn && turn.cost != null && turn.real) fields.costUsd = turn.cost;
|
|
307
|
+
if (model) fields.model = model;
|
|
308
|
+
return fields;
|
|
309
|
+
}
|
|
310
|
+
|
|
232
311
|
// ---------------------------------------------------------- discovery lane
|
|
233
312
|
//
|
|
234
313
|
// The chat flow reads vertically: one prompt, one answer, forever. That makes
|
|
@@ -1977,6 +2056,15 @@ async function spawnPath(card, spec) {
|
|
|
1977
2056
|
// text nor a tool call would trigger the empty-turn recovery — an
|
|
1978
2057
|
// extra dispatch this lane has no gathered context to justify.
|
|
1979
2058
|
tools: false,
|
|
2059
|
+
// `singlePass` is what makes that promise true on the pooled class.
|
|
2060
|
+
// The engine derives the pooled brain flag as
|
|
2061
|
+
// `singlePass ? false : autonomous ? true : undefined`, and this lane
|
|
2062
|
+
// set neither: on a `nexus-brain*` id the flag was `undefined`, so the
|
|
2063
|
+
// model id's own default decided — which is the fan-out. A card that
|
|
2064
|
+
// was meant to be one cheap 1024-token pass could therefore bill a
|
|
2065
|
+
// workers + synthesis dispatch, twice per turn. Explicit `false` =
|
|
2066
|
+
// one provider call, always.
|
|
2067
|
+
singlePass: true,
|
|
1980
2068
|
sessionId: id,
|
|
1981
2069
|
},
|
|
1982
2070
|
onDelta
|
|
@@ -1993,8 +2081,29 @@ async function spawnPath(card, spec) {
|
|
|
1993
2081
|
const bits = [spec.path.title];
|
|
1994
2082
|
if (data && data.model) bits.push(data.model);
|
|
1995
2083
|
else if (spec.model) bits.push(spec.model);
|
|
1996
|
-
|
|
1997
|
-
|
|
2084
|
+
// A lane card is a dispatch the same way a turn is, and it was the
|
|
2085
|
+
// un-costed half of the 3x story: it reported tokens with no charge beside
|
|
2086
|
+
// them, so the extra calls were the least visible thing on the screen.
|
|
2087
|
+
const flow = turnAccounting(data && data.usage, spec.model, {
|
|
2088
|
+
costUsd: data && typeof data.costUsd === 'number' ? data.costUsd : undefined,
|
|
2089
|
+
});
|
|
2090
|
+
if (flow.tokens != null) bits.push(`${flow.tokens} tokens`);
|
|
2091
|
+
if (flow.cost != null) bits.push(fmtCost(flow.cost, flow.real));
|
|
2092
|
+
// …and roll it into the session, which is the half that was missing. The
|
|
2093
|
+
// card shows this ONE dispatch; a session total that skipped it would be
|
|
2094
|
+
// the lane's calls — the extra ones this feature's cost story is made of —
|
|
2095
|
+
// being the only calls on screen that never get counted. `turns: 0`
|
|
2096
|
+
// because a discovery path is not a turn the user asked for: it
|
|
2097
|
+
// contributes tokens and calls, and leaves the turn count to real
|
|
2098
|
+
// exchanges.
|
|
2099
|
+
const roll = foldRoll(spec.parentSessionId, data && data.usage, {
|
|
2100
|
+
model: spec.model,
|
|
2101
|
+
costUsd: data && typeof data.costUsd === 'number' ? data.costUsd : undefined,
|
|
2102
|
+
calls: data && data.calls,
|
|
2103
|
+
turns: 0,
|
|
2104
|
+
});
|
|
2105
|
+
const rollLine = fmtRoll(roll);
|
|
2106
|
+
if (rollLine) bits.push(`session: ${rollLine}`);
|
|
1998
2107
|
meta.textContent = bits.join(' · ');
|
|
1999
2108
|
} catch (err) {
|
|
2000
2109
|
const message = err && err.message ? err.message : String(err);
|
|
@@ -2845,6 +2954,15 @@ function openSession(id) {
|
|
|
2845
2954
|
// its on-screen transcript — continuing it as sessionId reuses the same
|
|
2846
2955
|
// id and threadMessages carries the prior turns into the next send().
|
|
2847
2956
|
currentSessionId = s.id;
|
|
2957
|
+
// A resumed thread resumes its spend too. The shared store's ledger rows
|
|
2958
|
+
// carry `tokens`/`costUsd` (client/session-store.js recordExchange writes
|
|
2959
|
+
// exactly the shape rollMessages reads), so the rolling total is rebuilt
|
|
2960
|
+
// from what was really recorded rather than restarting at zero — the
|
|
2961
|
+
// CLI's aggregateSessionUsage, which sums history.jsonl for the same
|
|
2962
|
+
// reason. A row this window wrote before ledgerFields existed carries no
|
|
2963
|
+
// `tokens` and folds as unaccounted, which is stated rather than guessed.
|
|
2964
|
+
rollsBySession.set(s.id, rollMessages(msgs));
|
|
2965
|
+
renderRollMeter(s.id);
|
|
2848
2966
|
threadMessages = msgs
|
|
2849
2967
|
.filter((m) => m.role === 'user' || m.role === 'assistant')
|
|
2850
2968
|
.map((m) => ({ role: m.role, content: m.content || m.text || '' }));
|
|
@@ -2899,6 +3017,12 @@ function newChat() {
|
|
|
2899
3017
|
currentSessionId = null;
|
|
2900
3018
|
threadMessages = [];
|
|
2901
3019
|
flowCount = 0;
|
|
3020
|
+
// A fresh thread opens with an empty meter. The outgoing session's roll is
|
|
3021
|
+
// left in the map (reopening it rebuilds from the store anyway), but the
|
|
3022
|
+
// topbar must not keep showing the thread the user just left — `sessionId`
|
|
3023
|
+
// nulls out here and the new id is minted on the first send, so nothing is
|
|
3024
|
+
// hidden that will not reappear with this thread's own number.
|
|
3025
|
+
renderRollMeter(null);
|
|
2902
3026
|
}
|
|
2903
3027
|
|
|
2904
3028
|
/**
|
|
@@ -3071,12 +3195,41 @@ async function send() {
|
|
|
3071
3195
|
// class it is what sized the call, and a user cannot tell a 16k turn from a
|
|
3072
3196
|
// 64k one by looking at the answer.
|
|
3073
3197
|
else if (effort) bits.push(`effort: ${effort}`);
|
|
3074
|
-
|
|
3075
|
-
|
|
3198
|
+
// Tokens AND what they cost, on the CLI's rule: a pooled turn's settled
|
|
3199
|
+
// charge (`costUsd`) is reported verbatim, and only a turn without one is
|
|
3200
|
+
// priced from the local rate table and marked an estimate. Printing tokens
|
|
3201
|
+
// alone left this surface with no comparable meter at all — the desktop
|
|
3202
|
+
// and the CLI run the same engine, so a gap between them had to be
|
|
3203
|
+
// measured on the same prompt, and one side was not measuring.
|
|
3204
|
+
const turn = turnAccounting(data && data.usage, model, {
|
|
3205
|
+
costUsd: data && typeof data.costUsd === 'number' ? data.costUsd : undefined,
|
|
3206
|
+
});
|
|
3207
|
+
if (turn.tokens != null) bits.push(`tokens: ${turn.tokens}`);
|
|
3208
|
+
if (turn.cost != null) bits.push(fmtCost(turn.cost, turn.real));
|
|
3209
|
+
// …AND the running session total beside it, which is the number the CLI
|
|
3210
|
+
// prints. The per-turn figure answers "what did that call cost"; only the
|
|
3211
|
+
// rolling one answers "what has this conversation cost", and it was the
|
|
3212
|
+
// missing half. Folded here rather than only on the meter so a turn that
|
|
3213
|
+
// reported no usage still counts as a turn (rollTurn counts before it
|
|
3214
|
+
// tests the count) instead of vanishing from the session.
|
|
3215
|
+
const roll = foldRoll(sessionId, data && data.usage, {
|
|
3216
|
+
model,
|
|
3217
|
+
costUsd: data && typeof data.costUsd === 'number' ? data.costUsd : undefined,
|
|
3218
|
+
calls: data && data.calls,
|
|
3219
|
+
});
|
|
3220
|
+
const rollLine = fmtRoll(roll);
|
|
3221
|
+
if (rollLine) bits.push(`session: ${rollLine}`);
|
|
3076
3222
|
addMessage('assistant', text, bits.join(' · ') || undefined, sessionId, toolLog);
|
|
3077
3223
|
|
|
3078
3224
|
try {
|
|
3079
|
-
await sync.append(sessionId, {
|
|
3225
|
+
await sync.append(sessionId, {
|
|
3226
|
+
role: 'assistant',
|
|
3227
|
+
content: text,
|
|
3228
|
+
// The turn's ledger fields, so this window's spend survives the window
|
|
3229
|
+
// — see ledgerFields. This is what makes the rolling meter the same
|
|
3230
|
+
// quantity after a reopen as it was before one.
|
|
3231
|
+
...ledgerFields(data && data.usage, model, turn),
|
|
3232
|
+
});
|
|
3080
3233
|
await sync.save({ id: sessionId, title: prompt.slice(0, 60) });
|
|
3081
3234
|
} catch {
|
|
3082
3235
|
/* persistence is non-fatal */
|
|
@@ -3206,9 +3359,17 @@ async function init() {
|
|
|
3206
3359
|
});
|
|
3207
3360
|
updateAutonomousControlsVisibility();
|
|
3208
3361
|
|
|
3209
|
-
// The discovery lane is opt-in
|
|
3362
|
+
// The discovery lane is opt-in, remembered across restarts.
|
|
3363
|
+
//
|
|
3364
|
+
// This used to read `savedExplore === 'off'` against a checkbox that shipped
|
|
3365
|
+
// `checked` in index.html, i.e. the opposite of the comment above it: a fresh
|
|
3366
|
+
// install (no `aegis.explore` key, and nothing had ever written one) landed
|
|
3367
|
+
// with the lane ON and billed two extra model calls after every single turn —
|
|
3368
|
+
// ~3× the tokens of a plain reply, silently, because the lane is
|
|
3369
|
+
// fire-and-forget and never looks slow. Only an explicit 'on' turns it on
|
|
3370
|
+
// now; every other state, including "never asked", means off.
|
|
3210
3371
|
const savedExplore = localStorage.getItem(EXPLORE_KEY);
|
|
3211
|
-
|
|
3372
|
+
els.exploreToggle.checked = savedExplore === 'on';
|
|
3212
3373
|
els.exploreToggle.addEventListener('change', () => {
|
|
3213
3374
|
localStorage.setItem(EXPLORE_KEY, els.exploreToggle.checked ? 'on' : 'off');
|
|
3214
3375
|
if (!els.exploreToggle.checked) abortBranches();
|
package/renderer/index.html
CHANGED
|
@@ -22,6 +22,11 @@
|
|
|
22
22
|
<span class="dot" id="conn-dot"></span>
|
|
23
23
|
<span id="conn-text">connecting…</span>
|
|
24
24
|
</div>
|
|
25
|
+
<!-- The rolling session token total, counted the way the CLI counts it:
|
|
26
|
+
every finished turn folded into one running number rather than a
|
|
27
|
+
per-turn count that resets. Hidden until a turn has been accounted
|
|
28
|
+
for, so an untouched thread never shows a `0 tok` it never measured. -->
|
|
29
|
+
<div class="session-meter" id="session-meter" hidden></div>
|
|
25
30
|
<div class="spacer"></div>
|
|
26
31
|
<button type="button" id="new-chat" class="ghost-btn">New chat</button>
|
|
27
32
|
</header>
|
|
@@ -353,9 +358,12 @@
|
|
|
353
358
|
<label
|
|
354
359
|
class="flow-toggle"
|
|
355
360
|
for="explore-toggle"
|
|
356
|
-
title="After every answer, send the AI down two extra paths — an alternative angle and an unexpected discovery — shown as a horizontal discovery lane."
|
|
361
|
+
title="Optional, off by default. After every answer, send the AI down two extra paths — an alternative angle and an unexpected discovery — shown as a horizontal discovery lane. This costs two extra model calls per turn (roughly 3× the tokens of a plain reply)."
|
|
357
362
|
>
|
|
358
|
-
|
|
363
|
+
<!-- Unchecked in the markup on purpose: the lane bills two extra
|
|
364
|
+
model calls per turn, so it is opt-in. app.js only ticks this
|
|
365
|
+
when the user has explicitly saved 'on' (aegis.explore). -->
|
|
366
|
+
<input type="checkbox" id="explore-toggle" />
|
|
359
367
|
<span>discovery lane</span>
|
|
360
368
|
</label>
|
|
361
369
|
<button type="submit" id="send">Send</button>
|
package/renderer/style.css
CHANGED
|
@@ -114,6 +114,28 @@ body {
|
|
|
114
114
|
box-shadow: 0 0 6px var(--teal);
|
|
115
115
|
}
|
|
116
116
|
|
|
117
|
+
/* The rolling session token total — the CLI's status-bar figure, in the
|
|
118
|
+
topbar. Muted and monospaced so it reads as a meter rather than a heading,
|
|
119
|
+
and it never wraps: a number that reflows the topbar every turn is worse
|
|
120
|
+
than no number at all. Hidden via the `hidden` attribute, so the rule below
|
|
121
|
+
has to win against the flex display the base class would otherwise get. */
|
|
122
|
+
.session-meter {
|
|
123
|
+
font-size: 12px;
|
|
124
|
+
font-family: var(--font-mono);
|
|
125
|
+
color: var(--text2);
|
|
126
|
+
white-space: nowrap;
|
|
127
|
+
overflow: hidden;
|
|
128
|
+
text-overflow: ellipsis;
|
|
129
|
+
max-width: 40ch;
|
|
130
|
+
padding: 2px 8px;
|
|
131
|
+
border: 1px solid var(--border);
|
|
132
|
+
border-radius: 999px;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
.session-meter[hidden] {
|
|
136
|
+
display: none;
|
|
137
|
+
}
|
|
138
|
+
|
|
117
139
|
.spacer {
|
|
118
140
|
flex: 1;
|
|
119
141
|
}
|
package/renderer/usage.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
'use strict';
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
|
-
* Pure token-usage → display-number mapping.
|
|
4
|
+
* Pure token-usage → display-number mapping, and the cost half of it.
|
|
5
5
|
*
|
|
6
6
|
* Standalone from app.js (same reason as budget.js/stream-policy.js): it is
|
|
7
7
|
* requireable from a plain Node test without window.aegis. app.js only calls
|
|
@@ -18,6 +18,27 @@
|
|
|
18
18
|
* the call was silently being billed. Accepting both spellings, and deriving
|
|
19
19
|
* the total when the provider doesn't state one, is what makes the spend
|
|
20
20
|
* visible regardless of which endpoint answered.
|
|
21
|
+
*
|
|
22
|
+
* ── The cost half, on the CLI's principle ──────────────────────────────────
|
|
23
|
+
*
|
|
24
|
+
* The renderer printed a token count and nothing else, so the desktop had no
|
|
25
|
+
* meter to compare against `aegiscodex` on the same engine and the same
|
|
26
|
+
* prompt — the comparison that started this whole line of work. The CLI's
|
|
27
|
+
* rule (cli/src/tokens.js, ported here) has three parts, and all three matter:
|
|
28
|
+
*
|
|
29
|
+
* 1. A pooled turn's bill is settled SERVER-side and carries a margin and a
|
|
30
|
+
* prompt-cache discount the client cannot see. When the response states
|
|
31
|
+
* `costUsd`, that figure wins verbatim — it is the truth about the
|
|
32
|
+
* charge, not an estimate of it.
|
|
33
|
+
* 2. Only in the absence of a settled charge does the local rate table
|
|
34
|
+
* apply, and the result is labeled an estimate so nobody reads a guess as
|
|
35
|
+
* a bill.
|
|
36
|
+
* 3. The rate table is resolved by exact id, then longest prefix. The
|
|
37
|
+
* DeepSeek row is load-bearing: without it every direct DeepSeek turn
|
|
38
|
+
* fell through to Sonnet's $3.00/$15.00 per M against a real
|
|
39
|
+
* $0.14/$0.28 — 21x the input rate and 54x the output rate — which made
|
|
40
|
+
* the *meter* the largest single contributor to the apparent cost gap
|
|
41
|
+
* between two surfaces running identical code.
|
|
21
42
|
*/
|
|
22
43
|
|
|
23
44
|
/**
|
|
@@ -25,18 +46,352 @@
|
|
|
25
46
|
* reported none (an unknown count must render as nothing, never as `0`).
|
|
26
47
|
*
|
|
27
48
|
* @param {{total_tokens?: number, prompt_tokens?: number, completion_tokens?: number,
|
|
28
|
-
* input_tokens?: number, output_tokens?: number
|
|
49
|
+
* input_tokens?: number, output_tokens?: number,
|
|
50
|
+
* input?: number, output?: number}|null|undefined} usage
|
|
29
51
|
* @returns {number|null}
|
|
30
52
|
*/
|
|
31
53
|
function usageTokens(usage) {
|
|
32
54
|
if (!usage || typeof usage !== 'object') return null;
|
|
33
55
|
if (typeof usage.total_tokens === 'number') return usage.total_tokens;
|
|
34
|
-
|
|
35
|
-
|
|
56
|
+
// Three spellings of the same quantity: OpenAI's, Anthropic's, and this
|
|
57
|
+
// file's own bucket shape — which is also the shape a LEDGER row carries
|
|
58
|
+
// (cli/src/history.js writes `{input, output, cacheRead, cacheWrite}` into
|
|
59
|
+
// sessions.json, and the CLI's demo path writes `{input, output}`). Reading
|
|
60
|
+
// only the two wire spellings left every stored row uncountable, so a
|
|
61
|
+
// resumed session's rolling total started at zero even though its ledger
|
|
62
|
+
// said otherwise.
|
|
63
|
+
const input = usage.input_tokens ?? usage.prompt_tokens ?? usage.input;
|
|
64
|
+
const output = usage.output_tokens ?? usage.completion_tokens ?? usage.output;
|
|
36
65
|
if (typeof input !== 'number' && typeof output !== 'number') return null;
|
|
37
66
|
return (input || 0) + (output || 0);
|
|
38
67
|
}
|
|
39
68
|
|
|
69
|
+
// Per-million-token USD rates. Cache-read/write matter for long sessions.
|
|
70
|
+
//
|
|
71
|
+
// These are the PROVIDER's rates, for the fallback path where a turn has no
|
|
72
|
+
// server-settled charge (a direct provider, ollama, a custom endpoint). A
|
|
73
|
+
// pooled turn reports the ledger figure instead — see turnAccounting's
|
|
74
|
+
// `costUsd` — because the pool's bill carries a margin and a prompt-cache
|
|
75
|
+
// discount this table cannot see.
|
|
76
|
+
//
|
|
77
|
+
// Keys are matched by exact id first, then by prefix (ratesFor below), so a
|
|
78
|
+
// family row covers every dated variant of it.
|
|
79
|
+
const RATES = {
|
|
80
|
+
sonnet: { input: 3.00, output: 15.00, cacheRead: 0.30, cacheWrite: 3.75 },
|
|
81
|
+
default: { input: 3.00, output: 15.00, cacheRead: 0.30, cacheWrite: 3.75 },
|
|
82
|
+
fable: { input: 5.00, output: 25.00, cacheRead: 0.50, cacheWrite: 6.25 },
|
|
83
|
+
opus: { input: 5.00, output: 25.00, cacheRead: 0.50, cacheWrite: 6.25 },
|
|
84
|
+
// DeepSeek V4 — the provider's published rate, independently corroborated by
|
|
85
|
+
// aegis1 services/nexus_provider/catalog.py (cost_per_1k_input=0.00014,
|
|
86
|
+
// cost_per_1k_output=0.00028) and referenced in services/pricing.py.
|
|
87
|
+
//
|
|
88
|
+
// This row was MISSING upstream, and usageCost fell through to RATES.sonnet
|
|
89
|
+
// for every DeepSeek turn: $3.00/$15.00 per M against a real $0.14/$0.28 —
|
|
90
|
+
// 21x the input rate and 54x the output rate, on every direct DeepSeek call.
|
|
91
|
+
//
|
|
92
|
+
// cacheRead is 0.1x input (aegis1 services/pricing.py
|
|
93
|
+
// CACHE_READ_FRACTION_BY_COMPANY["deepseek"] = 0.1 -> $0.014/M).
|
|
94
|
+
// cacheWrite is the input rate: DeepSeek bills a cache write as ordinary
|
|
95
|
+
// input tokens and charges no separate write premium, unlike Anthropic's
|
|
96
|
+
// 1.25x. A zero here would understate a session that writes cache.
|
|
97
|
+
deepseek: { input: 0.14, output: 0.28, cacheRead: 0.014, cacheWrite: 0.14 },
|
|
98
|
+
};
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* The rate row for a model id, provider, or alias. Exact id wins, then the
|
|
102
|
+
* longest matching prefix, then the Sonnet-class default.
|
|
103
|
+
*/
|
|
104
|
+
function ratesFor(model) {
|
|
105
|
+
const id = String(model || '').toLowerCase();
|
|
106
|
+
if (!id) return RATES.default;
|
|
107
|
+
if (RATES[id]) return RATES[id];
|
|
108
|
+
let best = null;
|
|
109
|
+
let bestLen = 0;
|
|
110
|
+
for (const [prefix, rates] of Object.entries(RATES)) {
|
|
111
|
+
if (id.startsWith(prefix) && prefix.length > bestLen) {
|
|
112
|
+
best = rates;
|
|
113
|
+
bestLen = prefix.length;
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
return best || RATES.default;
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* The four billable buckets from a wire usage object, accepting both provider
|
|
121
|
+
* spellings. Cache fields have no Anthropic-compatible short form here because
|
|
122
|
+
* the desktop's transport normalises them (desktop/lib/local/providers.js).
|
|
123
|
+
*
|
|
124
|
+
* @param {object|null|undefined} usage
|
|
125
|
+
* @returns {{input: number, output: number, cacheRead: number, cacheWrite: number}}
|
|
126
|
+
*/
|
|
127
|
+
function usageBuckets(usage) {
|
|
128
|
+
const u = usage && typeof usage === 'object' ? usage : {};
|
|
129
|
+
const num = (...candidates) => {
|
|
130
|
+
for (const c of candidates) if (typeof c === 'number') return c;
|
|
131
|
+
return 0;
|
|
132
|
+
};
|
|
133
|
+
const input = num(u.input_tokens, u.prompt_tokens, u.input);
|
|
134
|
+
const output = num(u.output_tokens, u.completion_tokens, u.output);
|
|
135
|
+
// A stated total larger than the split means the provider counted tokens the
|
|
136
|
+
// split does not name (thinking, cached reads). Attribute the remainder to
|
|
137
|
+
// input rather than dropping it: dropping it would understate the bill.
|
|
138
|
+
const total = num(u.total_tokens);
|
|
139
|
+
const cacheRead = num(u.cache_read_input_tokens, u.cacheRead);
|
|
140
|
+
const cacheWrite = num(u.cache_creation_input_tokens, u.cacheWrite);
|
|
141
|
+
const accounted = input + output + cacheRead + cacheWrite;
|
|
142
|
+
return {
|
|
143
|
+
input: total > accounted ? input + (total - accounted) : input,
|
|
144
|
+
output,
|
|
145
|
+
cacheRead,
|
|
146
|
+
cacheWrite,
|
|
147
|
+
};
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/** Dollar cost of a usage record at the given model's rates (USD, estimate). */
|
|
151
|
+
function usageCost(usage, model = 'sonnet') {
|
|
152
|
+
const r = ratesFor(model);
|
|
153
|
+
const u = usageBuckets(usage);
|
|
154
|
+
const toD = (n, rate) => (n / 1_000_000) * rate;
|
|
155
|
+
return toD(u.input, r.input)
|
|
156
|
+
+ toD(u.output, r.output)
|
|
157
|
+
+ toD(u.cacheRead, r.cacheRead)
|
|
158
|
+
+ toD(u.cacheWrite, r.cacheWrite);
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
/**
|
|
162
|
+
* What one turn cost, as the CLI reports it: the server's settled charge when
|
|
163
|
+
* the response carries one, otherwise the rate table's estimate, always with
|
|
164
|
+
* the distinction preserved so it can be labeled.
|
|
165
|
+
*
|
|
166
|
+
* @param {object|null|undefined} usage the response's `usage` object
|
|
167
|
+
* @param {string} model the model id that answered
|
|
168
|
+
* @param {{costUsd?: number}} [opts] the server-settled charge, if any
|
|
169
|
+
* @returns {{tokens: number|null, cost: number|null, real: boolean, estimated: boolean}}
|
|
170
|
+
* `real` is true only when `cost` is the settled charge. `cost` is
|
|
171
|
+
* `null` when nothing was reported and no model was named — an
|
|
172
|
+
* unpriced turn must render as nothing, never as $0.0000.
|
|
173
|
+
*/
|
|
174
|
+
function turnAccounting(usage, model, opts = {}) {
|
|
175
|
+
const tokens = usageTokens(usage);
|
|
176
|
+
const u = usage && typeof usage === 'object' ? usage : {};
|
|
177
|
+
// The settled charge can arrive either on the response or folded into the
|
|
178
|
+
// usage object — aegiscodex-dev/src/main.js does the latter
|
|
179
|
+
// (`{ ...result.usage, costUsd: result.costUsd }`), so both are accepted.
|
|
180
|
+
const settled = typeof opts.costUsd === 'number'
|
|
181
|
+
? opts.costUsd
|
|
182
|
+
: (typeof u.costUsd === 'number' ? u.costUsd : undefined);
|
|
183
|
+
if (typeof settled === 'number') {
|
|
184
|
+
return { tokens, cost: settled, real: true, estimated: false };
|
|
185
|
+
}
|
|
186
|
+
if (tokens == null) return { tokens, cost: null, real: false, estimated: false };
|
|
187
|
+
const priced = usageCost(u, model);
|
|
188
|
+
// A model the table cannot place is still priced at the Sonnet-class default
|
|
189
|
+
// (ratesFor never returns nothing), so this is always an estimate.
|
|
190
|
+
return { tokens, cost: priced, real: false, estimated: true };
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* ── The rolling session tallies, on the CLI's rule ─────────────────────────
|
|
195
|
+
*
|
|
196
|
+
* The two surfaces did not merely print different numbers — they counted
|
|
197
|
+
* differently. The CLI never shows a turn's tokens in isolation: `recordTurn`
|
|
198
|
+
* (cli/src/app.js) folds each finished turn's usage into one `session` object
|
|
199
|
+
* — `tokens`, `inputTokens`, `outputTokens`, `calls` — and what the user reads
|
|
200
|
+
* back is that RUNNING TOTAL. The status bar prints `state.tokens`
|
|
201
|
+
* (renderStatus), and `ctrl+t` prints the tallies in one line
|
|
202
|
+
* (`tokenSummary`: `12,400 tok (10,100 in / 2,300 out) · 4 calls · €0.03`).
|
|
203
|
+
*
|
|
204
|
+
* The desktop counted per turn only. Every meta row was a fresh count that
|
|
205
|
+
* reset at the next call, so "what has this session spent" was answerable only
|
|
206
|
+
* by adding the rows up by eye across a scrollback — which is most of why the
|
|
207
|
+
* desktop looked like it accounted differently from the CLI on identical
|
|
208
|
+
* engine code and an identical prompt.
|
|
209
|
+
*
|
|
210
|
+
* Three properties of the CLI's fold are load-bearing and are reproduced here
|
|
211
|
+
* exactly, because dropping any one of them reintroduces a specific lie:
|
|
212
|
+
*
|
|
213
|
+
* 1. `turns` and `calls` are incremented BEFORE the "did usage come back?"
|
|
214
|
+
* gate (recordTurn counts first, then tests `tokens != null`). A turn
|
|
215
|
+
* that reported nothing still happened; a tally that skipped it would
|
|
216
|
+
* report the session as shorter and cheaper than it was.
|
|
217
|
+
* 2. `tokens`, `input` and `output` accumulate — never reset. A rolling
|
|
218
|
+
* total that resets per turn is the per-turn count it replaced.
|
|
219
|
+
* 3. A `null` count contributes nothing and is counted in `unknown`, so
|
|
220
|
+
* `tokens: 0` is only ever read as a real zero. Nothing is ever added as
|
|
221
|
+
* a fabricated 0 to make the arithmetic look complete.
|
|
222
|
+
*
|
|
223
|
+
* The money split is the CLI's too: a settled charge (the pool's ledger
|
|
224
|
+
* figure) rolls into `cost`, and a locally priced turn rolls into `estimate`.
|
|
225
|
+
* They are kept apart rather than summed so a `~`-estimate can never be read
|
|
226
|
+
* as part of the bill — the distinction `fmtCost` marks on a single turn, held
|
|
227
|
+
* across the session.
|
|
228
|
+
*/
|
|
229
|
+
|
|
230
|
+
/** A session with nothing accounted for yet. */
|
|
231
|
+
function emptyRoll() {
|
|
232
|
+
return {
|
|
233
|
+
turns: 0,
|
|
234
|
+
calls: 0,
|
|
235
|
+
tokens: 0,
|
|
236
|
+
input: 0,
|
|
237
|
+
output: 0,
|
|
238
|
+
cacheRead: 0,
|
|
239
|
+
cacheWrite: 0,
|
|
240
|
+
/** Dispatches that reported no usage at all — the honest gap in `tokens`. */
|
|
241
|
+
unknown: 0,
|
|
242
|
+
/** Settled charges (server-settled `costUsd`), rolled. */
|
|
243
|
+
cost: 0,
|
|
244
|
+
/** Locally priced turns, rolled. Never mixed into `cost`. */
|
|
245
|
+
estimate: 0,
|
|
246
|
+
};
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
/**
|
|
250
|
+
* Fold one completed dispatch into a session's rolling tallies. Pure: it
|
|
251
|
+
* returns a NEW roll and never mutates the one it was handed, so a half-applied
|
|
252
|
+
* fold cannot exist.
|
|
253
|
+
*
|
|
254
|
+
* @param {object} [roll] the roll so far (emptyRoll() when omitted)
|
|
255
|
+
* @param {object} [usage] the response's `usage` object
|
|
256
|
+
* @param {{model?: string, costUsd?: number, calls?: number, turns?: number}} [opts]
|
|
257
|
+
* `calls` defaults to 1; a pooled turn may report how many provider
|
|
258
|
+
* calls it actually made. `turns: 0` folds a dispatch that is not a
|
|
259
|
+
* turn of its own — the discovery-lane card, which bills like any other
|
|
260
|
+
* call but is not something the user asked for.
|
|
261
|
+
* @returns {object} the new roll
|
|
262
|
+
*/
|
|
263
|
+
function rollTurn(roll, usage, opts = {}) {
|
|
264
|
+
const next = Object.assign(emptyRoll(), roll || {});
|
|
265
|
+
next.turns += opts.turns === undefined ? 1 : Number(opts.turns) || 0;
|
|
266
|
+
next.calls += opts.calls === undefined ? 1 : Number(opts.calls) || 0;
|
|
267
|
+
const turn = turnAccounting(usage, opts.model, { costUsd: opts.costUsd });
|
|
268
|
+
if (turn.tokens == null) {
|
|
269
|
+
next.unknown += 1;
|
|
270
|
+
return next;
|
|
271
|
+
}
|
|
272
|
+
const b = usageBuckets(usage);
|
|
273
|
+
next.tokens += turn.tokens;
|
|
274
|
+
next.input += b.input;
|
|
275
|
+
next.output += b.output;
|
|
276
|
+
next.cacheRead += b.cacheRead;
|
|
277
|
+
next.cacheWrite += b.cacheWrite;
|
|
278
|
+
if (turn.real) next.cost += turn.cost;
|
|
279
|
+
else if (turn.cost != null) next.estimate += turn.cost;
|
|
280
|
+
return next;
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
/**
|
|
284
|
+
* Rebuild a session's rolling total from stored exchanges — the desktop's
|
|
285
|
+
* counterpart of the CLI's `aggregateSessionUsage`, which sums history.jsonl so
|
|
286
|
+
* a resumed session (and a compacted one) still reports everything it spent.
|
|
287
|
+
*
|
|
288
|
+
* Reads the shapes the shared store writes: an assistant message carrying
|
|
289
|
+
* `tokens: {input, output, cacheRead, cacheWrite}` and, when the pool settled
|
|
290
|
+
* the turn, `costUsd` (cli/src/history.js → session-store.recordExchange).
|
|
291
|
+
* Messages the desktop itself appended carry no `tokens` and fold as `unknown`
|
|
292
|
+
* — a resumed thread states what is known and does not invent the rest.
|
|
293
|
+
*
|
|
294
|
+
* @param {Array<{role?: string, tokens?: object, costUsd?: number, model?: string}>} [messages]
|
|
295
|
+
* @returns {object} the roll
|
|
296
|
+
*/
|
|
297
|
+
function rollMessages(messages) {
|
|
298
|
+
let roll = emptyRoll();
|
|
299
|
+
for (const m of Array.isArray(messages) ? messages : []) {
|
|
300
|
+
if (!m || m.role !== 'assistant') continue;
|
|
301
|
+
roll = rollTurn(roll, m.tokens, {
|
|
302
|
+
model: m.model,
|
|
303
|
+
costUsd: typeof m.costUsd === 'number' ? m.costUsd : undefined,
|
|
304
|
+
calls: m.calls,
|
|
305
|
+
});
|
|
306
|
+
}
|
|
307
|
+
return roll;
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
/**
|
|
311
|
+
* The CLI's rendering of a session tally — `cli/src/format.js fmtTokens`, which
|
|
312
|
+
* is the one `tokenSummary` actually imports (`cli/src/app.js:50`). NOT the
|
|
313
|
+
* `1.5k`/`12.3k` form in `cli/src/tokens.js`: that one belongs to the /cost
|
|
314
|
+
* panels, and using it here would print `12.4k` on the very total the CLI
|
|
315
|
+
* prints as `12,400` — a rendering difference stacked on top of the accounting
|
|
316
|
+
* difference this change exists to remove.
|
|
317
|
+
*
|
|
318
|
+
* Comma-grouped integer. Anything not finite and positive renders `0`, which is
|
|
319
|
+
* the CLI's rule and keeps a stray NaN from reaching the topbar.
|
|
320
|
+
*/
|
|
321
|
+
function fmtTokens(n) {
|
|
322
|
+
const v = Number(n);
|
|
323
|
+
if (!Number.isFinite(v) || v <= 0) return '0';
|
|
324
|
+
return Math.round(v)
|
|
325
|
+
.toString()
|
|
326
|
+
.replace(/\B(?=(\d{3})+(?!\d))/g, ',');
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
/**
|
|
330
|
+
* The rolling session total as one line — the desktop's counterpart of the
|
|
331
|
+
* CLI's `tokenSummary`. Empty string when nothing has been accounted for, so
|
|
332
|
+
* an untouched session adds no noise to a turn's meta line.
|
|
333
|
+
*
|
|
334
|
+
* @param {object} [roll]
|
|
335
|
+
* @returns {string} e.g. `12,400 tok (10,100 in / 2,300 out) · 4 calls · $0.0310`
|
|
336
|
+
*/
|
|
337
|
+
function fmtRoll(roll) {
|
|
338
|
+
const r = roll || emptyRoll();
|
|
339
|
+
if (!r.turns && !r.calls) return '';
|
|
340
|
+
// The first two fields are `tokenSummary` (cli/src/app.js:1413) verbatim,
|
|
341
|
+
// separator and all: `12,400 tok (10,100 in / 2,300 out) · 4 calls`. There the
|
|
342
|
+
// parenthetical is unconditional and the call count is singular at one; both
|
|
343
|
+
// are kept, because a line that is only *sometimes* shaped like the CLI's is a
|
|
344
|
+
// lookalike rather than the same quantity. The call count is also the number
|
|
345
|
+
// that reveals a fan-out, which is why it is not hidden at 1.
|
|
346
|
+
const bits = [
|
|
347
|
+
`${fmtTokens(r.tokens)} tok (${fmtTokens(r.input)} in / ${fmtTokens(r.output)} out)`,
|
|
348
|
+
`${r.calls} call${r.calls === 1 ? '' : 's'}`,
|
|
349
|
+
];
|
|
350
|
+
// Money is where this line departs from `tokenSummary`, deliberately: that
|
|
351
|
+
// one sums a single `session.cost` in EUR via fmtEur, while this surface keeps
|
|
352
|
+
// a settled charge and a local estimate apart so a `~`-estimate can never be
|
|
353
|
+
// read as part of the bill. Desktop's pre-existing fmtCost renders both, and
|
|
354
|
+
// its `$` convention is left exactly as it was.
|
|
355
|
+
if (r.cost > 0 || r.estimate > 0) {
|
|
356
|
+
const money = [];
|
|
357
|
+
if (r.cost > 0) money.push(fmtCost(r.cost, true));
|
|
358
|
+
if (r.estimate > 0) money.push(fmtCost(r.estimate, false));
|
|
359
|
+
bits.push(money.join(' + '));
|
|
360
|
+
}
|
|
361
|
+
// No counterpart in `tokenSummary`, which folds a usage-less turn silently and
|
|
362
|
+
// so reports a total short of the truth without saying so. Named here instead:
|
|
363
|
+
// the count appears only when something went unreported, and it never changes
|
|
364
|
+
// a number — it only says the number is not the whole story.
|
|
365
|
+
if (r.unknown) bits.push(`${r.unknown} unrpt`);
|
|
366
|
+
return bits.join(' · ');
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
/**
|
|
370
|
+
* A cost for display. Estimates are marked with `~` so an estimate is never
|
|
371
|
+
* mistaken for a settled charge.
|
|
372
|
+
*
|
|
373
|
+
* @param {number} cost
|
|
374
|
+
* @param {boolean} [real]
|
|
375
|
+
* @returns {string}
|
|
376
|
+
*/
|
|
377
|
+
function fmtCost(cost, real) {
|
|
378
|
+
if (typeof cost !== 'number' || !isFinite(cost)) return '';
|
|
379
|
+
return `${real ? '' : '~'}$${cost.toFixed(4)}`;
|
|
380
|
+
}
|
|
381
|
+
|
|
40
382
|
if (typeof module !== 'undefined' && module.exports) {
|
|
41
|
-
module.exports = {
|
|
383
|
+
module.exports = {
|
|
384
|
+
RATES,
|
|
385
|
+
usageTokens,
|
|
386
|
+
ratesFor,
|
|
387
|
+
usageBuckets,
|
|
388
|
+
usageCost,
|
|
389
|
+
turnAccounting,
|
|
390
|
+
emptyRoll,
|
|
391
|
+
rollTurn,
|
|
392
|
+
rollMessages,
|
|
393
|
+
fmtTokens,
|
|
394
|
+
fmtRoll,
|
|
395
|
+
fmtCost,
|
|
396
|
+
};
|
|
42
397
|
}
|