aegis-desktop 0.7.2 → 0.7.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/renderer/app.js +211 -13
- package/renderer/index.html +5 -0
- package/renderer/style.css +22 -0
- package/renderer/usage.js +516 -5
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "aegis-desktop",
|
|
3
3
|
"productName": "AEGIS Desktop",
|
|
4
|
-
"version": "0.7.
|
|
4
|
+
"version": "0.7.4",
|
|
5
5
|
"description": "Thin Electron host for AEGIS — a local chat UI over the shared client/aegis.js transport. Ships transport + UI only; engine logic stays server-side.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "AEGIS Code",
|
package/renderer/app.js
CHANGED
|
@@ -14,12 +14,13 @@
|
|
|
14
14
|
* wrong.
|
|
15
15
|
*
|
|
16
16
|
* `budgetFor`/`maxTokensCeiling`/`FLAT_CEILING`/`EFFORT_TOKEN_BUDGET` come from
|
|
17
|
-
* budget.js and `
|
|
18
|
-
* before this one (see index.html). There is no
|
|
19
|
-
* surface at all: the ceiling is display-only (what a
|
|
20
|
-
* limit is, reported in the Model hint) and
|
|
21
|
-
* Effort rung — what a request actually travels
|
|
22
|
-
* displayed-number mapping stays unit-testable without
|
|
17
|
+
* budget.js and `turnAccounting`/`fmtCost` from usage.js, sibling classic
|
|
18
|
+
* scripts loaded before this one (see index.html). There is no
|
|
19
|
+
* max-tokens control on this surface at all: the ceiling is display-only (what a
|
|
20
|
+
* model says its own output limit is, reported in the Model hint) and
|
|
21
|
+
* `budgetFor` answers — from the Effort rung — what a request actually travels
|
|
22
|
+
* with. The token-usage → displayed-number mapping stays unit-testable without
|
|
23
|
+
* window.aegis/models.
|
|
23
24
|
*/
|
|
24
25
|
|
|
25
26
|
// Everything below runs inside an IIFE. preload.js's contextBridge.exposeInMainWorld
|
|
@@ -105,6 +106,7 @@ const ELEMENT_IDS = {
|
|
|
105
106
|
sessionsHint: 'sessions-hint',
|
|
106
107
|
syncNow: 'sync-now',
|
|
107
108
|
syncStatus: 'sync-status',
|
|
109
|
+
sessionMeter: 'session-meter',
|
|
108
110
|
newChat: 'new-chat',
|
|
109
111
|
memorySearchForm: 'memory-search-form',
|
|
110
112
|
memoryQuery: 'memory-query',
|
|
@@ -229,6 +231,93 @@ let threadMessages = [];
|
|
|
229
231
|
let classOptions = [];
|
|
230
232
|
let modelMeta = new Map(); // model id -> raw model object from listModels() (P2 §6.3 ceiling)
|
|
231
233
|
|
|
234
|
+
// ------------------------------------------------------- rolling token meter
|
|
235
|
+
//
|
|
236
|
+
// Tokens are accounted the way the CLI accounts them: as a RUNNING SESSION
|
|
237
|
+
// TOTAL, folded turn by turn, not as a per-turn number that resets at the next
|
|
238
|
+
// call. The CLI's `recordTurn` (cli/src/app.js) folds every finished turn into
|
|
239
|
+
// one `session` object and prints that total in the status bar and in `ctrl+t`;
|
|
240
|
+
// this surface printed each turn's count and nothing else, so the only way to
|
|
241
|
+
// answer "what has this session spent" was to add the rows up by eye across the
|
|
242
|
+
// scrollback. That is the accounting difference between two surfaces running
|
|
243
|
+
// the same engine on the same prompt.
|
|
244
|
+
//
|
|
245
|
+
// Keyed by sessionId rather than held in one global, because the window holds
|
|
246
|
+
// several sessions across its life: switching threads must not carry one
|
|
247
|
+
// thread's spend into another's meter, and resuming a thread must not start its
|
|
248
|
+
// total at zero. `rollMessages` rebuilds a resumed session's total from the
|
|
249
|
+
// ledger rows the shared store already keeps.
|
|
250
|
+
const rollsBySession = new Map();
|
|
251
|
+
|
|
252
|
+
/** The rolling tallies for a session — empty, never undefined, when unseen. */
|
|
253
|
+
function rollFor(sessionId) {
|
|
254
|
+
const id = sessionId || '';
|
|
255
|
+
if (!rollsBySession.has(id)) rollsBySession.set(id, emptyRoll());
|
|
256
|
+
return rollsBySession.get(id);
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
/**
|
|
260
|
+
* Fold one finished dispatch into its session's rolling total and refresh the
|
|
261
|
+
* meter. Returns the turn's own accounting so the caller can still print the
|
|
262
|
+
* per-turn figure beside the session total (the CLI shows both: the meta row's
|
|
263
|
+
* `1.5k tok` and the status bar's rolling total).
|
|
264
|
+
*/
|
|
265
|
+
function foldRoll(sessionId, usage, opts = {}) {
|
|
266
|
+
const id = sessionId || '';
|
|
267
|
+
const next = rollTurn(rollFor(id), usage, opts);
|
|
268
|
+
rollsBySession.set(id, next);
|
|
269
|
+
renderRollMeter(id);
|
|
270
|
+
return next;
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
/**
|
|
274
|
+
* Paint the topbar meter. Hidden while a session has accounted for nothing —
|
|
275
|
+
* an unused thread must not display a `0 tok` it never measured — and shown
|
|
276
|
+
* the moment a turn is folded in.
|
|
277
|
+
*/
|
|
278
|
+
function renderRollMeter(sessionId) {
|
|
279
|
+
const el = els.sessionMeter;
|
|
280
|
+
if (!el) return;
|
|
281
|
+
const roll = rollsBySession.get(sessionId || '');
|
|
282
|
+
const line = roll && currentSessionId === sessionId ? fmtRoll(roll) : '';
|
|
283
|
+
el.textContent = line;
|
|
284
|
+
el.hidden = !line;
|
|
285
|
+
if (line) el.title = 'This session, counted the way the CLI counts it — every turn rolled into one running total';
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/**
|
|
289
|
+
* The ledger fields one finished turn must carry into the session store, so
|
|
290
|
+
* the rolling total can be REBUILT when the thread is reopened.
|
|
291
|
+
*
|
|
292
|
+
* Without this the rolling meter was a one-window illusion: the store kept
|
|
293
|
+
* `{role, content}` only, so `rollMessages` found no `tokens` on any row this
|
|
294
|
+
* window had written and a reopened thread came back as a stack of
|
|
295
|
+
* unaccounted turns while the CLI — whose `recordExchange` does write `tokens`
|
|
296
|
+
* and `costUsd` into the very same file — came back with its full total. That
|
|
297
|
+
* asymmetry is the accounting difference, not the rendering of it.
|
|
298
|
+
*
|
|
299
|
+
* One authority writes the row: `ledgerRow` in usage.js, which mirrors the
|
|
300
|
+
* CLI's `appendHistory` shape exactly. A turn the wire did not report on is
|
|
301
|
+
* STILL written — as the CLI writes it, an estimate from the turn's own text
|
|
302
|
+
* marked `real: false` — because that is what keeps the live roll and the
|
|
303
|
+
* rebuilt roll the same number. Only a dispatch with neither reported usage
|
|
304
|
+
* nor any text to estimate from writes nothing: a fabricated
|
|
305
|
+
* `{input: 0, output: 0}` row would read as a measured zero forever after,
|
|
306
|
+
* which is the one lie the token meter was built to avoid.
|
|
307
|
+
*/
|
|
308
|
+
function ledgerFields(usage, model, turn, text) {
|
|
309
|
+
const row = ledgerRow(usage, turn, {
|
|
310
|
+
model,
|
|
311
|
+
costUsd: turn && turn.real && typeof turn.cost === 'number' ? turn.cost : undefined,
|
|
312
|
+
calls: text && text.calls,
|
|
313
|
+
prompt: text && text.prompt,
|
|
314
|
+
reply: text && text.reply,
|
|
315
|
+
});
|
|
316
|
+
const fields = row ? Object.assign({}, row) : {};
|
|
317
|
+
if (model) fields.model = model;
|
|
318
|
+
return fields;
|
|
319
|
+
}
|
|
320
|
+
|
|
232
321
|
// ---------------------------------------------------------- discovery lane
|
|
233
322
|
//
|
|
234
323
|
// The chat flow reads vertically: one prompt, one answer, forever. That makes
|
|
@@ -2002,8 +2091,34 @@ async function spawnPath(card, spec) {
|
|
|
2002
2091
|
const bits = [spec.path.title];
|
|
2003
2092
|
if (data && data.model) bits.push(data.model);
|
|
2004
2093
|
else if (spec.model) bits.push(spec.model);
|
|
2005
|
-
|
|
2006
|
-
|
|
2094
|
+
// A lane card is a dispatch the same way a turn is, and it was the
|
|
2095
|
+
// un-costed half of the 3x story: it reported tokens with no charge beside
|
|
2096
|
+
// them, so the extra calls were the least visible thing on the screen.
|
|
2097
|
+
const flow = turnAccounting(data && data.usage, spec.model, {
|
|
2098
|
+
costUsd: data && typeof data.costUsd === 'number' ? data.costUsd : undefined,
|
|
2099
|
+
});
|
|
2100
|
+
if (flow.tokens != null) bits.push(`${flow.tokens} tokens`);
|
|
2101
|
+
if (flow.cost != null) bits.push(fmtCost(flow.cost, flow.real));
|
|
2102
|
+
// …and roll it into the session, which is the half that was missing. The
|
|
2103
|
+
// card shows this ONE dispatch; a session total that skipped it would be
|
|
2104
|
+
// the lane's calls — the extra ones this feature's cost story is made of —
|
|
2105
|
+
// being the only calls on screen that never get counted. `turns: 0`
|
|
2106
|
+
// because a discovery path is not a turn the user asked for: it
|
|
2107
|
+
// contributes tokens and calls, and leaves the turn count to real
|
|
2108
|
+
// exchanges.
|
|
2109
|
+
const roll = foldRoll(spec.parentSessionId, data && data.usage, {
|
|
2110
|
+
model: spec.model,
|
|
2111
|
+
costUsd: data && typeof data.costUsd === 'number' ? data.costUsd : undefined,
|
|
2112
|
+
calls: data && data.calls,
|
|
2113
|
+
turns: 0,
|
|
2114
|
+
// The dispatch's own prompt and stream, for the same reason as the turn
|
|
2115
|
+
// site: a path that reports no usage is estimated from its own text and
|
|
2116
|
+
// counted, instead of leaving the lane's calls out of the session total.
|
|
2117
|
+
prompt: `Original request:\n${spec.prompt}\n\n${spec.path.hint}`,
|
|
2118
|
+
reply: text,
|
|
2119
|
+
});
|
|
2120
|
+
const rollLine = fmtRoll(roll);
|
|
2121
|
+
if (rollLine) bits.push(`session: ${rollLine}`);
|
|
2007
2122
|
meta.textContent = bits.join(' · ');
|
|
2008
2123
|
} catch (err) {
|
|
2009
2124
|
const message = err && err.message ? err.message : String(err);
|
|
@@ -2854,6 +2969,15 @@ function openSession(id) {
|
|
|
2854
2969
|
// its on-screen transcript — continuing it as sessionId reuses the same
|
|
2855
2970
|
// id and threadMessages carries the prior turns into the next send().
|
|
2856
2971
|
currentSessionId = s.id;
|
|
2972
|
+
// A resumed thread resumes its spend too. The shared store's ledger rows
|
|
2973
|
+
// carry `tokens`/`costUsd` (client/session-store.js recordExchange writes
|
|
2974
|
+
// exactly the shape rollMessages reads), so the rolling total is rebuilt
|
|
2975
|
+
// from what was really recorded rather than restarting at zero — the
|
|
2976
|
+
// CLI's aggregateSessionUsage, which sums history.jsonl for the same
|
|
2977
|
+
// reason. A row this window wrote before ledgerFields existed carries no
|
|
2978
|
+
// `tokens` and folds as unaccounted, which is stated rather than guessed.
|
|
2979
|
+
rollsBySession.set(s.id, rollMessages(msgs));
|
|
2980
|
+
renderRollMeter(s.id);
|
|
2857
2981
|
threadMessages = msgs
|
|
2858
2982
|
.filter((m) => m.role === 'user' || m.role === 'assistant')
|
|
2859
2983
|
.map((m) => ({ role: m.role, content: m.content || m.text || '' }));
|
|
@@ -2908,6 +3032,12 @@ function newChat() {
|
|
|
2908
3032
|
currentSessionId = null;
|
|
2909
3033
|
threadMessages = [];
|
|
2910
3034
|
flowCount = 0;
|
|
3035
|
+
// A fresh thread opens with an empty meter. The outgoing session's roll is
|
|
3036
|
+
// left in the map (reopening it rebuilds from the store anyway), but the
|
|
3037
|
+
// topbar must not keep showing the thread the user just left — `sessionId`
|
|
3038
|
+
// nulls out here and the new id is minted on the first send, so nothing is
|
|
3039
|
+
// hidden that will not reappear with this thread's own number.
|
|
3040
|
+
renderRollMeter(null);
|
|
2911
3041
|
}
|
|
2912
3042
|
|
|
2913
3043
|
/**
|
|
@@ -3080,12 +3210,58 @@ async function send() {
|
|
|
3080
3210
|
// class it is what sized the call, and a user cannot tell a 16k turn from a
|
|
3081
3211
|
// 64k one by looking at the answer.
|
|
3082
3212
|
else if (effort) bits.push(`effort: ${effort}`);
|
|
3083
|
-
|
|
3084
|
-
|
|
3213
|
+
// Tokens AND what they cost, on the CLI's rule: a pooled turn's settled
|
|
3214
|
+
// charge (`costUsd`) is reported verbatim, and only a turn without one is
|
|
3215
|
+
// priced from the local rate table and marked an estimate. Printing tokens
|
|
3216
|
+
// alone left this surface with no comparable meter at all — the desktop
|
|
3217
|
+
// and the CLI run the same engine, so a gap between them had to be
|
|
3218
|
+
// measured on the same prompt, and one side was not measuring.
|
|
3219
|
+
const turn = turnAccounting(data && data.usage, model, {
|
|
3220
|
+
costUsd: data && typeof data.costUsd === 'number' ? data.costUsd : undefined,
|
|
3221
|
+
});
|
|
3222
|
+
if (turn.tokens != null) bits.push(`tokens: ${turn.tokens}`);
|
|
3223
|
+
// A turn the wire did not report on still gets a figure — the same text
|
|
3224
|
+
// estimate that goes into the session total, marked `~` so an inferred
|
|
3225
|
+
// count is never read as a reported one. Printing nothing here while the
|
|
3226
|
+
// session total moved was the other half of "the counter looks dead".
|
|
3227
|
+
else {
|
|
3228
|
+
const est = estimatedBuckets(prompt, text);
|
|
3229
|
+
if (est) bits.push(`~${est.input + est.output} tokens`);
|
|
3230
|
+
}
|
|
3231
|
+
if (turn.cost != null) bits.push(fmtCost(turn.cost, turn.real));
|
|
3232
|
+
// …AND the running session total beside it, which is the number the CLI
|
|
3233
|
+
// prints. The per-turn figure answers "what did that call cost"; only the
|
|
3234
|
+
// rolling one answers "what has this conversation cost", and it was the
|
|
3235
|
+
// missing half. Folded here rather than only on the meter so a turn that
|
|
3236
|
+
// reported no usage still counts as a turn (rollTurn counts before it
|
|
3237
|
+
// tests the count) instead of vanishing from the session.
|
|
3238
|
+
const roll = foldRoll(sessionId, data && data.usage, {
|
|
3239
|
+
model,
|
|
3240
|
+
costUsd: data && typeof data.costUsd === 'number' ? data.costUsd : undefined,
|
|
3241
|
+
calls: data && data.calls,
|
|
3242
|
+
// The turn's own text, used ONLY when the wire reported no usage — so
|
|
3243
|
+
// such a turn is estimated and counted rather than dropped. This is the
|
|
3244
|
+
// CLI's `appendHistory` rule and the reason the total moves every turn.
|
|
3245
|
+
prompt,
|
|
3246
|
+
reply: text,
|
|
3247
|
+
});
|
|
3248
|
+
const rollLine = fmtRoll(roll);
|
|
3249
|
+
if (rollLine) bits.push(`session: ${rollLine}`);
|
|
3085
3250
|
addMessage('assistant', text, bits.join(' · ') || undefined, sessionId, toolLog);
|
|
3086
3251
|
|
|
3087
3252
|
try {
|
|
3088
|
-
await sync.append(sessionId, {
|
|
3253
|
+
await sync.append(sessionId, {
|
|
3254
|
+
role: 'assistant',
|
|
3255
|
+
content: text,
|
|
3256
|
+
// The turn's ledger fields, so this window's spend survives the window
|
|
3257
|
+
// — see ledgerFields. This is what makes the rolling meter the same
|
|
3258
|
+
// quantity after a reopen as it was before one.
|
|
3259
|
+
...ledgerFields(data && data.usage, model, turn, {
|
|
3260
|
+
prompt,
|
|
3261
|
+
reply: text,
|
|
3262
|
+
calls: data && data.calls,
|
|
3263
|
+
}),
|
|
3264
|
+
});
|
|
3089
3265
|
await sync.save({ id: sessionId, title: prompt.slice(0, 60) });
|
|
3090
3266
|
} catch {
|
|
3091
3267
|
/* persistence is non-fatal */
|
|
@@ -3108,9 +3284,31 @@ async function send() {
|
|
|
3108
3284
|
if (isCancellation(err, { userStopped })) {
|
|
3109
3285
|
const text = streamedText || reasoningText || '(stopped before any output)';
|
|
3110
3286
|
threadMessages.push({ role: 'assistant', content: text });
|
|
3111
|
-
|
|
3287
|
+
// A stopped turn is a real exchange and the CLI records one: its
|
|
3288
|
+
// `appendHistory` writes a `status: 'stopped'` entry for every stopped
|
|
3289
|
+
// turn, and `aggregateSessionUsage` sums it like any other. The desktop
|
|
3290
|
+
// wrote `{role, content}` and folded nothing, so an Escape mid-answer
|
|
3291
|
+
// left the session total standing still on a turn the provider had
|
|
3292
|
+
// already billed. Folded here like any other turn; with no wire usage
|
|
3293
|
+
// on this path the figure is the text estimate, marked `est` — and
|
|
3294
|
+
// never a fabricated zero.
|
|
3295
|
+
const turn = turnAccounting(undefined, model, {});
|
|
3296
|
+
const roll = foldRoll(sessionId, undefined, { model, prompt, reply: text });
|
|
3297
|
+
const stopBits = ['stopped by you'];
|
|
3298
|
+
if (turn.tokens != null) stopBits.push(`tokens: ${turn.tokens}`);
|
|
3299
|
+
else {
|
|
3300
|
+
const est = estimatedBuckets(prompt, text);
|
|
3301
|
+
if (est) stopBits.push(`~${est.input + est.output} tokens`);
|
|
3302
|
+
}
|
|
3303
|
+
const stopRollLine = fmtRoll(roll);
|
|
3304
|
+
if (stopRollLine) stopBits.push(`session: ${stopRollLine}`);
|
|
3305
|
+
addMessage('assistant', text, stopBits.join(' · '), sessionId, toolLog);
|
|
3112
3306
|
try {
|
|
3113
|
-
await sync.append(sessionId, {
|
|
3307
|
+
await sync.append(sessionId, {
|
|
3308
|
+
role: 'assistant',
|
|
3309
|
+
content: text,
|
|
3310
|
+
...ledgerFields(undefined, model, turn, { prompt, reply: text }),
|
|
3311
|
+
});
|
|
3114
3312
|
} catch {
|
|
3115
3313
|
/* persistence is non-fatal */
|
|
3116
3314
|
}
|
package/renderer/index.html
CHANGED
|
@@ -22,6 +22,11 @@
|
|
|
22
22
|
<span class="dot" id="conn-dot"></span>
|
|
23
23
|
<span id="conn-text">connecting…</span>
|
|
24
24
|
</div>
|
|
25
|
+
<!-- The rolling session token total, counted the way the CLI counts it:
|
|
26
|
+
every finished turn folded into one running number rather than a
|
|
27
|
+
per-turn count that resets. Hidden until a turn has been accounted
|
|
28
|
+
for, so an untouched thread never shows a `0 tok` it never measured. -->
|
|
29
|
+
<div class="session-meter" id="session-meter" hidden></div>
|
|
25
30
|
<div class="spacer"></div>
|
|
26
31
|
<button type="button" id="new-chat" class="ghost-btn">New chat</button>
|
|
27
32
|
</header>
|
package/renderer/style.css
CHANGED
|
@@ -114,6 +114,28 @@ body {
|
|
|
114
114
|
box-shadow: 0 0 6px var(--teal);
|
|
115
115
|
}
|
|
116
116
|
|
|
117
|
+
/* The rolling session token total — the CLI's status-bar figure, in the
|
|
118
|
+
topbar. Muted and monospaced so it reads as a meter rather than a heading,
|
|
119
|
+
and it never wraps: a number that reflows the topbar every turn is worse
|
|
120
|
+
than no number at all. Hidden via the `hidden` attribute, so the rule below
|
|
121
|
+
has to win against the flex display the base class would otherwise get. */
|
|
122
|
+
.session-meter {
|
|
123
|
+
font-size: 12px;
|
|
124
|
+
font-family: var(--font-mono);
|
|
125
|
+
color: var(--text2);
|
|
126
|
+
white-space: nowrap;
|
|
127
|
+
overflow: hidden;
|
|
128
|
+
text-overflow: ellipsis;
|
|
129
|
+
max-width: 40ch;
|
|
130
|
+
padding: 2px 8px;
|
|
131
|
+
border: 1px solid var(--border);
|
|
132
|
+
border-radius: 999px;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
.session-meter[hidden] {
|
|
136
|
+
display: none;
|
|
137
|
+
}
|
|
138
|
+
|
|
117
139
|
.spacer {
|
|
118
140
|
flex: 1;
|
|
119
141
|
}
|
package/renderer/usage.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
'use strict';
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
|
-
* Pure token-usage → display-number mapping.
|
|
4
|
+
* Pure token-usage → display-number mapping, and the cost half of it.
|
|
5
5
|
*
|
|
6
6
|
* Standalone from app.js (same reason as budget.js/stream-policy.js): it is
|
|
7
7
|
* requireable from a plain Node test without window.aegis. app.js only calls
|
|
@@ -18,6 +18,27 @@
|
|
|
18
18
|
* the call was silently being billed. Accepting both spellings, and deriving
|
|
19
19
|
* the total when the provider doesn't state one, is what makes the spend
|
|
20
20
|
* visible regardless of which endpoint answered.
|
|
21
|
+
*
|
|
22
|
+
* ── The cost half, on the CLI's principle ──────────────────────────────────
|
|
23
|
+
*
|
|
24
|
+
* The renderer printed a token count and nothing else, so the desktop had no
|
|
25
|
+
* meter to compare against `aegiscodex` on the same engine and the same
|
|
26
|
+
* prompt — the comparison that started this whole line of work. The CLI's
|
|
27
|
+
* rule (cli/src/tokens.js, ported here) has three parts, and all three matter:
|
|
28
|
+
*
|
|
29
|
+
* 1. A pooled turn's bill is settled SERVER-side and carries a margin and a
|
|
30
|
+
* prompt-cache discount the client cannot see. When the response states
|
|
31
|
+
* `costUsd`, that figure wins verbatim — it is the truth about the
|
|
32
|
+
* charge, not an estimate of it.
|
|
33
|
+
* 2. Only in the absence of a settled charge does the local rate table
|
|
34
|
+
* apply, and the result is labeled an estimate so nobody reads a guess as
|
|
35
|
+
* a bill.
|
|
36
|
+
* 3. The rate table is resolved by exact id, then longest prefix. The
|
|
37
|
+
* DeepSeek row is load-bearing: without it every direct DeepSeek turn
|
|
38
|
+
* fell through to Sonnet's $3.00/$15.00 per M against a real
|
|
39
|
+
* $0.14/$0.28 — 21x the input rate and 54x the output rate — which made
|
|
40
|
+
* the *meter* the largest single contributor to the apparent cost gap
|
|
41
|
+
* between two surfaces running identical code.
|
|
21
42
|
*/
|
|
22
43
|
|
|
23
44
|
/**
|
|
@@ -25,18 +46,508 @@
|
|
|
25
46
|
* reported none (an unknown count must render as nothing, never as `0`).
|
|
26
47
|
*
|
|
27
48
|
* @param {{total_tokens?: number, prompt_tokens?: number, completion_tokens?: number,
|
|
28
|
-
* input_tokens?: number, output_tokens?: number
|
|
49
|
+
* input_tokens?: number, output_tokens?: number,
|
|
50
|
+
* input?: number, output?: number}|null|undefined} usage
|
|
29
51
|
* @returns {number|null}
|
|
30
52
|
*/
|
|
31
53
|
function usageTokens(usage) {
|
|
32
54
|
if (!usage || typeof usage !== 'object') return null;
|
|
33
55
|
if (typeof usage.total_tokens === 'number') return usage.total_tokens;
|
|
34
|
-
|
|
35
|
-
|
|
56
|
+
// Three spellings of the same quantity: OpenAI's, Anthropic's, and this
|
|
57
|
+
// file's own bucket shape — which is also the shape a LEDGER row carries
|
|
58
|
+
// (cli/src/history.js writes `{input, output, cacheRead, cacheWrite}` into
|
|
59
|
+
// sessions.json, and the CLI's demo path writes `{input, output}`). Reading
|
|
60
|
+
// only the two wire spellings left every stored row uncountable, so a
|
|
61
|
+
// resumed session's rolling total started at zero even though its ledger
|
|
62
|
+
// said otherwise.
|
|
63
|
+
const input = usage.input_tokens ?? usage.prompt_tokens ?? usage.input;
|
|
64
|
+
const output = usage.output_tokens ?? usage.completion_tokens ?? usage.output;
|
|
36
65
|
if (typeof input !== 'number' && typeof output !== 'number') return null;
|
|
37
66
|
return (input || 0) + (output || 0);
|
|
38
67
|
}
|
|
39
68
|
|
|
69
|
+
/**
|
|
70
|
+
* Rough token count for text the wire never measured — a direct port of the
|
|
71
|
+
* CLI's own estimator (`aegiscodex-dev/src/tokens.js estimateTokens`), and the
|
|
72
|
+
* reason its session total is MONOTONIC.
|
|
73
|
+
*
|
|
74
|
+
* This is the whole difference being fixed. The CLI's `appendHistory`
|
|
75
|
+
* (aegiscodex-dev/src/history.js) writes a `tokens` object for EVERY finished
|
|
76
|
+
* exchange: `{input, output, cacheRead, cacheWrite, real: true}` when the wire
|
|
77
|
+
* reported usage, and `{input: estimateTokens(prompt), output:
|
|
78
|
+
* estimateTokens(reply), real: false}` when it did not. `real` is a FLAG, not a
|
|
79
|
+
* gate — `aggregateSessionUsage` adds `t.input || 0` for every row regardless,
|
|
80
|
+
* so a turn without usage still moves the total and only marks it as partly
|
|
81
|
+
* estimated.
|
|
82
|
+
*
|
|
83
|
+
* The desktop rolled only reported usage and dropped everything else, so its
|
|
84
|
+
* total was flat across any turn the pool did not report on — the meter looked
|
|
85
|
+
* dead while the conversation was being billed. ~4 characters per token, the
|
|
86
|
+
* CLI's heuristic, kept identical so the two surfaces estimate the same turn
|
|
87
|
+
* the same way. Empty text is 0, never the `Math.max(1, …)` floor.
|
|
88
|
+
*/
|
|
89
|
+
function estimateTokens(text) {
|
|
90
|
+
if (!text) return 0;
|
|
91
|
+
return Math.max(1, Math.ceil([...String(text)].length / 4));
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* The bucket shape an exchange gets when the wire reported no usage: the CLI's
|
|
96
|
+
* `{input: estimateTokens(prompt), output: estimateTokens(reply)}`.
|
|
97
|
+
*
|
|
98
|
+
* `null` when there is no text at all to estimate from, so a dispatch that
|
|
99
|
+
* genuinely reported nothing (no usage, no prompt, no reply) still lands in
|
|
100
|
+
* `unknown` instead of being handed a fabricated zero.
|
|
101
|
+
*
|
|
102
|
+
* @param {string} [prompt]
|
|
103
|
+
* @param {string} [reply]
|
|
104
|
+
* @returns {{input: number, output: number, cacheRead: number, cacheWrite: number}|null}
|
|
105
|
+
*/
|
|
106
|
+
function estimatedBuckets(prompt, reply) {
|
|
107
|
+
const hasPrompt = typeof prompt === 'string' && prompt.length > 0;
|
|
108
|
+
const hasReply = typeof reply === 'string' && reply.length > 0;
|
|
109
|
+
if (!hasPrompt && !hasReply) return null;
|
|
110
|
+
return {
|
|
111
|
+
input: estimateTokens(prompt),
|
|
112
|
+
output: estimateTokens(reply),
|
|
113
|
+
cacheRead: 0,
|
|
114
|
+
cacheWrite: 0,
|
|
115
|
+
real: false,
|
|
116
|
+
};
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
// Per-million-token USD rates. Cache-read/write matter for long sessions.
|
|
120
|
+
//
|
|
121
|
+
// These are the PROVIDER's rates, for the fallback path where a turn has no
|
|
122
|
+
// server-settled charge (a direct provider, ollama, a custom endpoint). A
|
|
123
|
+
// pooled turn reports the ledger figure instead — see turnAccounting's
|
|
124
|
+
// `costUsd` — because the pool's bill carries a margin and a prompt-cache
|
|
125
|
+
// discount this table cannot see.
|
|
126
|
+
//
|
|
127
|
+
// Keys are matched by exact id first, then by prefix (ratesFor below), so a
|
|
128
|
+
// family row covers every dated variant of it.
|
|
129
|
+
const RATES = {
|
|
130
|
+
sonnet: { input: 3.00, output: 15.00, cacheRead: 0.30, cacheWrite: 3.75 },
|
|
131
|
+
default: { input: 3.00, output: 15.00, cacheRead: 0.30, cacheWrite: 3.75 },
|
|
132
|
+
fable: { input: 5.00, output: 25.00, cacheRead: 0.50, cacheWrite: 6.25 },
|
|
133
|
+
opus: { input: 5.00, output: 25.00, cacheRead: 0.50, cacheWrite: 6.25 },
|
|
134
|
+
// DeepSeek V4 — the provider's published rate, independently corroborated by
|
|
135
|
+
// aegis1 services/nexus_provider/catalog.py (cost_per_1k_input=0.00014,
|
|
136
|
+
// cost_per_1k_output=0.00028) and referenced in services/pricing.py.
|
|
137
|
+
//
|
|
138
|
+
// This row was MISSING upstream, and usageCost fell through to RATES.sonnet
|
|
139
|
+
// for every DeepSeek turn: $3.00/$15.00 per M against a real $0.14/$0.28 —
|
|
140
|
+
// 21x the input rate and 54x the output rate, on every direct DeepSeek call.
|
|
141
|
+
//
|
|
142
|
+
// cacheRead is 0.1x input (aegis1 services/pricing.py
|
|
143
|
+
// CACHE_READ_FRACTION_BY_COMPANY["deepseek"] = 0.1 -> $0.014/M).
|
|
144
|
+
// cacheWrite is the input rate: DeepSeek bills a cache write as ordinary
|
|
145
|
+
// input tokens and charges no separate write premium, unlike Anthropic's
|
|
146
|
+
// 1.25x. A zero here would understate a session that writes cache.
|
|
147
|
+
deepseek: { input: 0.14, output: 0.28, cacheRead: 0.014, cacheWrite: 0.14 },
|
|
148
|
+
};
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* The rate row for a model id, provider, or alias. Exact id wins, then the
|
|
152
|
+
* longest matching prefix, then the Sonnet-class default.
|
|
153
|
+
*/
|
|
154
|
+
function ratesFor(model) {
|
|
155
|
+
const id = String(model || '').toLowerCase();
|
|
156
|
+
if (!id) return RATES.default;
|
|
157
|
+
if (RATES[id]) return RATES[id];
|
|
158
|
+
let best = null;
|
|
159
|
+
let bestLen = 0;
|
|
160
|
+
for (const [prefix, rates] of Object.entries(RATES)) {
|
|
161
|
+
if (id.startsWith(prefix) && prefix.length > bestLen) {
|
|
162
|
+
best = rates;
|
|
163
|
+
bestLen = prefix.length;
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
return best || RATES.default;
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* The four billable buckets from a wire usage object, accepting both provider
|
|
171
|
+
* spellings. Cache fields have no Anthropic-compatible short form here because
|
|
172
|
+
* the desktop's transport normalises them (desktop/lib/local/providers.js).
|
|
173
|
+
*
|
|
174
|
+
* @param {object|null|undefined} usage
|
|
175
|
+
* @returns {{input: number, output: number, cacheRead: number, cacheWrite: number}}
|
|
176
|
+
*/
|
|
177
|
+
function usageBuckets(usage) {
|
|
178
|
+
const u = usage && typeof usage === 'object' ? usage : {};
|
|
179
|
+
const num = (...candidates) => {
|
|
180
|
+
for (const c of candidates) if (typeof c === 'number') return c;
|
|
181
|
+
return 0;
|
|
182
|
+
};
|
|
183
|
+
const input = num(u.input_tokens, u.prompt_tokens, u.input);
|
|
184
|
+
const output = num(u.output_tokens, u.completion_tokens, u.output);
|
|
185
|
+
// A stated total larger than the split means the provider counted tokens the
|
|
186
|
+
// split does not name (thinking, cached reads). Attribute the remainder to
|
|
187
|
+
// input rather than dropping it: dropping it would understate the bill.
|
|
188
|
+
const total = num(u.total_tokens);
|
|
189
|
+
const cacheRead = num(u.cache_read_input_tokens, u.cacheRead);
|
|
190
|
+
const cacheWrite = num(u.cache_creation_input_tokens, u.cacheWrite);
|
|
191
|
+
const accounted = input + output + cacheRead + cacheWrite;
|
|
192
|
+
return {
|
|
193
|
+
input: total > accounted ? input + (total - accounted) : input,
|
|
194
|
+
output,
|
|
195
|
+
cacheRead,
|
|
196
|
+
cacheWrite,
|
|
197
|
+
};
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/** Dollar cost of a usage record at the given model's rates (USD, estimate). */
|
|
201
|
+
function usageCost(usage, model = 'sonnet') {
|
|
202
|
+
const r = ratesFor(model);
|
|
203
|
+
const u = usageBuckets(usage);
|
|
204
|
+
const toD = (n, rate) => (n / 1_000_000) * rate;
|
|
205
|
+
return toD(u.input, r.input)
|
|
206
|
+
+ toD(u.output, r.output)
|
|
207
|
+
+ toD(u.cacheRead, r.cacheRead)
|
|
208
|
+
+ toD(u.cacheWrite, r.cacheWrite);
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
/**
|
|
212
|
+
* What one turn cost, as the CLI reports it: the server's settled charge when
|
|
213
|
+
* the response carries one, otherwise the rate table's estimate, always with
|
|
214
|
+
* the distinction preserved so it can be labeled.
|
|
215
|
+
*
|
|
216
|
+
* @param {object|null|undefined} usage the response's `usage` object
|
|
217
|
+
* @param {string} model the model id that answered
|
|
218
|
+
* @param {{costUsd?: number}} [opts] the server-settled charge, if any
|
|
219
|
+
* @returns {{tokens: number|null, cost: number|null, real: boolean, estimated: boolean}}
|
|
220
|
+
* `real` is true only when `cost` is the settled charge. `cost` is
|
|
221
|
+
* `null` when nothing was reported and no model was named — an
|
|
222
|
+
* unpriced turn must render as nothing, never as $0.0000.
|
|
223
|
+
*/
|
|
224
|
+
function turnAccounting(usage, model, opts = {}) {
|
|
225
|
+
const tokens = usageTokens(usage);
|
|
226
|
+
const u = usage && typeof usage === 'object' ? usage : {};
|
|
227
|
+
// The settled charge can arrive either on the response or folded into the
|
|
228
|
+
// usage object — aegiscodex-dev/src/main.js does the latter
|
|
229
|
+
// (`{ ...result.usage, costUsd: result.costUsd }`), so both are accepted.
|
|
230
|
+
const settled = typeof opts.costUsd === 'number'
|
|
231
|
+
? opts.costUsd
|
|
232
|
+
: (typeof u.costUsd === 'number' ? u.costUsd : undefined);
|
|
233
|
+
if (typeof settled === 'number') {
|
|
234
|
+
return { tokens, cost: settled, real: true, estimated: false };
|
|
235
|
+
}
|
|
236
|
+
if (tokens == null) return { tokens, cost: null, real: false, estimated: false };
|
|
237
|
+
const priced = usageCost(u, model);
|
|
238
|
+
// A model the table cannot place is still priced at the Sonnet-class default
|
|
239
|
+
// (ratesFor never returns nothing), so this is always an estimate.
|
|
240
|
+
return { tokens, cost: priced, real: false, estimated: true };
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
/**
|
|
244
|
+
* ── The rolling session tallies, on the CLI's rule ─────────────────────────
|
|
245
|
+
*
|
|
246
|
+
* The two surfaces did not merely print different numbers — they counted
|
|
247
|
+
* differently. The CLI never shows a turn's tokens in isolation: `recordTurn`
|
|
248
|
+
* (cli/src/app.js) folds each finished turn's usage into one `session` object
|
|
249
|
+
* — `tokens`, `inputTokens`, `outputTokens`, `calls` — and what the user reads
|
|
250
|
+
* back is that RUNNING TOTAL. The status bar prints `state.tokens`
|
|
251
|
+
* (renderStatus), and `ctrl+t` prints the tallies in one line
|
|
252
|
+
* (`tokenSummary`: `12,400 tok (10,100 in / 2,300 out) · 4 calls · €0.03`).
|
|
253
|
+
*
|
|
254
|
+
* The desktop counted per turn only. Every meta row was a fresh count that
|
|
255
|
+
* reset at the next call, so "what has this session spent" was answerable only
|
|
256
|
+
* by adding the rows up by eye across a scrollback — which is most of why the
|
|
257
|
+
* desktop looked like it accounted differently from the CLI on identical
|
|
258
|
+
* engine code and an identical prompt.
|
|
259
|
+
*
|
|
260
|
+
* Three properties of the CLI's fold are load-bearing and are reproduced here
|
|
261
|
+
* exactly, because dropping any one of them reintroduces a specific lie:
|
|
262
|
+
*
|
|
263
|
+
* 1. `turns` and `calls` are incremented BEFORE the "did usage come back?"
|
|
264
|
+
* gate (recordTurn counts first, then tests `tokens != null`). A turn
|
|
265
|
+
* that reported nothing still happened; a tally that skipped it would
|
|
266
|
+
* report the session as shorter and cheaper than it was.
|
|
267
|
+
* 2. `tokens`, `input` and `output` accumulate — never reset. A rolling
|
|
268
|
+
* total that resets per turn is the per-turn count it replaced.
|
|
269
|
+
* 3. A `null` count contributes nothing and is counted in `unknown`, so
|
|
270
|
+
* `tokens: 0` is only ever read as a real zero. Nothing is ever added as
|
|
271
|
+
* a fabricated 0 to make the arithmetic look complete.
|
|
272
|
+
*
|
|
273
|
+
* The money split is the CLI's too: a settled charge (the pool's ledger
|
|
274
|
+
* figure) rolls into `cost`, and a locally priced turn rolls into `estimate`.
|
|
275
|
+
* They are kept apart rather than summed so a `~`-estimate can never be read
|
|
276
|
+
* as part of the bill — the distinction `fmtCost` marks on a single turn, held
|
|
277
|
+
* across the session.
|
|
278
|
+
*/
|
|
279
|
+
|
|
280
|
+
/** A session with nothing accounted for yet. */
|
|
281
|
+
function emptyRoll() {
|
|
282
|
+
return {
|
|
283
|
+
turns: 0,
|
|
284
|
+
calls: 0,
|
|
285
|
+
tokens: 0,
|
|
286
|
+
input: 0,
|
|
287
|
+
output: 0,
|
|
288
|
+
cacheRead: 0,
|
|
289
|
+
cacheWrite: 0,
|
|
290
|
+
/** Dispatches that reported no usage at all — the honest gap in `tokens`. */
|
|
291
|
+
unknown: 0,
|
|
292
|
+
/**
|
|
293
|
+
* Exchanges folded from TEXT rather than reported usage, on the CLI's
|
|
294
|
+
* `real: false` rule. They are counted in `tokens` — that is the point —
|
|
295
|
+
* and named here so an estimated figure is never read as a measured one.
|
|
296
|
+
*/
|
|
297
|
+
estimated: 0,
|
|
298
|
+
/** Settled charges (server-settled `costUsd`), rolled. */
|
|
299
|
+
cost: 0,
|
|
300
|
+
/** Locally priced turns, rolled. Never mixed into `cost`. */
|
|
301
|
+
estimate: 0,
|
|
302
|
+
};
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
/**
|
|
306
|
+
* Fold one completed dispatch into a session's rolling tallies. Pure: it
|
|
307
|
+
* returns a NEW roll and never mutates the one it was handed, so a half-applied
|
|
308
|
+
* fold cannot exist.
|
|
309
|
+
*
|
|
310
|
+
* @param {object} [roll] the roll so far (emptyRoll() when omitted)
|
|
311
|
+
* @param {object} [usage] the response's `usage` object
|
|
312
|
+
* @param {{model?: string, costUsd?: number, calls?: number, turns?: number,
|
|
313
|
+
* prompt?: string, reply?: string, estimated?: boolean}} [opts]
|
|
314
|
+
* `calls` defaults to 1; a pooled turn may report how many provider
|
|
315
|
+
* calls it actually made. `turns: 0` folds a dispatch that is not a
|
|
316
|
+
* turn of its own — the discovery-lane card, which bills like any other
|
|
317
|
+
* call but is not something the user asked for. `prompt`/`reply` are the
|
|
318
|
+
* turn's text, used ONLY when the wire reported no usage, so the turn is
|
|
319
|
+
* estimated rather than dropped (the CLI's `appendHistory` rule).
|
|
320
|
+
* @returns {object} the new roll
|
|
321
|
+
*/
|
|
322
|
+
function rollTurn(roll, usage, opts = {}) {
|
|
323
|
+
const next = Object.assign(emptyRoll(), roll || {});
|
|
324
|
+
next.turns += opts.turns === undefined ? 1 : Number(opts.turns) || 0;
|
|
325
|
+
next.calls += opts.calls === undefined ? 1 : Number(opts.calls) || 0;
|
|
326
|
+
const turn = turnAccounting(usage, opts.model, { costUsd: opts.costUsd });
|
|
327
|
+
// A row folded from text rather than from the wire: either this turn's own
|
|
328
|
+
// fallback below, or a stored ledger row carrying `real: false` — the shape
|
|
329
|
+
// the CLI's `appendHistory` writes for an exchange the provider did not
|
|
330
|
+
// report on.
|
|
331
|
+
const estimated = opts.estimated === true || (usage && usage.real === false);
|
|
332
|
+
if (turn.tokens == null) {
|
|
333
|
+
// No reported usage. The CLI does not let such a turn vanish from the
|
|
334
|
+
// total: `appendHistory` estimates it from the text and marks it
|
|
335
|
+
// `real: false` (aegiscodex-dev/src/history.js), and
|
|
336
|
+
// `aggregateSessionUsage` then adds it like any other row. Folding the same
|
|
337
|
+
// estimate here is what makes the live roll and the rebuilt roll the same
|
|
338
|
+
// number, and what stops the total standing still on exactly the turns the
|
|
339
|
+
// pool declined to report on — the symptom this fallback exists to remove.
|
|
340
|
+
const est = estimatedBuckets(opts.prompt, opts.reply);
|
|
341
|
+
if (!est) {
|
|
342
|
+
// Nothing reported AND no text to estimate from. The single case that
|
|
343
|
+
// stays uncounted: `unknown` names the gap, where a fabricated zero would
|
|
344
|
+
// read as a measurement.
|
|
345
|
+
next.unknown += 1;
|
|
346
|
+
return next;
|
|
347
|
+
}
|
|
348
|
+
next.tokens += est.input + est.output;
|
|
349
|
+
next.input += est.input;
|
|
350
|
+
next.output += est.output;
|
|
351
|
+
next.estimated += 1;
|
|
352
|
+
return next;
|
|
353
|
+
}
|
|
354
|
+
const b = usageBuckets(usage);
|
|
355
|
+
next.tokens += turn.tokens;
|
|
356
|
+
next.input += b.input;
|
|
357
|
+
next.output += b.output;
|
|
358
|
+
next.cacheRead += b.cacheRead;
|
|
359
|
+
next.cacheWrite += b.cacheWrite;
|
|
360
|
+
if (estimated) {
|
|
361
|
+
// Money stays out of an estimated exchange, deliberately and in both
|
|
362
|
+
// directions: the CLI writes no `costUsd` for one, so pricing it here would
|
|
363
|
+
// make the live roll drift from the rebuilt one — and pricing tokens that
|
|
364
|
+
// were themselves guessed would stack one guess on another.
|
|
365
|
+
next.estimated += 1;
|
|
366
|
+
return next;
|
|
367
|
+
}
|
|
368
|
+
if (turn.real) next.cost += turn.cost;
|
|
369
|
+
else if (turn.cost != null) next.estimate += turn.cost;
|
|
370
|
+
return next;
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
/**
|
|
374
|
+
* Rebuild a session's rolling total from stored exchanges — the desktop's
|
|
375
|
+
* counterpart of the CLI's `aggregateSessionUsage`, which sums history.jsonl so
|
|
376
|
+
* a resumed session (and a compacted one) still reports everything it spent.
|
|
377
|
+
*
|
|
378
|
+
* Reads the shapes the shared store writes: an assistant message carrying
|
|
379
|
+
* `tokens: {input, output, cacheRead, cacheWrite}` and, when the pool settled
|
|
380
|
+
* the turn, `costUsd` (cli/src/history.js → session-store.recordExchange).
|
|
381
|
+
* Messages the desktop itself appended carry no `tokens` and fold as `unknown`
|
|
382
|
+
* — a resumed thread states what is known and does not invent the rest.
|
|
383
|
+
*
|
|
384
|
+
* @param {Array<{role?: string, tokens?: object, costUsd?: number, model?: string}>} [messages]
|
|
385
|
+
* @returns {object} the roll
|
|
386
|
+
*/
|
|
387
|
+
function rollMessages(messages) {
|
|
388
|
+
let roll = emptyRoll();
|
|
389
|
+
for (const m of Array.isArray(messages) ? messages : []) {
|
|
390
|
+
if (!m || m.role !== 'assistant') continue;
|
|
391
|
+
roll = rollTurn(roll, m.tokens, {
|
|
392
|
+
model: m.model,
|
|
393
|
+
costUsd: typeof m.costUsd === 'number' ? m.costUsd : undefined,
|
|
394
|
+
calls: m.calls,
|
|
395
|
+
});
|
|
396
|
+
}
|
|
397
|
+
return roll;
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
/**
|
|
401
|
+
* The ledger row one finished dispatch must carry into the shared session
|
|
402
|
+
* store, so the rolling total can be REBUILT when the thread is reopened —
|
|
403
|
+
* the desktop's counterpart of the CLI's `appendHistory`
|
|
404
|
+
* (aegiscodex-dev/src/history.js:36).
|
|
405
|
+
*
|
|
406
|
+
* One authority for the row, on purpose. The CLI writes a `tokens` object for
|
|
407
|
+
* EVERY exchange and skips none:
|
|
408
|
+
*
|
|
409
|
+
* tokens: usage
|
|
410
|
+
* ? { input, output, cacheRead, cacheWrite, real: true }
|
|
411
|
+
* : { input: estimateTokens(prompt), output: estimateTokens(reply), real: false }
|
|
412
|
+
*
|
|
413
|
+
* …and `aggregateSessionUsage` then sums `t.input || 0` over every entry with
|
|
414
|
+
* `real` as a FLAG, not a gate. Mirroring that shape here is what makes the
|
|
415
|
+
* live roll and the rebuilt roll the same number: `rollTurn` is handed this
|
|
416
|
+
* exact object on reopen, so an estimated exchange adds the same buckets it
|
|
417
|
+
* added live and is counted under `estimated` in both.
|
|
418
|
+
*
|
|
419
|
+
* Returns `null` when there is neither reported usage nor text to estimate
|
|
420
|
+
* from — the one case that must stay unrecorded, because a fabricated
|
|
421
|
+
* `{input: 0, output: 0}` row would read as a measured zero forever after.
|
|
422
|
+
*
|
|
423
|
+
* @param {object} [usage] the response's `usage` object
|
|
424
|
+
* @param {object} [turn] its accounting (turnAccounting), when already done
|
|
425
|
+
* @param {{model?: string, costUsd?: number, calls?: number, prompt?: string,
|
|
426
|
+
* reply?: string}} [opts]
|
|
427
|
+
* @returns {object|null} the row's ledger fields
|
|
428
|
+
*/
|
|
429
|
+
function ledgerRow(usage, turn, opts = {}) {
|
|
430
|
+
const t = turn || turnAccounting(usage, opts.model, { costUsd: opts.costUsd });
|
|
431
|
+
const calls = Number.isFinite(opts.calls) && opts.calls > 0 ? opts.calls : undefined;
|
|
432
|
+
if (t.tokens != null) {
|
|
433
|
+
const row = { tokens: usageBuckets(usage) };
|
|
434
|
+
// Only a SETTLED charge is persisted. The CLI writes `costUsd` for a real
|
|
435
|
+
// charge only, and the rebuild prices an unpriced row from the same rate
|
|
436
|
+
// table this window used — persisting a local guess would let a stale
|
|
437
|
+
// table outlive the change that wrote it.
|
|
438
|
+
if (t.real && t.cost != null) row.costUsd = t.cost;
|
|
439
|
+
if (calls !== undefined) row.calls = calls;
|
|
440
|
+
return row;
|
|
441
|
+
}
|
|
442
|
+
const est = estimatedBuckets(opts.prompt, opts.reply);
|
|
443
|
+
if (!est) return null;
|
|
444
|
+
const row = { tokens: est };
|
|
445
|
+
if (calls !== undefined) row.calls = calls;
|
|
446
|
+
return row;
|
|
447
|
+
}
|
|
448
|
+
|
|
449
|
+
/**
|
|
450
|
+
* The CLI's rendering of a session tally — `cli/src/format.js fmtTokens`, which
|
|
451
|
+
* is the one `tokenSummary` actually imports (`cli/src/app.js:50`). NOT the
|
|
452
|
+
* `1.5k`/`12.3k` form in `cli/src/tokens.js`: that one belongs to the /cost
|
|
453
|
+
* panels, and using it here would print `12.4k` on the very total the CLI
|
|
454
|
+
* prints as `12,400` — a rendering difference stacked on top of the accounting
|
|
455
|
+
* difference this change exists to remove.
|
|
456
|
+
*
|
|
457
|
+
* Comma-grouped integer. Anything not finite and positive renders `0`, which is
|
|
458
|
+
* the CLI's rule and keeps a stray NaN from reaching the topbar.
|
|
459
|
+
*/
|
|
460
|
+
function fmtTokens(n) {
|
|
461
|
+
const v = Number(n);
|
|
462
|
+
if (!Number.isFinite(v) || v <= 0) return '0';
|
|
463
|
+
return Math.round(v)
|
|
464
|
+
.toString()
|
|
465
|
+
.replace(/\B(?=(\d{3})+(?!\d))/g, ',');
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
/**
|
|
469
|
+
* The rolling session total as one line — the desktop's counterpart of the
|
|
470
|
+
* CLI's `tokenSummary`. Empty string when nothing has been accounted for, so
|
|
471
|
+
* an untouched session adds no noise to a turn's meta line.
|
|
472
|
+
*
|
|
473
|
+
* @param {object} [roll]
|
|
474
|
+
* @returns {string} e.g. `12,400 tok (10,100 in / 2,300 out) · 4 calls · $0.0310`
|
|
475
|
+
*/
|
|
476
|
+
function fmtRoll(roll) {
|
|
477
|
+
const r = roll || emptyRoll();
|
|
478
|
+
if (!r.turns && !r.calls) return '';
|
|
479
|
+
// The first two fields are `tokenSummary` (cli/src/app.js:1413) verbatim,
|
|
480
|
+
// separator and all: `12,400 tok (10,100 in / 2,300 out) · 4 calls`. There the
|
|
481
|
+
// parenthetical is unconditional and the call count is singular at one; both
|
|
482
|
+
// are kept, because a line that is only *sometimes* shaped like the CLI's is a
|
|
483
|
+
// lookalike rather than the same quantity. The call count is also the number
|
|
484
|
+
// that reveals a fan-out, which is why it is not hidden at 1.
|
|
485
|
+
const bits = [];
|
|
486
|
+
// The token half only when something was actually counted. `rollTurn` counts
|
|
487
|
+
// turns and calls BEFORE it looks at the token count, so a dispatch that
|
|
488
|
+
// reported nothing still reaches here — and printing its empty tally would
|
|
489
|
+
// put `0 tok (0 in / 0 out)` on the topbar, a figure the meter never took,
|
|
490
|
+
// which reads as a counter that does not move. The turn count is still
|
|
491
|
+
// stated, because that much is true.
|
|
492
|
+
if (r.tokens > 0) {
|
|
493
|
+
bits.push(
|
|
494
|
+
`${fmtTokens(r.tokens)} tok (${fmtTokens(r.input)} in / ${fmtTokens(r.output)} out)`
|
|
495
|
+
);
|
|
496
|
+
}
|
|
497
|
+
bits.push(`${r.calls} call${r.calls === 1 ? '' : 's'}`);
|
|
498
|
+
// Money is where this line departs from `tokenSummary`, deliberately: that
|
|
499
|
+
// one sums a single `session.cost` in EUR via fmtEur, while this surface keeps
|
|
500
|
+
// a settled charge and a local estimate apart so a `~`-estimate can never be
|
|
501
|
+
// read as part of the bill. Desktop's pre-existing fmtCost renders both, and
|
|
502
|
+
// its `$` convention is left exactly as it was.
|
|
503
|
+
if (r.cost > 0 || r.estimate > 0) {
|
|
504
|
+
const money = [];
|
|
505
|
+
if (r.cost > 0) money.push(fmtCost(r.cost, true));
|
|
506
|
+
if (r.estimate > 0) money.push(fmtCost(r.estimate, false));
|
|
507
|
+
bits.push(money.join(' + '));
|
|
508
|
+
}
|
|
509
|
+
// No counterpart in `tokenSummary`, which folds a usage-less turn silently and
|
|
510
|
+
// so reports a total short of the truth without saying so. Named here instead:
|
|
511
|
+
// the count appears only when something went unreported, and it never changes
|
|
512
|
+
// a number — it only says the number is not the whole story.
|
|
513
|
+
if (r.unknown) bits.push(`${r.unknown} unrpt`);
|
|
514
|
+
// The other half of the same honesty: a total that includes estimated
|
|
515
|
+
// exchanges says so, because the CLI marks the same rows `real: false` and a
|
|
516
|
+
// reader is entitled to know which figure they are looking at. It changes no
|
|
517
|
+
// number — it says the number is partly inferred.
|
|
518
|
+
if (r.estimated) bits.push(`${r.estimated} est`);
|
|
519
|
+
return bits.join(' · ');
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
/**
|
|
523
|
+
* A cost for display. Estimates are marked with `~` so an estimate is never
|
|
524
|
+
* mistaken for a settled charge.
|
|
525
|
+
*
|
|
526
|
+
* @param {number} cost
|
|
527
|
+
* @param {boolean} [real]
|
|
528
|
+
* @returns {string}
|
|
529
|
+
*/
|
|
530
|
+
function fmtCost(cost, real) {
|
|
531
|
+
if (typeof cost !== 'number' || !isFinite(cost)) return '';
|
|
532
|
+
return `${real ? '' : '~'}$${cost.toFixed(4)}`;
|
|
533
|
+
}
|
|
534
|
+
|
|
40
535
|
if (typeof module !== 'undefined' && module.exports) {
|
|
41
|
-
module.exports = {
|
|
536
|
+
module.exports = {
|
|
537
|
+
RATES,
|
|
538
|
+
usageTokens,
|
|
539
|
+
ratesFor,
|
|
540
|
+
usageBuckets,
|
|
541
|
+
usageCost,
|
|
542
|
+
turnAccounting,
|
|
543
|
+
estimateTokens,
|
|
544
|
+
estimatedBuckets,
|
|
545
|
+
emptyRoll,
|
|
546
|
+
rollTurn,
|
|
547
|
+
rollMessages,
|
|
548
|
+
ledgerRow,
|
|
549
|
+
fmtTokens,
|
|
550
|
+
fmtRoll,
|
|
551
|
+
fmtCost,
|
|
552
|
+
};
|
|
42
553
|
}
|