mixdog 1.0.1 → 1.0.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/README.md +8 -0
  2. package/package.json +2 -2
  3. package/src/defaults/skills/browser-use/SKILL.md +14 -6
  4. package/src/defaults/skills/setup/references/actions.md +6 -3
  5. package/src/defaults/skills/setup/references/surfaces.md +8 -4
  6. package/src/runtime/agent/orchestrator/context/role-instructions.mjs +0 -1
  7. package/src/runtime/agent/orchestrator/providers/anthropic-midstream-recovery.mjs +10 -1
  8. package/src/runtime/agent/orchestrator/session/cache/scoped-cache.mjs +40 -99
  9. package/src/runtime/agent/orchestrator/session/compact/runner.mjs +6 -1
  10. package/src/runtime/agent/orchestrator/session/loop/fresh-context.mjs +2 -0
  11. package/src/runtime/agent/orchestrator/session/loop/tool-exec/before-hook.mjs +7 -3
  12. package/src/runtime/agent/orchestrator/session/manager/session-prompt-composition.mjs +1 -1
  13. package/src/runtime/agent/orchestrator/tools/builtin/cache-layers.mjs +18 -38
  14. package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-path-fanout.mjs +59 -32
  15. package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-pattern-fanout.mjs +15 -10
  16. package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-request.mjs +1 -1
  17. package/src/runtime/agent/orchestrator/tools/builtin/lib/shared-native-scan.mjs +12 -0
  18. package/src/runtime/agent/orchestrator/tools/builtin/read-source-windows.mjs +14 -0
  19. package/src/runtime/agent/orchestrator/tools/graph-manifest.json +13 -13
  20. package/src/runtime/agent/orchestrator/tools/patch/apply-patch/codex-batch.mjs +3 -1
  21. package/src/runtime/agent/orchestrator/tools/patch/dispatch.mjs +17 -9
  22. package/src/runtime/agent/orchestrator/tools/patch/orchestrator.mjs +1 -0
  23. package/src/runtime/agent/orchestrator/tools/patch/wave.mjs +4 -0
  24. package/src/runtime/browser-bridge/input-fields.mjs +1 -1
  25. package/src/runtime/browser-bridge/tool-defs.mjs +1 -1
  26. package/src/runtime/computer-bridge/actions.mjs +0 -13
  27. package/src/runtime/media/adapters/antigravity-image.mjs +7 -3
  28. package/src/runtime/media/adapters/codex-image.mjs +12 -1
  29. package/src/runtime/media/adapters/gemini-image.mjs +17 -6
  30. package/src/runtime/media/adapters/gemini-video.mjs +19 -9
  31. package/src/runtime/media/adapters/xai-media.mjs +20 -7
  32. package/src/runtime/media/jobs.mjs +35 -3
  33. package/src/runtime/media/media-usage.mjs +126 -0
  34. package/src/runtime/media/tool.mjs +16 -4
  35. package/src/runtime/shared/llm/cost.mjs +50 -3
  36. package/src/runtime/shared/llm/model-catalog-projection.mjs +7 -2
  37. package/src/runtime/shared/llm/model-catalog.mjs +2 -1
  38. package/src/runtime/shared/llm/model-pricing-rates.mjs +21 -0
  39. package/src/runtime/shared/llm/usage-accounting.mjs +25 -5
  40. package/src/runtime/shared/llm/usage-context.mjs +6 -0
  41. package/src/runtime/shared/llm/usage-ledger-import.mjs +1 -0
  42. package/src/session-runtime/cwd-plugins/plugin-status.mjs +24 -4
  43. package/src/session-runtime/internal-tool-executor/feature-tools.mjs +9 -1
  44. package/src/session-runtime/services/agent-tool/index-write-queue.mjs +38 -6
  45. package/src/session-runtime/services/agent-tool/worker-index/row-store.mjs +9 -2
  46. package/src/session-runtime/services/agent-tool/worker-index.mjs +6 -3
  47. package/src/session-runtime/services/hook-bus/event-runner.mjs +10 -2
  48. package/src/session-runtime/services/hook-bus/tool-gate.mjs +5 -2
  49. package/src/session-runtime/setup-tool/tool-defs.mjs +1 -1
  50. package/src/session-runtime/workflow-agents-api/agent-route.mjs +8 -5
  51. package/src/tui/app/doctor.mjs +78 -2
  52. package/src/tui/dist/index.mjs +7 -3
  53. package/scripts/lib/bench-sandbox.test.mjs +0 -54
  54. package/scripts/lib/cli-args.test.mjs +0 -64
  55. package/scripts/lib/embedding-model-bench-core.test.mjs +0 -47
  56. package/scripts/lib/smoke-loop-shared.test.mjs +0 -22
  57. package/scripts/lib/trace-row.test.mjs +0 -47
  58. package/scripts/lib/trace-stats.test.mjs +0 -62
@@ -12,7 +12,9 @@ import { boundedSignal } from '../bounded-signal.mjs';
12
12
  import { decodeBase64Media, downloadGeminiMedia } from '../download.mjs';
13
13
  import { mediaError } from '../lanes.mjs';
14
14
  import { upstreamError } from '../upstream-error.mjs';
15
+ import { interactionsUsage, withReportedUsage } from '../media-usage.mjs';
15
16
 
17
+ const VEO_DEFAULT_SECONDS = 8;
16
18
  const BASE_URL = 'https://generativelanguage.googleapis.com/v1beta';
17
19
  const OMNI_TIMEOUT_MS = 600_000;
18
20
  const POLL_INTERVAL_MS = 8_000;
@@ -54,14 +56,15 @@ async function generateViaOmni({ model, prompt, options, references = [], signal
54
56
  });
55
57
  if (!res.ok) throw upstreamError('Gemini Omni video', res.status, await res.text().catch(() => ''));
56
58
  const data = await res.json();
59
+ const usage = interactionsUsage(data?.usage);
57
60
  for (const step of data?.steps || []) {
58
61
  const content = Array.isArray(step?.content) ? step.content : [step?.content].filter(Boolean);
59
62
  const video = content.find((item) => item?.type === 'video' && typeof item?.data === 'string');
60
63
  if (video) {
61
- return { bytes: decodeBase64Media(video.data, 'Gemini Omni video'), mime: video.mime_type || 'video/mp4' };
64
+ return { bytes: decodeBase64Media(video.data, 'Gemini Omni video'), mime: video.mime_type || 'video/mp4', usage };
62
65
  }
63
66
  }
64
- throw mediaError('Gemini Omni returned no video data', 'MEDIA_EMPTY_RESULT', 502);
67
+ throw withReportedUsage(mediaError('Gemini Omni returned no video data', 'MEDIA_EMPTY_RESULT', 502), usage);
65
68
  }
66
69
 
67
70
  async function generateViaVeo({ model, prompt, options, references = [], signal, onProgress, key }) {
@@ -114,13 +117,20 @@ async function generateViaVeo({ model, prompt, options, references = [], signal,
114
117
  const sample = data?.response?.generateVideoResponse?.generatedSamples?.[0] || data?.response?.generatedVideos?.[0];
115
118
  const uri = sample?.video?.uri || sample?.video?.fileUri;
116
119
  if (!uri) throw mediaError('Veo finished without a video URI', 'MEDIA_EMPTY_RESULT', 502);
117
- return {
118
- // The download is part of the generation budget: unbounded, it held an
119
- // active job slot (and the upstream connection) open indefinitely after
120
- // the poll loop finished.
121
- bytes: await downloadGeminiMedia(uri, { key, signal: boundedSignal(signal, deadline, TOTAL_TIMEOUT_MS) }),
122
- mime: 'video/mp4',
123
- };
120
+ // Veo reports no usage; it bills per generated second. The response does
121
+ // not echo the length, so the requested one (default 8s) is the billed one.
122
+ // A failed download still leaves the generated seconds billed.
123
+ const usage = { seconds: parameters.durationSeconds || VEO_DEFAULT_SECONDS, resolution: parameters.resolution || '720p' };
124
+ // The download is part of the generation budget: unbounded, it held an
125
+ // active job slot (and the upstream connection) open indefinitely after
126
+ // the poll loop finished.
127
+ const bytes = await downloadGeminiMedia(uri, {
128
+ key,
129
+ signal: boundedSignal(signal, deadline, TOTAL_TIMEOUT_MS),
130
+ }).catch((error) => {
131
+ throw withReportedUsage(error, usage);
132
+ });
133
+ return { bytes, mime: 'video/mp4', usage };
124
134
  }
125
135
  }
126
136
 
@@ -10,6 +10,7 @@ import { boundedSignal, timeoutSignal } from '../bounded-signal.mjs';
10
10
  import { decodeBase64Media, downloadPublicMedia } from '../download.mjs';
11
11
  import { mediaError } from '../lanes.mjs';
12
12
  import { upstreamError } from '../upstream-error.mjs';
13
+ import { withReportedUsage, xaiUsage } from '../media-usage.mjs';
13
14
 
14
15
  const POLL_INTERVAL_MS = 4_000;
15
16
  const START_TIMEOUT_MS = 60_000;
@@ -72,6 +73,7 @@ export async function generateImage({ lane, model, prompt, options = {}, referen
72
73
  bytes: decodeBase64Media(entry.b64_json, 'xAI image'),
73
74
  mime: entry.mime_type || 'image/png',
74
75
  revisedPrompt: entry.revised_prompt || null,
76
+ usage: xaiUsage(data?.usage),
75
77
  };
76
78
  }
77
79
 
@@ -114,18 +116,29 @@ export async function generateVideo({ lane, model, prompt, options = {}, referen
114
116
  if (typeof data?.progress === 'number' && typeof onProgress === 'function') onProgress(data.progress);
115
117
  if (data?.status === 'done') {
116
118
  const url = data?.video?.url;
117
- if (!url) throw mediaError('xAI video finished without a URL', 'MEDIA_EMPTY_RESULT', 502);
119
+ // The billed cost (when reported) is the price; seconds + resolution
120
+ // let the catalog's per-second rate price it otherwise.
121
+ const seconds = Number(data?.video?.duration) || duration;
122
+ const usage = { ...xaiUsage(data?.usage), seconds, resolution };
123
+ if (!url) throw withReportedUsage(mediaError('xAI video finished without a URL', 'MEDIA_EMPTY_RESULT', 502), usage);
124
+ const bytes = await downloadPublicMedia(url, { signal, label: 'xAI video' }).catch((error) => {
125
+ throw withReportedUsage(error, usage);
126
+ });
118
127
  return {
119
- bytes: await downloadPublicMedia(url, { signal, label: 'xAI video' }),
128
+ bytes,
120
129
  mime: 'video/mp4',
121
- durationSeconds: Number(data?.video?.duration) || duration,
130
+ durationSeconds: seconds,
131
+ usage,
122
132
  };
123
133
  }
124
134
  if (data?.status === 'failed' || data?.status === 'expired') {
125
- throw mediaError(
126
- `xAI video ${data.status}${data?.error?.code ? `: ${data.error.code}` : ''}`,
127
- 'MEDIA_UPSTREAM_FAILED',
128
- 502
135
+ throw withReportedUsage(
136
+ mediaError(
137
+ `xAI video ${data.status}${data?.error?.code ? `: ${data.error.code}` : ''}`,
138
+ 'MEDIA_UPSTREAM_FAILED',
139
+ 502
140
+ ),
141
+ xaiUsage(data?.usage)
129
142
  );
130
143
  }
131
144
  }
@@ -11,6 +11,7 @@ import { MAX_GENERATED_MEDIA_BYTES } from './download.mjs';
11
11
  import { mediaError, resolveMediaRequest } from './lanes.mjs';
12
12
  import { saveMediaAsset } from './store.mjs';
13
13
  import { setMediaDefault } from './defaults.mjs';
14
+ import { recordMediaUsage } from './media-usage.mjs';
14
15
 
15
16
  const JOBS = new Map();
16
17
  // Finished jobs stay readable for a while so a slow poller still sees the
@@ -77,7 +78,16 @@ async function runAdapter({ lane, kind, model, requestModel, prompt, options, re
77
78
  * Validate + start one generation. Returns the initial snapshot immediately;
78
79
  * the caller polls getMediaJob for progress and the finished asset id.
79
80
  */
80
- export async function startMediaJob({ lane: laneId, kind, model, prompt, options = {}, references = [] } = {}) {
81
+ export async function startMediaJob({
82
+ lane: laneId,
83
+ kind,
84
+ model,
85
+ prompt,
86
+ options = {},
87
+ references = [],
88
+ sessionId = '',
89
+ sourceType = '',
90
+ } = {}) {
81
91
  const text = String(prompt || '').trim();
82
92
  if (!text) throw mediaError('prompt is required', 'MEDIA_PROMPT_REQUIRED');
83
93
  if (text.length > MAX_PROMPT_CHARS) throw mediaError('prompt is too long', 'MEDIA_PROMPT_TOO_LONG');
@@ -128,13 +138,31 @@ export async function startMediaJob({ lane: laneId, kind, model, prompt, options
128
138
  controller,
129
139
  };
130
140
  JOBS.set(job.id, job);
131
- void runJob(job, resolved, { requestModel: modelEntry?.requestModel, options, references: refs });
141
+ void runJob(job, resolved, {
142
+ requestModel: modelEntry?.requestModel,
143
+ options,
144
+ references: refs,
145
+ sessionId,
146
+ sourceType,
147
+ });
132
148
  return snapshot(job);
133
149
  }
134
150
 
135
151
  /** Drive one started job to a terminal state; never rejects. */
136
- async function runJob(job, resolved, { requestModel, options, references }) {
152
+ async function runJob(job, resolved, { requestModel, options, references, sessionId, sourceType }) {
137
153
  const { controller } = job;
154
+ // One ledger row per generation. The ChatGPT lane bills the orchestrator
155
+ // model that ran the hosted tool, not the "auto" image route.
156
+ const record = (usage) =>
157
+ recordMediaUsage({
158
+ lane: job.lane,
159
+ model: job.model,
160
+ pricingModel: job.lane === 'openai-oauth' ? requestModel : undefined,
161
+ usage,
162
+ sessionId,
163
+ sourceType,
164
+ durationMs: Date.now() - job.startedAt,
165
+ });
138
166
  try {
139
167
  const result = await runAdapter({
140
168
  lane: resolved.lane,
@@ -154,6 +182,8 @@ async function runJob(job, resolved, { requestModel, options, references }) {
154
182
  if (next > job.progress) job.progress = next;
155
183
  },
156
184
  });
185
+ // Billed once the provider returned a result, whatever happens to the bytes next.
186
+ await record(result?.usage);
157
187
  if (!Buffer.isBuffer(result?.bytes) || !result.bytes.length || result.bytes.length > MAX_GENERATED_MEDIA_BYTES) {
158
188
  throw mediaError('generated media exceeds the media size limit', 'MEDIA_RESULT_TOO_LARGE', 502);
159
189
  }
@@ -175,6 +205,8 @@ async function runJob(job, resolved, { requestModel, options, references }) {
175
205
  job.status = 'done';
176
206
  } catch (err) {
177
207
  const canceled = controller.signal.aborted || err?.code === 'MEDIA_CANCELED' || err?.name === 'AbortError';
208
+ // A failure is recorded only when the provider reported usage for it.
209
+ if (err?.usage) await record(err.usage);
178
210
  job.status = canceled ? 'canceled' : 'failed';
179
211
  job.error = canceled ? 'canceled' : String(err?.message || err).slice(0, 500);
180
212
  job.errorCode = err?.code || null;
@@ -0,0 +1,126 @@
1
+ /**
2
+ * Usage ledger rows for media generations.
3
+ *
4
+ * Adapters return (or attach to their error) a normalized `usage` object read
5
+ * from what the upstream response actually reported:
6
+ * { inputTokens, outputTokens, cachedTokens, outputImageTokens, outputVideoTokens,
7
+ * costUsd, images, seconds, resolution }
8
+ * `jobs.mjs` records exactly one row per generation through recordMediaUsage.
9
+ * Nothing here estimates tokens: a field the provider did not report stays absent.
10
+ */
11
+ import { getUsageLedger, makeUsageRecord } from '../shared/llm/usage-ledger.mjs';
12
+ import { ACCOUNT_PROVIDERS } from '../shared/provider-accounts.mjs';
13
+ import { currentProviderAccountId } from '../shared/provider-auth-binding.mjs';
14
+
15
+ const count = (value) => (Number.isFinite(Number(value)) && Number(value) > 0 ? Math.trunc(Number(value)) : 0);
16
+ const pick = (source, ...keys) => {
17
+ for (const key of keys) if (source?.[key] != null) return source[key];
18
+ return undefined;
19
+ };
20
+
21
+ const TICKS_PER_USD = 1e10;
22
+
23
+ /** Token usage when the provider reported any; null otherwise. */
24
+ function tokenUsage(fields) {
25
+ const usage = Object.fromEntries(Object.entries(fields).filter(([, value]) => value > 0));
26
+ return Object.keys(usage).length ? usage : null;
27
+ }
28
+
29
+ /** Gemini `generateContent` / Antigravity `usageMetadata` (camelCase or snake_case). */
30
+ export function geminiUsage(meta) {
31
+ if (!meta || typeof meta !== 'object') return null;
32
+ const details = pick(meta, 'candidatesTokensDetails', 'candidates_tokens_details');
33
+ const imageTokens = (Array.isArray(details) ? details : [])
34
+ .filter((entry) => String(entry?.modality || '').toUpperCase() === 'IMAGE')
35
+ .reduce((sum, entry) => sum + count(pick(entry, 'tokenCount', 'token_count')), 0);
36
+ return tokenUsage({
37
+ inputTokens: count(pick(meta, 'promptTokenCount', 'prompt_token_count')),
38
+ cachedTokens: count(pick(meta, 'cachedContentTokenCount', 'cached_content_token_count')),
39
+ outputTokens:
40
+ count(pick(meta, 'candidatesTokenCount', 'candidates_token_count')) +
41
+ count(pick(meta, 'thoughtsTokenCount', 'thoughts_token_count')),
42
+ outputImageTokens: imageTokens,
43
+ });
44
+ }
45
+
46
+ /** Gemini Interactions API `usage` (omni video). */
47
+ export function interactionsUsage(usage) {
48
+ if (!usage || typeof usage !== 'object') return null;
49
+ const videoTokens = (Array.isArray(usage.output_tokens_by_modality) ? usage.output_tokens_by_modality : [])
50
+ .filter((entry) => String(entry?.modality || '').toLowerCase() === 'video')
51
+ .reduce((sum, entry) => sum + count(entry?.tokens), 0);
52
+ return tokenUsage({
53
+ inputTokens: count(usage.total_input_tokens),
54
+ cachedTokens: count(usage.total_cached_tokens),
55
+ outputTokens: count(usage.total_output_tokens) + count(usage.total_thought_tokens),
56
+ outputVideoTokens: videoTokens,
57
+ });
58
+ }
59
+
60
+ /** OpenAI / Codex Responses `usage` from response.completed. */
61
+ export function responsesUsage(usage) {
62
+ if (!usage || typeof usage !== 'object') return null;
63
+ return tokenUsage({
64
+ inputTokens: count(usage.input_tokens),
65
+ cachedTokens: count(usage.input_tokens_details?.cached_tokens),
66
+ outputTokens: count(usage.output_tokens),
67
+ });
68
+ }
69
+
70
+ /** xAI image/video `usage.cost_in_usd_ticks` (1 USD = 1e10 ticks); the billed cost. */
71
+ export function xaiUsage(usage) {
72
+ const ticks = usage?.cost_in_usd_ticks;
73
+ return typeof ticks === 'number' && Number.isFinite(ticks) && ticks >= 0 ? { costUsd: ticks / TICKS_PER_USD } : null;
74
+ }
75
+
76
+ /** True when the provider reported something billable for the request. */
77
+ export function hasReportedUsage(usage) {
78
+ if (!usage) return false;
79
+ // Seconds are only attached once a video generation completed (billed per second).
80
+ return (
81
+ typeof usage.costUsd === 'number' ||
82
+ ['inputTokens', 'outputTokens', 'cachedTokens', 'seconds'].some((k) => usage[k] > 0)
83
+ );
84
+ }
85
+
86
+ /** Attach provider-reported usage to a failure so jobs.mjs can still record it. */
87
+ export function withReportedUsage(error, usage) {
88
+ if (hasReportedUsage(usage)) error.usage = usage;
89
+ return error;
90
+ }
91
+
92
+ /**
93
+ * Append one ledger row for a generation. A ledger failure never fails the
94
+ * generation; it is logged the way usage-accounting does.
95
+ */
96
+ export async function recordMediaUsage(args, getLedger = getUsageLedger) {
97
+ try {
98
+ const usage = args.usage || {};
99
+ const row = makeUsageRecord({
100
+ ts: Date.now(),
101
+ provider: args.lane,
102
+ model: args.model,
103
+ requestedModel: args.model,
104
+ pricingModel: args.pricingModel,
105
+ sessionId: args.sessionId,
106
+ sourceType: args.sourceType,
107
+ inputTokens: usage.inputTokens,
108
+ outputTokens: usage.outputTokens,
109
+ cacheReadTokens: usage.cachedTokens,
110
+ costUsd: usage.costUsd,
111
+ media: {
112
+ reportedCostUsd: usage.costUsd,
113
+ outputImageTokens: usage.outputImageTokens,
114
+ outputVideoTokens: usage.outputVideoTokens,
115
+ images: usage.images,
116
+ seconds: usage.seconds,
117
+ resolution: usage.resolution,
118
+ },
119
+ account: ACCOUNT_PROVIDERS.includes(args.lane) ? currentProviderAccountId(args.lane) : '',
120
+ durationMs: args.durationMs,
121
+ });
122
+ await getLedger()?.recordQueued(row);
123
+ } catch (error) {
124
+ process.stderr.write(`[usage-ledger] RECORD NOT SAVED: ${String(error?.message || error)}\n`);
125
+ }
126
+ }
@@ -280,7 +280,7 @@ function jobView(job) {
280
280
  };
281
281
  }
282
282
 
283
- async function generate(args, { cwd, signal, deps }) {
283
+ async function generate(args, { cwd, signal, deps, sessionId, sourceType }) {
284
284
  const kind = clean(args.kind);
285
285
  if (!MEDIA_KINDS.includes(kind)) throw new MediaToolError(`generate requires kind: ${MEDIA_KINDS.join(' | ')}`);
286
286
  const prompt = clean(args.prompt);
@@ -314,7 +314,16 @@ async function generate(args, { cwd, signal, deps }) {
314
314
  quality: clean(args.quality),
315
315
  });
316
316
  const references = await readReferences(args.references, cwd, model.controls);
317
- const started = await jobs.startMediaJob({ lane: lane.id, kind, model: modelId, prompt, options, references });
317
+ const started = await jobs.startMediaJob({
318
+ lane: lane.id,
319
+ kind,
320
+ model: modelId,
321
+ prompt,
322
+ options,
323
+ references,
324
+ sessionId,
325
+ sourceType,
326
+ });
318
327
  const base = { lane: lane.id, model: modelId, laneSource, options, referenceCount: references.length, prompt };
319
328
  if (args.wait === false) {
320
329
  return {
@@ -371,7 +380,10 @@ async function cancel(args, { deps }) {
371
380
  return { ok: true, ...jobView(jobs.getMediaJob(id)), canceled: canceled === true };
372
381
  }
373
382
 
374
- export async function executeMediaTool(args = {}, { cwd = process.cwd(), signal = null, deps = null } = {}) {
383
+ export async function executeMediaTool(
384
+ args = {},
385
+ { cwd = process.cwd(), signal = null, deps = null, sessionId = '', sourceType = '' } = {}
386
+ ) {
375
387
  const action = clean(args.action).toLowerCase();
376
388
  try {
377
389
  if (!MEDIA_ACTIONS.includes(action))
@@ -388,7 +400,7 @@ export async function executeMediaTool(args = {}, { cwd = process.cwd(), signal
388
400
  ),
389
401
  });
390
402
  }
391
- if (action === 'generate') return mediaToolResult(await generate(args, { cwd, signal, deps }));
403
+ if (action === 'generate') return mediaToolResult(await generate(args, { cwd, signal, deps, sessionId, sourceType }));
392
404
  if (action === 'status') return mediaToolResult(await status(args, { cwd, deps }));
393
405
  return mediaToolResult(await cancel(args, { deps }));
394
406
  } catch (error) {
@@ -39,6 +39,35 @@ export function billableInputTokensForProvider(provider, inputTokens, cacheReadT
39
39
  return Math.max(input - (Number(cacheReadTokens) || 0) - (Number(cacheWriteTokens) || 0), 0);
40
40
  }
41
41
 
42
+ /**
43
+ * Media billed by something other than the text token slots: image/video output
44
+ * tokens, whole images, or video seconds (per resolution when the catalog lists
45
+ * one). A used unit without a catalog rate is reported missing, never free.
46
+ */
47
+ function mediaCharges(meta, media, imageOut, videoOut) {
48
+ const n = (value) => (Number.isFinite(Number(value)) && Number(value) > 0 ? Number(value) : 0);
49
+ const out = { usd: 0, charged: false, missing: [], rates: {} };
50
+ const bill = (amount, key, rate, scale = 1) => {
51
+ if (amount <= 0) return;
52
+ out.charged = true;
53
+ if (rate == null) out.missing.push(key);
54
+ else {
55
+ out.usd += (amount * rate) / scale;
56
+ out.rates[key] = rate;
57
+ }
58
+ };
59
+ bill(imageOut, 'outputImageCostPerM', meta.outputImageCostPerM, 1_000_000);
60
+ bill(videoOut, 'outputVideoCostPerM', meta.outputVideoCostPerM, 1_000_000);
61
+ bill(n(media.images), 'outputCostPerImage', meta.outputCostPerImage);
62
+ const resolution = String(media.resolution || '').toLowerCase();
63
+ bill(
64
+ n(media.seconds),
65
+ 'outputCostPerSecond',
66
+ meta.outputCostPerSecondByResolution?.[resolution] ?? meta.outputCostPerSecond
67
+ );
68
+ return out;
69
+ }
70
+
42
71
  /**
43
72
  * Price normalized token slots once. Null means unknown, not a free request.
44
73
  * Rates are returned so a durable record keeps the price applied at ingestion.
@@ -68,6 +97,15 @@ export function priceUsage(args) {
68
97
  ...(args.serviceTier ? { serviceTier: args.serviceTier } : {}),
69
98
  ...(written1h ? { cacheWrite1hTokens: written1h } : {}),
70
99
  };
100
+ // A provider-billed figure for a media request (e.g. xAI cost_in_usd_ticks)
101
+ // is the price, with or without a catalog row.
102
+ const media = args.media || null;
103
+ if (typeof media?.reportedCostUsd === 'number' && Number.isFinite(media.reportedCostUsd) && media.reportedCostUsd >= 0)
104
+ return {
105
+ input,
106
+ costUsd: Number(media.reportedCostUsd.toFixed(6)),
107
+ rates: { ...provenance, pricingSource: 'provider' },
108
+ };
71
109
  if (args.inputTokensKnown === false || !meta)
72
110
  return {
73
111
  input,
@@ -107,7 +145,11 @@ export function priceUsage(args) {
107
145
  // Fast mode bills 2x standard rates on every fast-capable Opus.
108
146
  if (anthropicFast) multiplier *= 2;
109
147
  const keys = PRICING_RATE_KEYS;
110
- const tokens = [input, n(args.outputTokens), cached, written - written1h];
148
+ // Image/video output tokens bill above the text output rate; the row still
149
+ // carries the full output count.
150
+ const imageOut = media ? Math.min(n(media.outputImageTokens), n(args.outputTokens)) : 0;
151
+ const videoOut = media ? Math.min(n(media.outputVideoTokens), n(args.outputTokens) - imageOut) : 0;
152
+ const tokens = [input, n(args.outputTokens) - imageOut - videoOut, cached, written - written1h];
111
153
  const tierRates = ratesForPrompt(rateMeta, promptTokens);
112
154
  const rates = {
113
155
  ...provenance,
@@ -116,6 +158,10 @@ export function priceUsage(args) {
116
158
  if (written1h) rates.cacheWrite1hCostPerM = rates.inputCostPerM === null ? null : rates.inputCostPerM * 2;
117
159
  const missingRates = keys.filter((key, i) => tokens[i] > 0 && rates[key] === null);
118
160
  if (written1h && rates.cacheWrite1hCostPerM === null) missingRates.push('cacheWrite1hCostPerM');
161
+ const charge = media ? mediaCharges(meta, media, imageOut, videoOut) : { usd: 0, charged: false, missing: [] };
162
+ missingRates.push(...charge.missing);
163
+ if (media && !charge.charged && tokens.every((amount) => amount === 0) && !written1h)
164
+ return { input, costUsd: null, rates: { ...rates, unpricedReason: 'usage-not-reported' } };
119
165
  if (missingRates.length) {
120
166
  rates.unpricedReason = 'missing-rate';
121
167
  rates.missingRates = missingRates;
@@ -124,8 +170,9 @@ export function priceUsage(args) {
124
170
  const costUsd =
125
171
  (tokens.reduce((sum, amount, i) => sum + amount * (rates[keys[i]] ?? 0), 0) +
126
172
  written1h * (rates.cacheWrite1hCostPerM ?? 0)) /
127
- 1_000_000;
128
- return { input, costUsd: Number(costUsd.toFixed(6)), rates };
173
+ 1_000_000 +
174
+ charge.usd;
175
+ return { input, costUsd: Number(costUsd.toFixed(6)), rates: { ...rates, ...charge.rates } };
129
176
  }
130
177
 
131
178
  /**
@@ -124,6 +124,10 @@ const LITELLM_NUMBER_FIELDS = [
124
124
  'output_cost_per_token',
125
125
  'cache_read_input_token_cost',
126
126
  'cache_creation_input_token_cost',
127
+ 'output_cost_per_image',
128
+ 'output_cost_per_image_token',
129
+ 'output_cost_per_video_token',
130
+ 'output_cost_per_second',
127
131
  ];
128
132
  // _normalize tests each of these with `=== true`, so only a true value carries
129
133
  // information; anything else is indistinguishable from absent.
@@ -148,9 +152,10 @@ function projectLitellmRow(row) {
148
152
  // litellmPricing), not just the base price.
149
153
  for (const [field, value] of Object.entries(row)) {
150
154
  if (
151
- /^(?:input_cost_per_token|output_cost_per_token|cache_read_input_token_cost|cache_creation_input_token_cost)(?:_above_\d+k_tokens)?(?:_priority)?$/.test(
155
+ (/^(?:input_cost_per_token|output_cost_per_token|cache_read_input_token_cost|cache_creation_input_token_cost)(?:_above_\d+k_tokens)?(?:_priority)?$/.test(
152
156
  field
153
- ) &&
157
+ ) ||
158
+ /^output_cost_per_second_\w+$/.test(field)) &&
154
159
  typeof value === 'number' &&
155
160
  Number.isFinite(value)
156
161
  )
@@ -26,7 +26,7 @@ import {
26
26
  cachedProviderModelListsSync,
27
27
  providerCachedModelsSync,
28
28
  } from './provider-catalog-cache.mjs';
29
- import { litellmPricing, modelsDevPricing, PRICING_RATE_KEYS } from './model-pricing-rates.mjs';
29
+ import { litellmMediaPricing, litellmPricing, modelsDevPricing, PRICING_RATE_KEYS } from './model-pricing-rates.mjs';
30
30
  // Both overlays are narrowed to their read surface before becoming resident;
31
31
  // the disk caches below still receive the full payload.
32
32
  import { projectLitellmCatalog, projectModelsDevCatalog } from './model-catalog-projection.mjs';
@@ -597,6 +597,7 @@ function _normalize(entry) {
597
597
  contextWindow: entry.max_input_tokens || entry.max_tokens || null,
598
598
  outputTokens: entry.max_output_tokens || null,
599
599
  ...litellmPricing(entry),
600
+ ...litellmMediaPricing(entry),
600
601
  ...(PRICING_RATE_KEYS.some((key) => fastPricing[key] != null) ? { fastPricing } : {}),
601
602
  ...(entry.off_peak_multiplier ? { offPeakMultiplier: entry.off_peak_multiplier } : {}),
602
603
  supportsVision: entry.supports_vision === true,
@@ -51,6 +51,27 @@ export function litellmPricing(entry, suffix = '') {
51
51
  };
52
52
  }
53
53
 
54
+ /**
55
+ * Non-token media rates published by LiteLLM: USD per generated image, USD per
56
+ * generated video second (optionally per resolution, `output_cost_per_second_<res>`),
57
+ * and USD/M for image / video output tokens, which bill above the text output rate.
58
+ */
59
+ export function litellmMediaPricing(entry) {
60
+ const perM = (value) => (validRate(value) ? value * 1_000_000 : null);
61
+ const byResolution = {};
62
+ for (const [key, value] of Object.entries(entry || {})) {
63
+ const match = key.match(/^output_cost_per_second_(.+)$/);
64
+ if (match && validRate(value)) byResolution[match[1].toLowerCase()] = value;
65
+ }
66
+ return {
67
+ outputImageCostPerM: perM(entry?.output_cost_per_image_token),
68
+ outputVideoCostPerM: perM(entry?.output_cost_per_video_token),
69
+ outputCostPerImage: validRate(entry?.output_cost_per_image) ? entry.output_cost_per_image : null,
70
+ outputCostPerSecond: validRate(entry?.output_cost_per_second) ? entry.output_cost_per_second : null,
71
+ outputCostPerSecondByResolution: byResolution,
72
+ };
73
+ }
74
+
54
75
  export function modelsDevPricing(cost) {
55
76
  // Structured tiers supersede the older context_over_200k compatibility
56
77
  // field; it can coexist with a tier whose actual boundary is not 200k.
@@ -4,6 +4,13 @@ import { withUsageContext } from './usage-context.mjs';
4
4
  import { ACCOUNT_PROVIDERS } from '../provider-accounts.mjs';
5
5
  import { currentProviderAccountId } from '../provider-auth-binding.mjs';
6
6
 
7
+ // Results and errors already written. A provider-local re-send (model
8
+ // fallback, catalog retry) runs its own accounting and its result or error
9
+ // then returns through the enclosing send, which must not write it again.
10
+ const recorded = new WeakSet();
11
+ const hasTokens = (usage) =>
12
+ ['inputTokens', 'outputTokens', 'cachedTokens', 'cacheWriteTokens'].some((key) => Number(usage?.[key]) > 0);
13
+
7
14
  /**
8
15
  * Runs at the common provider boundary, not inside optional diagnostic IO.
9
16
  * Provider-local retries remain owned by the provider. Accounting failure must
@@ -12,7 +19,9 @@ import { currentProviderAccountId } from '../provider-auth-binding.mjs';
12
19
  export async function accountProviderSend(provider, instance, send, model, opts = {}) {
13
20
  const requestId = randomUUID();
14
21
  const startedAt = Date.now();
15
- const sessionId = opts.sessionId || opts.session?.id;
22
+ // usageSessionId: a request isolated under its own provider session id
23
+ // (compaction's `:compact`) whose spend belongs to the source session.
24
+ const sessionId = opts.usageSessionId || opts.sessionId || opts.session?.id;
16
25
  const sourceType = opts.session?.sourceType || opts.sourceType || opts.requestKind || '';
17
26
  const inputTokensInclusive = instance.constructor?.inputExcludesCache !== true;
18
27
  let ledger;
@@ -30,16 +39,18 @@ export async function accountProviderSend(provider, instance, send, model, opts
30
39
  sourceType,
31
40
  inputTokensInclusive,
32
41
  };
33
- const record = async (result) => {
42
+ const record = async (result, id, owner) => {
34
43
  if (!result?.usage) return;
35
44
  // A nested send (e.g. a fallback model re-send) already stamped its own
36
45
  // final attempt's tier; the outer context only saw the abandoned attempt.
37
46
  result.requestServiceTier ??= identity.requestServiceTier || '';
47
+ if (recorded.has(owner)) return;
38
48
  if (openingError) throw openingError;
39
49
  if (!ledger) return;
50
+ recorded.add(owner);
40
51
  const usage = result.usage;
41
52
  const row = makeUsageRecord({
42
- id: result.responseId ? undefined : requestId,
53
+ id: result.responseId ? undefined : id,
43
54
  ts: Date.now(),
44
55
  provider,
45
56
  model: result.model || model,
@@ -68,23 +79,32 @@ export async function accountProviderSend(provider, instance, send, model, opts
68
79
  // SQLite write no longer runs on the event loop.
69
80
  await ledger.recordQueued(row);
70
81
  };
71
- const save = async (result) => {
82
+ const save = async (result, id = requestId, owner = result) => {
72
83
  try {
73
- await record(result);
84
+ await record(result, id, owner);
74
85
  } catch (error) {
75
86
  result.usageAccountingError = String(error?.message || error);
76
87
  process.stderr.write(`[usage-ledger] RECORD NOT SAVED: ${result.usageAccountingError}\n`);
77
88
  }
78
89
  };
90
+ const saveAbandoned = async () => {
91
+ for (const [index, attempt] of (identity.abandonedUsage || []).entries()) {
92
+ if (hasTokens(attempt.usage)) await save(attempt, `${requestId}:abandoned:${index}`);
93
+ }
94
+ };
79
95
  let result;
80
96
  try {
81
97
  result = await withUsageContext(identity, send);
82
98
  } catch (error) {
99
+ await saveAbandoned();
83
100
  // Only provider-reported partial usage is recordable; never invent
84
101
  // tokens for a failed request or reinterpret an error as a success.
85
102
  if (error?.usage) await save(error);
103
+ else if (hasTokens(error?.partialUsage))
104
+ await save({ usage: error.partialUsage, model: error.partialModel }, requestId, error);
86
105
  throw error;
87
106
  }
107
+ await saveAbandoned();
88
108
  await save(result);
89
109
  return result;
90
110
  }
@@ -11,3 +11,9 @@ export const noteRequestServiceTier = (tier) => {
11
11
  const identity = context.getStore();
12
12
  if (identity) identity.requestServiceTier = tier || '';
13
13
  };
14
+ // A provider-local retry abandons an attempt the provider still billed; the
15
+ // enclosing send records it alongside its own final usage.
16
+ export const noteAbandonedUsage = (usage, model) => {
17
+ const identity = context.getStore();
18
+ if (identity && usage) (identity.abandonedUsage ||= []).push({ usage, model });
19
+ };
@@ -43,6 +43,7 @@ export function importTraceRow(row) {
43
43
  uncachedInputTokens: raw ? (row.uncached_input_tokens ?? payload.uncached_input_tokens) : undefined,
44
44
  cacheReadTokens: raw ? row.cached_tokens : row.cacheReadTokens,
45
45
  cacheWriteTokens: raw ? row.cache_write_tokens : row.cacheWriteTokens,
46
+ cacheWrite1hTokens: raw ? payload.raw_usage?.cache_creation?.ephemeral_1h_input_tokens : undefined,
46
47
  costUsd: raw ? undefined : row.costUsd,
47
48
  sessionId: row.session_id || row.sessionId,
48
49
  sourceType: row.sourceType || row.source_type || payload.sourceType || payload.source_type,