mixdog 1.0.2 → 1.0.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +8 -0
  2. package/package.json +1 -1
  3. package/src/defaults/skills/browser-use/SKILL.md +14 -6
  4. package/src/defaults/skills/goal-management/SKILL.md +2 -2
  5. package/src/runtime/agent/orchestrator/providers/anthropic-midstream-recovery.mjs +10 -1
  6. package/src/runtime/agent/orchestrator/runtime-core/goal-tool-defs.mjs +1 -1
  7. package/src/runtime/agent/orchestrator/session/cache/scoped-cache.mjs +40 -99
  8. package/src/runtime/agent/orchestrator/session/compact/runner.mjs +6 -1
  9. package/src/runtime/agent/orchestrator/session/loop/fresh-context.mjs +2 -0
  10. package/src/runtime/agent/orchestrator/session/loop/tool-exec/before-hook.mjs +7 -3
  11. package/src/runtime/agent/orchestrator/tools/builtin/cache-layers.mjs +18 -38
  12. package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-path-fanout.mjs +59 -32
  13. package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-pattern-fanout.mjs +15 -10
  14. package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-request.mjs +1 -1
  15. package/src/runtime/agent/orchestrator/tools/builtin/lib/shared-native-scan.mjs +12 -0
  16. package/src/runtime/agent/orchestrator/tools/builtin/read-source-windows.mjs +14 -0
  17. package/src/runtime/agent/orchestrator/tools/graph-manifest.json +13 -13
  18. package/src/runtime/agent/orchestrator/tools/patch/apply-patch/codex-batch.mjs +3 -1
  19. package/src/runtime/agent/orchestrator/tools/patch/dispatch.mjs +17 -9
  20. package/src/runtime/agent/orchestrator/tools/patch/orchestrator.mjs +1 -0
  21. package/src/runtime/agent/orchestrator/tools/patch/wave.mjs +4 -0
  22. package/src/runtime/browser-bridge/input-fields.mjs +1 -1
  23. package/src/runtime/browser-bridge/tool-defs.mjs +1 -1
  24. package/src/runtime/computer-bridge/actions.mjs +0 -13
  25. package/src/runtime/media/adapters/antigravity-image.mjs +7 -3
  26. package/src/runtime/media/adapters/codex-image.mjs +12 -1
  27. package/src/runtime/media/adapters/gemini-image.mjs +17 -6
  28. package/src/runtime/media/adapters/gemini-video.mjs +19 -9
  29. package/src/runtime/media/adapters/xai-media.mjs +20 -7
  30. package/src/runtime/media/jobs.mjs +35 -3
  31. package/src/runtime/media/media-usage.mjs +126 -0
  32. package/src/runtime/media/tool.mjs +16 -4
  33. package/src/runtime/shared/llm/cost.mjs +50 -3
  34. package/src/runtime/shared/llm/model-catalog-projection.mjs +7 -2
  35. package/src/runtime/shared/llm/model-catalog.mjs +2 -1
  36. package/src/runtime/shared/llm/model-pricing-rates.mjs +21 -0
  37. package/src/runtime/shared/llm/usage-accounting.mjs +25 -5
  38. package/src/runtime/shared/llm/usage-context.mjs +6 -0
  39. package/src/runtime/shared/llm/usage-ledger-import.mjs +1 -0
  40. package/src/session-runtime/internal-tool-executor/feature-tools.mjs +9 -1
  41. package/src/session-runtime/services/agent-tool/index-write-queue.mjs +38 -6
  42. package/src/session-runtime/services/agent-tool/worker-index/row-store.mjs +9 -2
  43. package/src/session-runtime/services/agent-tool/worker-index.mjs +6 -3
  44. package/src/session-runtime/services/hook-bus/event-runner.mjs +10 -2
  45. package/src/session-runtime/services/hook-bus/tool-gate.mjs +5 -2
  46. package/src/tui/dist/index.mjs +7 -3
@@ -11,6 +11,7 @@ import { MAX_GENERATED_MEDIA_BYTES } from './download.mjs';
11
11
  import { mediaError, resolveMediaRequest } from './lanes.mjs';
12
12
  import { saveMediaAsset } from './store.mjs';
13
13
  import { setMediaDefault } from './defaults.mjs';
14
+ import { recordMediaUsage } from './media-usage.mjs';
14
15
 
15
16
  const JOBS = new Map();
16
17
  // Finished jobs stay readable for a while so a slow poller still sees the
@@ -77,7 +78,16 @@ async function runAdapter({ lane, kind, model, requestModel, prompt, options, re
77
78
  * Validate + start one generation. Returns the initial snapshot immediately;
78
79
  * the caller polls getMediaJob for progress and the finished asset id.
79
80
  */
80
- export async function startMediaJob({ lane: laneId, kind, model, prompt, options = {}, references = [] } = {}) {
81
+ export async function startMediaJob({
82
+ lane: laneId,
83
+ kind,
84
+ model,
85
+ prompt,
86
+ options = {},
87
+ references = [],
88
+ sessionId = '',
89
+ sourceType = '',
90
+ } = {}) {
81
91
  const text = String(prompt || '').trim();
82
92
  if (!text) throw mediaError('prompt is required', 'MEDIA_PROMPT_REQUIRED');
83
93
  if (text.length > MAX_PROMPT_CHARS) throw mediaError('prompt is too long', 'MEDIA_PROMPT_TOO_LONG');
@@ -128,13 +138,31 @@ export async function startMediaJob({ lane: laneId, kind, model, prompt, options
128
138
  controller,
129
139
  };
130
140
  JOBS.set(job.id, job);
131
- void runJob(job, resolved, { requestModel: modelEntry?.requestModel, options, references: refs });
141
+ void runJob(job, resolved, {
142
+ requestModel: modelEntry?.requestModel,
143
+ options,
144
+ references: refs,
145
+ sessionId,
146
+ sourceType,
147
+ });
132
148
  return snapshot(job);
133
149
  }
134
150
 
135
151
  /** Drive one started job to a terminal state; never rejects. */
136
- async function runJob(job, resolved, { requestModel, options, references }) {
152
+ async function runJob(job, resolved, { requestModel, options, references, sessionId, sourceType }) {
137
153
  const { controller } = job;
154
+ // One ledger row per generation. The ChatGPT lane bills the orchestrator
155
+ // model that ran the hosted tool, not the "auto" image route.
156
+ const record = (usage) =>
157
+ recordMediaUsage({
158
+ lane: job.lane,
159
+ model: job.model,
160
+ pricingModel: job.lane === 'openai-oauth' ? requestModel : undefined,
161
+ usage,
162
+ sessionId,
163
+ sourceType,
164
+ durationMs: Date.now() - job.startedAt,
165
+ });
138
166
  try {
139
167
  const result = await runAdapter({
140
168
  lane: resolved.lane,
@@ -154,6 +182,8 @@ async function runJob(job, resolved, { requestModel, options, references }) {
154
182
  if (next > job.progress) job.progress = next;
155
183
  },
156
184
  });
185
+ // Billed once the provider returned a result, whatever happens to the bytes next.
186
+ await record(result?.usage);
157
187
  if (!Buffer.isBuffer(result?.bytes) || !result.bytes.length || result.bytes.length > MAX_GENERATED_MEDIA_BYTES) {
158
188
  throw mediaError('generated media exceeds the media size limit', 'MEDIA_RESULT_TOO_LARGE', 502);
159
189
  }
@@ -175,6 +205,8 @@ async function runJob(job, resolved, { requestModel, options, references }) {
175
205
  job.status = 'done';
176
206
  } catch (err) {
177
207
  const canceled = controller.signal.aborted || err?.code === 'MEDIA_CANCELED' || err?.name === 'AbortError';
208
+ // A failure is recorded only when the provider reported usage for it.
209
+ if (err?.usage) await record(err.usage);
178
210
  job.status = canceled ? 'canceled' : 'failed';
179
211
  job.error = canceled ? 'canceled' : String(err?.message || err).slice(0, 500);
180
212
  job.errorCode = err?.code || null;
@@ -0,0 +1,126 @@
1
+ /**
2
+ * Usage ledger rows for media generations.
3
+ *
4
+ * Adapters return (or attach to their error) a normalized `usage` object read
5
+ * from what the upstream response actually reported:
6
+ * { inputTokens, outputTokens, cachedTokens, outputImageTokens, outputVideoTokens,
7
+ * costUsd, images, seconds, resolution }
8
+ * `jobs.mjs` records exactly one row per generation through recordMediaUsage.
9
+ * Nothing here estimates tokens: a field the provider did not report stays absent.
10
+ */
11
+ import { getUsageLedger, makeUsageRecord } from '../shared/llm/usage-ledger.mjs';
12
+ import { ACCOUNT_PROVIDERS } from '../shared/provider-accounts.mjs';
13
+ import { currentProviderAccountId } from '../shared/provider-auth-binding.mjs';
14
+
15
+ const count = (value) => (Number.isFinite(Number(value)) && Number(value) > 0 ? Math.trunc(Number(value)) : 0);
16
+ const pick = (source, ...keys) => {
17
+ for (const key of keys) if (source?.[key] != null) return source[key];
18
+ return undefined;
19
+ };
20
+
21
+ const TICKS_PER_USD = 1e10;
22
+
23
+ /** Token usage when the provider reported any; null otherwise. */
24
+ function tokenUsage(fields) {
25
+ const usage = Object.fromEntries(Object.entries(fields).filter(([, value]) => value > 0));
26
+ return Object.keys(usage).length ? usage : null;
27
+ }
28
+
29
+ /** Gemini `generateContent` / Antigravity `usageMetadata` (camelCase or snake_case). */
30
+ export function geminiUsage(meta) {
31
+ if (!meta || typeof meta !== 'object') return null;
32
+ const details = pick(meta, 'candidatesTokensDetails', 'candidates_tokens_details');
33
+ const imageTokens = (Array.isArray(details) ? details : [])
34
+ .filter((entry) => String(entry?.modality || '').toUpperCase() === 'IMAGE')
35
+ .reduce((sum, entry) => sum + count(pick(entry, 'tokenCount', 'token_count')), 0);
36
+ return tokenUsage({
37
+ inputTokens: count(pick(meta, 'promptTokenCount', 'prompt_token_count')),
38
+ cachedTokens: count(pick(meta, 'cachedContentTokenCount', 'cached_content_token_count')),
39
+ outputTokens:
40
+ count(pick(meta, 'candidatesTokenCount', 'candidates_token_count')) +
41
+ count(pick(meta, 'thoughtsTokenCount', 'thoughts_token_count')),
42
+ outputImageTokens: imageTokens,
43
+ });
44
+ }
45
+
46
+ /** Gemini Interactions API `usage` (omni video). */
47
+ export function interactionsUsage(usage) {
48
+ if (!usage || typeof usage !== 'object') return null;
49
+ const videoTokens = (Array.isArray(usage.output_tokens_by_modality) ? usage.output_tokens_by_modality : [])
50
+ .filter((entry) => String(entry?.modality || '').toLowerCase() === 'video')
51
+ .reduce((sum, entry) => sum + count(entry?.tokens), 0);
52
+ return tokenUsage({
53
+ inputTokens: count(usage.total_input_tokens),
54
+ cachedTokens: count(usage.total_cached_tokens),
55
+ outputTokens: count(usage.total_output_tokens) + count(usage.total_thought_tokens),
56
+ outputVideoTokens: videoTokens,
57
+ });
58
+ }
59
+
60
+ /** OpenAI / Codex Responses `usage` from response.completed. */
61
+ export function responsesUsage(usage) {
62
+ if (!usage || typeof usage !== 'object') return null;
63
+ return tokenUsage({
64
+ inputTokens: count(usage.input_tokens),
65
+ cachedTokens: count(usage.input_tokens_details?.cached_tokens),
66
+ outputTokens: count(usage.output_tokens),
67
+ });
68
+ }
69
+
70
+ /** xAI image/video `usage.cost_in_usd_ticks` (1 USD = 1e10 ticks); the billed cost. */
71
+ export function xaiUsage(usage) {
72
+ const ticks = usage?.cost_in_usd_ticks;
73
+ return typeof ticks === 'number' && Number.isFinite(ticks) && ticks >= 0 ? { costUsd: ticks / TICKS_PER_USD } : null;
74
+ }
75
+
76
+ /** True when the provider reported something billable for the request. */
77
+ export function hasReportedUsage(usage) {
78
+ if (!usage) return false;
79
+ // Seconds are only attached once a video generation completed (billed per second).
80
+ return (
81
+ typeof usage.costUsd === 'number' ||
82
+ ['inputTokens', 'outputTokens', 'cachedTokens', 'seconds'].some((k) => usage[k] > 0)
83
+ );
84
+ }
85
+
86
+ /** Attach provider-reported usage to a failure so jobs.mjs can still record it. */
87
+ export function withReportedUsage(error, usage) {
88
+ if (hasReportedUsage(usage)) error.usage = usage;
89
+ return error;
90
+ }
91
+
92
+ /**
93
+ * Append one ledger row for a generation. A ledger failure never fails the
94
+ * generation; it is logged the way usage-accounting does.
95
+ */
96
+ export async function recordMediaUsage(args, getLedger = getUsageLedger) {
97
+ try {
98
+ const usage = args.usage || {};
99
+ const row = makeUsageRecord({
100
+ ts: Date.now(),
101
+ provider: args.lane,
102
+ model: args.model,
103
+ requestedModel: args.model,
104
+ pricingModel: args.pricingModel,
105
+ sessionId: args.sessionId,
106
+ sourceType: args.sourceType,
107
+ inputTokens: usage.inputTokens,
108
+ outputTokens: usage.outputTokens,
109
+ cacheReadTokens: usage.cachedTokens,
110
+ costUsd: usage.costUsd,
111
+ media: {
112
+ reportedCostUsd: usage.costUsd,
113
+ outputImageTokens: usage.outputImageTokens,
114
+ outputVideoTokens: usage.outputVideoTokens,
115
+ images: usage.images,
116
+ seconds: usage.seconds,
117
+ resolution: usage.resolution,
118
+ },
119
+ account: ACCOUNT_PROVIDERS.includes(args.lane) ? currentProviderAccountId(args.lane) : '',
120
+ durationMs: args.durationMs,
121
+ });
122
+ await getLedger()?.recordQueued(row);
123
+ } catch (error) {
124
+ process.stderr.write(`[usage-ledger] RECORD NOT SAVED: ${String(error?.message || error)}\n`);
125
+ }
126
+ }
@@ -280,7 +280,7 @@ function jobView(job) {
280
280
  };
281
281
  }
282
282
 
283
- async function generate(args, { cwd, signal, deps }) {
283
+ async function generate(args, { cwd, signal, deps, sessionId, sourceType }) {
284
284
  const kind = clean(args.kind);
285
285
  if (!MEDIA_KINDS.includes(kind)) throw new MediaToolError(`generate requires kind: ${MEDIA_KINDS.join(' | ')}`);
286
286
  const prompt = clean(args.prompt);
@@ -314,7 +314,16 @@ async function generate(args, { cwd, signal, deps }) {
314
314
  quality: clean(args.quality),
315
315
  });
316
316
  const references = await readReferences(args.references, cwd, model.controls);
317
- const started = await jobs.startMediaJob({ lane: lane.id, kind, model: modelId, prompt, options, references });
317
+ const started = await jobs.startMediaJob({
318
+ lane: lane.id,
319
+ kind,
320
+ model: modelId,
321
+ prompt,
322
+ options,
323
+ references,
324
+ sessionId,
325
+ sourceType,
326
+ });
318
327
  const base = { lane: lane.id, model: modelId, laneSource, options, referenceCount: references.length, prompt };
319
328
  if (args.wait === false) {
320
329
  return {
@@ -371,7 +380,10 @@ async function cancel(args, { deps }) {
371
380
  return { ok: true, ...jobView(jobs.getMediaJob(id)), canceled: canceled === true };
372
381
  }
373
382
 
374
- export async function executeMediaTool(args = {}, { cwd = process.cwd(), signal = null, deps = null } = {}) {
383
+ export async function executeMediaTool(
384
+ args = {},
385
+ { cwd = process.cwd(), signal = null, deps = null, sessionId = '', sourceType = '' } = {}
386
+ ) {
375
387
  const action = clean(args.action).toLowerCase();
376
388
  try {
377
389
  if (!MEDIA_ACTIONS.includes(action))
@@ -388,7 +400,7 @@ export async function executeMediaTool(args = {}, { cwd = process.cwd(), signal
388
400
  ),
389
401
  });
390
402
  }
391
- if (action === 'generate') return mediaToolResult(await generate(args, { cwd, signal, deps }));
403
+ if (action === 'generate') return mediaToolResult(await generate(args, { cwd, signal, deps, sessionId, sourceType }));
392
404
  if (action === 'status') return mediaToolResult(await status(args, { cwd, deps }));
393
405
  return mediaToolResult(await cancel(args, { deps }));
394
406
  } catch (error) {
@@ -39,6 +39,35 @@ export function billableInputTokensForProvider(provider, inputTokens, cacheReadT
39
39
  return Math.max(input - (Number(cacheReadTokens) || 0) - (Number(cacheWriteTokens) || 0), 0);
40
40
  }
41
41
 
42
+ /**
43
+ * Media billed by something other than the text token slots: image/video output
44
+ * tokens, whole images, or video seconds (per resolution when the catalog lists
45
+ * one). A used unit without a catalog rate is reported missing, never free.
46
+ */
47
+ function mediaCharges(meta, media, imageOut, videoOut) {
48
+ const n = (value) => (Number.isFinite(Number(value)) && Number(value) > 0 ? Number(value) : 0);
49
+ const out = { usd: 0, charged: false, missing: [], rates: {} };
50
+ const bill = (amount, key, rate, scale = 1) => {
51
+ if (amount <= 0) return;
52
+ out.charged = true;
53
+ if (rate == null) out.missing.push(key);
54
+ else {
55
+ out.usd += (amount * rate) / scale;
56
+ out.rates[key] = rate;
57
+ }
58
+ };
59
+ bill(imageOut, 'outputImageCostPerM', meta.outputImageCostPerM, 1_000_000);
60
+ bill(videoOut, 'outputVideoCostPerM', meta.outputVideoCostPerM, 1_000_000);
61
+ bill(n(media.images), 'outputCostPerImage', meta.outputCostPerImage);
62
+ const resolution = String(media.resolution || '').toLowerCase();
63
+ bill(
64
+ n(media.seconds),
65
+ 'outputCostPerSecond',
66
+ meta.outputCostPerSecondByResolution?.[resolution] ?? meta.outputCostPerSecond
67
+ );
68
+ return out;
69
+ }
70
+
42
71
  /**
43
72
  * Price normalized token slots once. Null means unknown, not a free request.
44
73
  * Rates are returned so a durable record keeps the price applied at ingestion.
@@ -68,6 +97,15 @@ export function priceUsage(args) {
68
97
  ...(args.serviceTier ? { serviceTier: args.serviceTier } : {}),
69
98
  ...(written1h ? { cacheWrite1hTokens: written1h } : {}),
70
99
  };
100
+ // A provider-billed figure for a media request (e.g. xAI cost_in_usd_ticks)
101
+ // is the price, with or without a catalog row.
102
+ const media = args.media || null;
103
+ if (typeof media?.reportedCostUsd === 'number' && Number.isFinite(media.reportedCostUsd) && media.reportedCostUsd >= 0)
104
+ return {
105
+ input,
106
+ costUsd: Number(media.reportedCostUsd.toFixed(6)),
107
+ rates: { ...provenance, pricingSource: 'provider' },
108
+ };
71
109
  if (args.inputTokensKnown === false || !meta)
72
110
  return {
73
111
  input,
@@ -107,7 +145,11 @@ export function priceUsage(args) {
107
145
  // Fast mode bills 2x standard rates on every fast-capable Opus.
108
146
  if (anthropicFast) multiplier *= 2;
109
147
  const keys = PRICING_RATE_KEYS;
110
- const tokens = [input, n(args.outputTokens), cached, written - written1h];
148
+ // Image/video output tokens bill above the text output rate; the row still
149
+ // carries the full output count.
150
+ const imageOut = media ? Math.min(n(media.outputImageTokens), n(args.outputTokens)) : 0;
151
+ const videoOut = media ? Math.min(n(media.outputVideoTokens), n(args.outputTokens) - imageOut) : 0;
152
+ const tokens = [input, n(args.outputTokens) - imageOut - videoOut, cached, written - written1h];
111
153
  const tierRates = ratesForPrompt(rateMeta, promptTokens);
112
154
  const rates = {
113
155
  ...provenance,
@@ -116,6 +158,10 @@ export function priceUsage(args) {
116
158
  if (written1h) rates.cacheWrite1hCostPerM = rates.inputCostPerM === null ? null : rates.inputCostPerM * 2;
117
159
  const missingRates = keys.filter((key, i) => tokens[i] > 0 && rates[key] === null);
118
160
  if (written1h && rates.cacheWrite1hCostPerM === null) missingRates.push('cacheWrite1hCostPerM');
161
+ const charge = media ? mediaCharges(meta, media, imageOut, videoOut) : { usd: 0, charged: false, missing: [] };
162
+ missingRates.push(...charge.missing);
163
+ if (media && !charge.charged && tokens.every((amount) => amount === 0) && !written1h)
164
+ return { input, costUsd: null, rates: { ...rates, unpricedReason: 'usage-not-reported' } };
119
165
  if (missingRates.length) {
120
166
  rates.unpricedReason = 'missing-rate';
121
167
  rates.missingRates = missingRates;
@@ -124,8 +170,9 @@ export function priceUsage(args) {
124
170
  const costUsd =
125
171
  (tokens.reduce((sum, amount, i) => sum + amount * (rates[keys[i]] ?? 0), 0) +
126
172
  written1h * (rates.cacheWrite1hCostPerM ?? 0)) /
127
- 1_000_000;
128
- return { input, costUsd: Number(costUsd.toFixed(6)), rates };
173
+ 1_000_000 +
174
+ charge.usd;
175
+ return { input, costUsd: Number(costUsd.toFixed(6)), rates: { ...rates, ...charge.rates } };
129
176
  }
130
177
 
131
178
  /**
@@ -124,6 +124,10 @@ const LITELLM_NUMBER_FIELDS = [
124
124
  'output_cost_per_token',
125
125
  'cache_read_input_token_cost',
126
126
  'cache_creation_input_token_cost',
127
+ 'output_cost_per_image',
128
+ 'output_cost_per_image_token',
129
+ 'output_cost_per_video_token',
130
+ 'output_cost_per_second',
127
131
  ];
128
132
  // _normalize tests each of these with `=== true`, so only a true value carries
129
133
  // information; anything else is indistinguishable from absent.
@@ -148,9 +152,10 @@ function projectLitellmRow(row) {
148
152
  // litellmPricing), not just the base price.
149
153
  for (const [field, value] of Object.entries(row)) {
150
154
  if (
151
- /^(?:input_cost_per_token|output_cost_per_token|cache_read_input_token_cost|cache_creation_input_token_cost)(?:_above_\d+k_tokens)?(?:_priority)?$/.test(
155
+ (/^(?:input_cost_per_token|output_cost_per_token|cache_read_input_token_cost|cache_creation_input_token_cost)(?:_above_\d+k_tokens)?(?:_priority)?$/.test(
152
156
  field
153
- ) &&
157
+ ) ||
158
+ /^output_cost_per_second_\w+$/.test(field)) &&
154
159
  typeof value === 'number' &&
155
160
  Number.isFinite(value)
156
161
  )
@@ -26,7 +26,7 @@ import {
26
26
  cachedProviderModelListsSync,
27
27
  providerCachedModelsSync,
28
28
  } from './provider-catalog-cache.mjs';
29
- import { litellmPricing, modelsDevPricing, PRICING_RATE_KEYS } from './model-pricing-rates.mjs';
29
+ import { litellmMediaPricing, litellmPricing, modelsDevPricing, PRICING_RATE_KEYS } from './model-pricing-rates.mjs';
30
30
  // Both overlays are narrowed to their read surface before becoming resident;
31
31
  // the disk caches below still receive the full payload.
32
32
  import { projectLitellmCatalog, projectModelsDevCatalog } from './model-catalog-projection.mjs';
@@ -597,6 +597,7 @@ function _normalize(entry) {
597
597
  contextWindow: entry.max_input_tokens || entry.max_tokens || null,
598
598
  outputTokens: entry.max_output_tokens || null,
599
599
  ...litellmPricing(entry),
600
+ ...litellmMediaPricing(entry),
600
601
  ...(PRICING_RATE_KEYS.some((key) => fastPricing[key] != null) ? { fastPricing } : {}),
601
602
  ...(entry.off_peak_multiplier ? { offPeakMultiplier: entry.off_peak_multiplier } : {}),
602
603
  supportsVision: entry.supports_vision === true,
@@ -51,6 +51,27 @@ export function litellmPricing(entry, suffix = '') {
51
51
  };
52
52
  }
53
53
 
54
+ /**
55
+ * Non-token media rates published by LiteLLM: USD per generated image, USD per
56
+ * generated video second (optionally per resolution, `output_cost_per_second_<res>`),
57
+ * and USD/M for image / video output tokens, which bill above the text output rate.
58
+ */
59
+ export function litellmMediaPricing(entry) {
60
+ const perM = (value) => (validRate(value) ? value * 1_000_000 : null);
61
+ const byResolution = {};
62
+ for (const [key, value] of Object.entries(entry || {})) {
63
+ const match = key.match(/^output_cost_per_second_(.+)$/);
64
+ if (match && validRate(value)) byResolution[match[1].toLowerCase()] = value;
65
+ }
66
+ return {
67
+ outputImageCostPerM: perM(entry?.output_cost_per_image_token),
68
+ outputVideoCostPerM: perM(entry?.output_cost_per_video_token),
69
+ outputCostPerImage: validRate(entry?.output_cost_per_image) ? entry.output_cost_per_image : null,
70
+ outputCostPerSecond: validRate(entry?.output_cost_per_second) ? entry.output_cost_per_second : null,
71
+ outputCostPerSecondByResolution: byResolution,
72
+ };
73
+ }
74
+
54
75
  export function modelsDevPricing(cost) {
55
76
  // Structured tiers supersede the older context_over_200k compatibility
56
77
  // field; it can coexist with a tier whose actual boundary is not 200k.
@@ -4,6 +4,13 @@ import { withUsageContext } from './usage-context.mjs';
4
4
  import { ACCOUNT_PROVIDERS } from '../provider-accounts.mjs';
5
5
  import { currentProviderAccountId } from '../provider-auth-binding.mjs';
6
6
 
7
+ // Results and errors already written. A provider-local re-send (model
8
+ // fallback, catalog retry) runs its own accounting and its result or error
9
+ // then returns through the enclosing send, which must not write it again.
10
+ const recorded = new WeakSet();
11
+ const hasTokens = (usage) =>
12
+ ['inputTokens', 'outputTokens', 'cachedTokens', 'cacheWriteTokens'].some((key) => Number(usage?.[key]) > 0);
13
+
7
14
  /**
8
15
  * Runs at the common provider boundary, not inside optional diagnostic IO.
9
16
  * Provider-local retries remain owned by the provider. Accounting failure must
@@ -12,7 +19,9 @@ import { currentProviderAccountId } from '../provider-auth-binding.mjs';
12
19
  export async function accountProviderSend(provider, instance, send, model, opts = {}) {
13
20
  const requestId = randomUUID();
14
21
  const startedAt = Date.now();
15
- const sessionId = opts.sessionId || opts.session?.id;
22
+ // usageSessionId: a request isolated under its own provider session id
23
+ // (compaction's `:compact`) whose spend belongs to the source session.
24
+ const sessionId = opts.usageSessionId || opts.sessionId || opts.session?.id;
16
25
  const sourceType = opts.session?.sourceType || opts.sourceType || opts.requestKind || '';
17
26
  const inputTokensInclusive = instance.constructor?.inputExcludesCache !== true;
18
27
  let ledger;
@@ -30,16 +39,18 @@ export async function accountProviderSend(provider, instance, send, model, opts
30
39
  sourceType,
31
40
  inputTokensInclusive,
32
41
  };
33
- const record = async (result) => {
42
+ const record = async (result, id, owner) => {
34
43
  if (!result?.usage) return;
35
44
  // A nested send (e.g. a fallback model re-send) already stamped its own
36
45
  // final attempt's tier; the outer context only saw the abandoned attempt.
37
46
  result.requestServiceTier ??= identity.requestServiceTier || '';
47
+ if (recorded.has(owner)) return;
38
48
  if (openingError) throw openingError;
39
49
  if (!ledger) return;
50
+ recorded.add(owner);
40
51
  const usage = result.usage;
41
52
  const row = makeUsageRecord({
42
- id: result.responseId ? undefined : requestId,
53
+ id: result.responseId ? undefined : id,
43
54
  ts: Date.now(),
44
55
  provider,
45
56
  model: result.model || model,
@@ -68,23 +79,32 @@ export async function accountProviderSend(provider, instance, send, model, opts
68
79
  // SQLite write no longer runs on the event loop.
69
80
  await ledger.recordQueued(row);
70
81
  };
71
- const save = async (result) => {
82
+ const save = async (result, id = requestId, owner = result) => {
72
83
  try {
73
- await record(result);
84
+ await record(result, id, owner);
74
85
  } catch (error) {
75
86
  result.usageAccountingError = String(error?.message || error);
76
87
  process.stderr.write(`[usage-ledger] RECORD NOT SAVED: ${result.usageAccountingError}\n`);
77
88
  }
78
89
  };
90
+ const saveAbandoned = async () => {
91
+ for (const [index, attempt] of (identity.abandonedUsage || []).entries()) {
92
+ if (hasTokens(attempt.usage)) await save(attempt, `${requestId}:abandoned:${index}`);
93
+ }
94
+ };
79
95
  let result;
80
96
  try {
81
97
  result = await withUsageContext(identity, send);
82
98
  } catch (error) {
99
+ await saveAbandoned();
83
100
  // Only provider-reported partial usage is recordable; never invent
84
101
  // tokens for a failed request or reinterpret an error as a success.
85
102
  if (error?.usage) await save(error);
103
+ else if (hasTokens(error?.partialUsage))
104
+ await save({ usage: error.partialUsage, model: error.partialModel }, requestId, error);
86
105
  throw error;
87
106
  }
107
+ await saveAbandoned();
88
108
  await save(result);
89
109
  return result;
90
110
  }
@@ -11,3 +11,9 @@ export const noteRequestServiceTier = (tier) => {
11
11
  const identity = context.getStore();
12
12
  if (identity) identity.requestServiceTier = tier || '';
13
13
  };
14
+ // A provider-local retry abandons an attempt the provider still billed; the
15
+ // enclosing send records it alongside its own final usage.
16
+ export const noteAbandonedUsage = (usage, model) => {
17
+ const identity = context.getStore();
18
+ if (identity && usage) (identity.abandonedUsage ||= []).push({ usage, model });
19
+ };
@@ -43,6 +43,7 @@ export function importTraceRow(row) {
43
43
  uncachedInputTokens: raw ? (row.uncached_input_tokens ?? payload.uncached_input_tokens) : undefined,
44
44
  cacheReadTokens: raw ? row.cached_tokens : row.cacheReadTokens,
45
45
  cacheWriteTokens: raw ? row.cache_write_tokens : row.cacheWriteTokens,
46
+ cacheWrite1hTokens: raw ? payload.raw_usage?.cache_creation?.ephemeral_1h_input_tokens : undefined,
46
47
  costUsd: raw ? undefined : row.costUsd,
47
48
  sessionId: row.session_id || row.sessionId,
48
49
  sourceType: row.sourceType || row.source_type || payload.sourceType || payload.source_type,
@@ -27,7 +27,15 @@ export function createFeatureToolHandlers({ rt, setupTool, officeToolsEnabled, m
27
27
  media: async (args, { callerCtx, callerCwd }) => {
28
28
  requireEnabled(callerCtx, mediaToolEnabled, 'media');
29
29
  const { executeMediaTool } = await import('../../runtime/media/tool.mjs');
30
- return await executeMediaTool(args, { cwd: callerCwd, signal: signalFor(callerCtx) });
30
+ const { getSession } = await import('../../runtime/agent/orchestrator/session/manager/session-crud.mjs');
31
+ // Generation spend is attributed to the calling session in the usage ledger.
32
+ const sessionId = sessionIdFor(callerCtx) || '';
33
+ return await executeMediaTool(args, {
34
+ cwd: callerCwd,
35
+ signal: signalFor(callerCtx),
36
+ sessionId,
37
+ sourceType: (sessionId && getSession(sessionId)?.sourceType) || '',
38
+ });
31
39
  },
32
40
  tidy: async (args, { callerCtx, callerCwd }) => {
33
41
  requireEnabled(callerCtx, tidyToolEnabled, 'tidy');
@@ -51,6 +51,38 @@ function recordOps(args, run) {
51
51
  });
52
52
  }
53
53
 
54
+ /**
55
+ * Collapse a sequence of op sets into one op set per argument slot with the
56
+ * same sequential outcome: every key ever deleted is deleted (so a foreign
57
+ * writer's key is still removed), then each surviving key is set once with its
58
+ * final value, ordered as sequential insertion would leave it (an overwrite
59
+ * keeps its position, a delete/reinsert moves to the end). Set adds are unioned.
60
+ */
61
+ function compactOps(batch) {
62
+ const slots = [];
63
+ for (const ops of batch) {
64
+ ops.forEach((op, index) => {
65
+ if (!op) return;
66
+ let slot = slots[index];
67
+ if (!slot) {
68
+ slot = slots[index] = op.add ? { add: new Set() } : { del: new Set(), final: new Map() };
69
+ }
70
+ if (slot.add) {
71
+ for (const value of op.add) slot.add.add(value);
72
+ return;
73
+ }
74
+ for (const key of op.del) {
75
+ slot.del.add(key);
76
+ slot.final.delete(key);
77
+ }
78
+ for (const [key, json] of op.set) slot.final.set(key, json);
79
+ });
80
+ }
81
+ return slots.map((slot) =>
82
+ !slot ? null : slot.add ? { add: slot.add } : { del: slot.del, set: slot.final }
83
+ );
84
+ }
85
+
54
86
  function replayOps(ops, args) {
55
87
  args.forEach((arg, index) => {
56
88
  const op = ops[index];
@@ -64,6 +96,10 @@ function replayOps(ops, args) {
64
96
  });
65
97
  }
66
98
 
99
+ function replayBatch(batch, args) {
100
+ replayOps(compactOps(batch), args);
101
+ }
102
+
67
103
  /**
68
104
  * @param {object} options
69
105
  * @param {string|null} options.file index file; null → every write is a no-op
@@ -99,9 +135,7 @@ export function createIndexWriteQueue({ file, rewrite, readDoc, onPersisted = ()
99
135
  await updateJsonAtomic(
100
136
  file,
101
137
  (cur) =>
102
- rewrite(cur, (...args) => {
103
- for (const ops of batch) replayOps(ops, args);
104
- }),
138
+ rewrite(cur, (...args) => replayBatch(batch, args)),
105
139
  { lock: true }
106
140
  );
107
141
  return;
@@ -153,9 +187,7 @@ export function createIndexWriteQueue({ file, rewrite, readDoc, onPersisted = ()
153
187
  updateJsonAtomicSync(
154
188
  file,
155
189
  (cur) =>
156
- rewrite(cur, (...args) => {
157
- for (const set of ops) replayOps(set, args);
158
- }),
190
+ rewrite(cur, (...args) => replayBatch(ops, args)),
159
191
  { lock: true, timeoutMs: EXIT_LOCK_TIMEOUT_MS }
160
192
  );
161
193
  } catch {
@@ -101,9 +101,15 @@ export function createWorkerRowStore(file) {
101
101
  return rows;
102
102
  }
103
103
 
104
+ // Rows and tombstones must come from the same read/projection. In
105
+ // particular, do not stat and reload the file between the two collections.
106
+ function readSnapshot() {
107
+ const rows = readAll();
108
+ return { rows, tombstones: (projectedView() || cache)?.tombstones || [] };
109
+ }
110
+
104
111
  function readTombstones() {
105
- readAll();
106
- return (projectedView() || cache)?.tombstones || [];
112
+ return readSnapshot().tombstones;
107
113
  }
108
114
 
109
115
  // Single writer path: the mutator runs now over keyed maps against the
@@ -113,6 +119,7 @@ export function createWorkerRowStore(file) {
113
119
 
114
120
  return {
115
121
  readAll,
122
+ readSnapshot,
116
123
  readTombstones,
117
124
  write,
118
125
  /** Resolves once every write so far is on disk. */
@@ -27,16 +27,19 @@ export function createWorkerIndex({ dataDir, cfgMod, mgr, tags, tagAgents, tagCw
27
27
  const activeWorkerKeys = new Set();
28
28
 
29
29
  function readWorkerRows(context = {}) {
30
- const rows = store.readAll();
30
+ const { rows, tombstones: savedTombstones } = store.readSnapshot();
31
31
  if (rows.length === 0) return rows;
32
- const tombstones = new Map(readAllTagTombstones().map((row) => [tagTombstoneKey(row), row]));
32
+ const tombstones = new Map(recoverTombstoneOwners(savedTombstones).map((row) => [tagTombstoneKey(row), row]));
33
33
  return rows.filter(
34
34
  (row) => rowMatchesContext(row, context) && !tombstoneBlocksWork(row, findTagTombstone(row, tombstones))
35
35
  );
36
36
  }
37
37
 
38
38
  function readAllTagTombstones() {
39
- const rows = store.readTombstones();
39
+ return recoverTombstoneOwners(store.readTombstones());
40
+ }
41
+
42
+ function recoverTombstoneOwners(rows) {
40
43
  if (!rows.some((row) => !clean(row.parentSessionId || row.ownerSessionId))) return rows;
41
44
  const sessions = typeof mgr.listSessions === 'function' ? mgr.listSessions({ includeClosed: true }) : [];
42
45
  // Resolve legacy ownership against all sessions, never just the caller's