mixdog 1.0.2 → 1.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -0
- package/package.json +1 -1
- package/src/defaults/skills/browser-use/SKILL.md +14 -6
- package/src/defaults/skills/goal-management/SKILL.md +2 -2
- package/src/runtime/agent/orchestrator/providers/anthropic-midstream-recovery.mjs +10 -1
- package/src/runtime/agent/orchestrator/runtime-core/goal-tool-defs.mjs +1 -1
- package/src/runtime/agent/orchestrator/session/cache/scoped-cache.mjs +40 -99
- package/src/runtime/agent/orchestrator/session/compact/runner.mjs +6 -1
- package/src/runtime/agent/orchestrator/session/loop/fresh-context.mjs +2 -0
- package/src/runtime/agent/orchestrator/session/loop/tool-exec/before-hook.mjs +7 -3
- package/src/runtime/agent/orchestrator/tools/builtin/cache-layers.mjs +18 -38
- package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-path-fanout.mjs +59 -32
- package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-pattern-fanout.mjs +15 -10
- package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-request.mjs +1 -1
- package/src/runtime/agent/orchestrator/tools/builtin/lib/shared-native-scan.mjs +12 -0
- package/src/runtime/agent/orchestrator/tools/builtin/read-source-windows.mjs +14 -0
- package/src/runtime/agent/orchestrator/tools/graph-manifest.json +13 -13
- package/src/runtime/agent/orchestrator/tools/patch/apply-patch/codex-batch.mjs +3 -1
- package/src/runtime/agent/orchestrator/tools/patch/dispatch.mjs +17 -9
- package/src/runtime/agent/orchestrator/tools/patch/orchestrator.mjs +1 -0
- package/src/runtime/agent/orchestrator/tools/patch/wave.mjs +4 -0
- package/src/runtime/browser-bridge/input-fields.mjs +1 -1
- package/src/runtime/browser-bridge/tool-defs.mjs +1 -1
- package/src/runtime/computer-bridge/actions.mjs +0 -13
- package/src/runtime/media/adapters/antigravity-image.mjs +7 -3
- package/src/runtime/media/adapters/codex-image.mjs +12 -1
- package/src/runtime/media/adapters/gemini-image.mjs +17 -6
- package/src/runtime/media/adapters/gemini-video.mjs +19 -9
- package/src/runtime/media/adapters/xai-media.mjs +20 -7
- package/src/runtime/media/jobs.mjs +35 -3
- package/src/runtime/media/media-usage.mjs +126 -0
- package/src/runtime/media/tool.mjs +16 -4
- package/src/runtime/shared/llm/cost.mjs +50 -3
- package/src/runtime/shared/llm/model-catalog-projection.mjs +7 -2
- package/src/runtime/shared/llm/model-catalog.mjs +2 -1
- package/src/runtime/shared/llm/model-pricing-rates.mjs +21 -0
- package/src/runtime/shared/llm/usage-accounting.mjs +25 -5
- package/src/runtime/shared/llm/usage-context.mjs +6 -0
- package/src/runtime/shared/llm/usage-ledger-import.mjs +1 -0
- package/src/session-runtime/internal-tool-executor/feature-tools.mjs +9 -1
- package/src/session-runtime/services/agent-tool/index-write-queue.mjs +38 -6
- package/src/session-runtime/services/agent-tool/worker-index/row-store.mjs +9 -2
- package/src/session-runtime/services/agent-tool/worker-index.mjs +6 -3
- package/src/session-runtime/services/hook-bus/event-runner.mjs +10 -2
- package/src/session-runtime/services/hook-bus/tool-gate.mjs +5 -2
- package/src/tui/dist/index.mjs +7 -3
|
@@ -11,6 +11,7 @@ import { MAX_GENERATED_MEDIA_BYTES } from './download.mjs';
|
|
|
11
11
|
import { mediaError, resolveMediaRequest } from './lanes.mjs';
|
|
12
12
|
import { saveMediaAsset } from './store.mjs';
|
|
13
13
|
import { setMediaDefault } from './defaults.mjs';
|
|
14
|
+
import { recordMediaUsage } from './media-usage.mjs';
|
|
14
15
|
|
|
15
16
|
const JOBS = new Map();
|
|
16
17
|
// Finished jobs stay readable for a while so a slow poller still sees the
|
|
@@ -77,7 +78,16 @@ async function runAdapter({ lane, kind, model, requestModel, prompt, options, re
|
|
|
77
78
|
* Validate + start one generation. Returns the initial snapshot immediately;
|
|
78
79
|
* the caller polls getMediaJob for progress and the finished asset id.
|
|
79
80
|
*/
|
|
80
|
-
export async function startMediaJob({
|
|
81
|
+
export async function startMediaJob({
|
|
82
|
+
lane: laneId,
|
|
83
|
+
kind,
|
|
84
|
+
model,
|
|
85
|
+
prompt,
|
|
86
|
+
options = {},
|
|
87
|
+
references = [],
|
|
88
|
+
sessionId = '',
|
|
89
|
+
sourceType = '',
|
|
90
|
+
} = {}) {
|
|
81
91
|
const text = String(prompt || '').trim();
|
|
82
92
|
if (!text) throw mediaError('prompt is required', 'MEDIA_PROMPT_REQUIRED');
|
|
83
93
|
if (text.length > MAX_PROMPT_CHARS) throw mediaError('prompt is too long', 'MEDIA_PROMPT_TOO_LONG');
|
|
@@ -128,13 +138,31 @@ export async function startMediaJob({ lane: laneId, kind, model, prompt, options
|
|
|
128
138
|
controller,
|
|
129
139
|
};
|
|
130
140
|
JOBS.set(job.id, job);
|
|
131
|
-
void runJob(job, resolved, {
|
|
141
|
+
void runJob(job, resolved, {
|
|
142
|
+
requestModel: modelEntry?.requestModel,
|
|
143
|
+
options,
|
|
144
|
+
references: refs,
|
|
145
|
+
sessionId,
|
|
146
|
+
sourceType,
|
|
147
|
+
});
|
|
132
148
|
return snapshot(job);
|
|
133
149
|
}
|
|
134
150
|
|
|
135
151
|
/** Drive one started job to a terminal state; never rejects. */
|
|
136
|
-
async function runJob(job, resolved, { requestModel, options, references }) {
|
|
152
|
+
async function runJob(job, resolved, { requestModel, options, references, sessionId, sourceType }) {
|
|
137
153
|
const { controller } = job;
|
|
154
|
+
// One ledger row per generation. The ChatGPT lane bills the orchestrator
|
|
155
|
+
// model that ran the hosted tool, not the "auto" image route.
|
|
156
|
+
const record = (usage) =>
|
|
157
|
+
recordMediaUsage({
|
|
158
|
+
lane: job.lane,
|
|
159
|
+
model: job.model,
|
|
160
|
+
pricingModel: job.lane === 'openai-oauth' ? requestModel : undefined,
|
|
161
|
+
usage,
|
|
162
|
+
sessionId,
|
|
163
|
+
sourceType,
|
|
164
|
+
durationMs: Date.now() - job.startedAt,
|
|
165
|
+
});
|
|
138
166
|
try {
|
|
139
167
|
const result = await runAdapter({
|
|
140
168
|
lane: resolved.lane,
|
|
@@ -154,6 +182,8 @@ async function runJob(job, resolved, { requestModel, options, references }) {
|
|
|
154
182
|
if (next > job.progress) job.progress = next;
|
|
155
183
|
},
|
|
156
184
|
});
|
|
185
|
+
// Billed once the provider returned a result, whatever happens to the bytes next.
|
|
186
|
+
await record(result?.usage);
|
|
157
187
|
if (!Buffer.isBuffer(result?.bytes) || !result.bytes.length || result.bytes.length > MAX_GENERATED_MEDIA_BYTES) {
|
|
158
188
|
throw mediaError('generated media exceeds the media size limit', 'MEDIA_RESULT_TOO_LARGE', 502);
|
|
159
189
|
}
|
|
@@ -175,6 +205,8 @@ async function runJob(job, resolved, { requestModel, options, references }) {
|
|
|
175
205
|
job.status = 'done';
|
|
176
206
|
} catch (err) {
|
|
177
207
|
const canceled = controller.signal.aborted || err?.code === 'MEDIA_CANCELED' || err?.name === 'AbortError';
|
|
208
|
+
// A failure is recorded only when the provider reported usage for it.
|
|
209
|
+
if (err?.usage) await record(err.usage);
|
|
178
210
|
job.status = canceled ? 'canceled' : 'failed';
|
|
179
211
|
job.error = canceled ? 'canceled' : String(err?.message || err).slice(0, 500);
|
|
180
212
|
job.errorCode = err?.code || null;
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Usage ledger rows for media generations.
|
|
3
|
+
*
|
|
4
|
+
* Adapters return (or attach to their error) a normalized `usage` object read
|
|
5
|
+
* from what the upstream response actually reported:
|
|
6
|
+
* { inputTokens, outputTokens, cachedTokens, outputImageTokens, outputVideoTokens,
|
|
7
|
+
* costUsd, images, seconds, resolution }
|
|
8
|
+
* `jobs.mjs` records exactly one row per generation through recordMediaUsage.
|
|
9
|
+
* Nothing here estimates tokens: a field the provider did not report stays absent.
|
|
10
|
+
*/
|
|
11
|
+
import { getUsageLedger, makeUsageRecord } from '../shared/llm/usage-ledger.mjs';
|
|
12
|
+
import { ACCOUNT_PROVIDERS } from '../shared/provider-accounts.mjs';
|
|
13
|
+
import { currentProviderAccountId } from '../shared/provider-auth-binding.mjs';
|
|
14
|
+
|
|
15
|
+
const count = (value) => (Number.isFinite(Number(value)) && Number(value) > 0 ? Math.trunc(Number(value)) : 0);
|
|
16
|
+
const pick = (source, ...keys) => {
|
|
17
|
+
for (const key of keys) if (source?.[key] != null) return source[key];
|
|
18
|
+
return undefined;
|
|
19
|
+
};
|
|
20
|
+
|
|
21
|
+
const TICKS_PER_USD = 1e10;
|
|
22
|
+
|
|
23
|
+
/** Token usage when the provider reported any; null otherwise. */
|
|
24
|
+
function tokenUsage(fields) {
|
|
25
|
+
const usage = Object.fromEntries(Object.entries(fields).filter(([, value]) => value > 0));
|
|
26
|
+
return Object.keys(usage).length ? usage : null;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** Gemini `generateContent` / Antigravity `usageMetadata` (camelCase or snake_case). */
|
|
30
|
+
export function geminiUsage(meta) {
|
|
31
|
+
if (!meta || typeof meta !== 'object') return null;
|
|
32
|
+
const details = pick(meta, 'candidatesTokensDetails', 'candidates_tokens_details');
|
|
33
|
+
const imageTokens = (Array.isArray(details) ? details : [])
|
|
34
|
+
.filter((entry) => String(entry?.modality || '').toUpperCase() === 'IMAGE')
|
|
35
|
+
.reduce((sum, entry) => sum + count(pick(entry, 'tokenCount', 'token_count')), 0);
|
|
36
|
+
return tokenUsage({
|
|
37
|
+
inputTokens: count(pick(meta, 'promptTokenCount', 'prompt_token_count')),
|
|
38
|
+
cachedTokens: count(pick(meta, 'cachedContentTokenCount', 'cached_content_token_count')),
|
|
39
|
+
outputTokens:
|
|
40
|
+
count(pick(meta, 'candidatesTokenCount', 'candidates_token_count')) +
|
|
41
|
+
count(pick(meta, 'thoughtsTokenCount', 'thoughts_token_count')),
|
|
42
|
+
outputImageTokens: imageTokens,
|
|
43
|
+
});
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** Gemini Interactions API `usage` (omni video). */
|
|
47
|
+
export function interactionsUsage(usage) {
|
|
48
|
+
if (!usage || typeof usage !== 'object') return null;
|
|
49
|
+
const videoTokens = (Array.isArray(usage.output_tokens_by_modality) ? usage.output_tokens_by_modality : [])
|
|
50
|
+
.filter((entry) => String(entry?.modality || '').toLowerCase() === 'video')
|
|
51
|
+
.reduce((sum, entry) => sum + count(entry?.tokens), 0);
|
|
52
|
+
return tokenUsage({
|
|
53
|
+
inputTokens: count(usage.total_input_tokens),
|
|
54
|
+
cachedTokens: count(usage.total_cached_tokens),
|
|
55
|
+
outputTokens: count(usage.total_output_tokens) + count(usage.total_thought_tokens),
|
|
56
|
+
outputVideoTokens: videoTokens,
|
|
57
|
+
});
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/** OpenAI / Codex Responses `usage` from response.completed. */
|
|
61
|
+
export function responsesUsage(usage) {
|
|
62
|
+
if (!usage || typeof usage !== 'object') return null;
|
|
63
|
+
return tokenUsage({
|
|
64
|
+
inputTokens: count(usage.input_tokens),
|
|
65
|
+
cachedTokens: count(usage.input_tokens_details?.cached_tokens),
|
|
66
|
+
outputTokens: count(usage.output_tokens),
|
|
67
|
+
});
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** xAI image/video `usage.cost_in_usd_ticks` (1 USD = 1e10 ticks); the billed cost. */
|
|
71
|
+
export function xaiUsage(usage) {
|
|
72
|
+
const ticks = usage?.cost_in_usd_ticks;
|
|
73
|
+
return typeof ticks === 'number' && Number.isFinite(ticks) && ticks >= 0 ? { costUsd: ticks / TICKS_PER_USD } : null;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** True when the provider reported something billable for the request. */
|
|
77
|
+
export function hasReportedUsage(usage) {
|
|
78
|
+
if (!usage) return false;
|
|
79
|
+
// Seconds are only attached once a video generation completed (billed per second).
|
|
80
|
+
return (
|
|
81
|
+
typeof usage.costUsd === 'number' ||
|
|
82
|
+
['inputTokens', 'outputTokens', 'cachedTokens', 'seconds'].some((k) => usage[k] > 0)
|
|
83
|
+
);
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** Attach provider-reported usage to a failure so jobs.mjs can still record it. */
|
|
87
|
+
export function withReportedUsage(error, usage) {
|
|
88
|
+
if (hasReportedUsage(usage)) error.usage = usage;
|
|
89
|
+
return error;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Append one ledger row for a generation. A ledger failure never fails the
|
|
94
|
+
* generation; it is logged the way usage-accounting does.
|
|
95
|
+
*/
|
|
96
|
+
export async function recordMediaUsage(args, getLedger = getUsageLedger) {
|
|
97
|
+
try {
|
|
98
|
+
const usage = args.usage || {};
|
|
99
|
+
const row = makeUsageRecord({
|
|
100
|
+
ts: Date.now(),
|
|
101
|
+
provider: args.lane,
|
|
102
|
+
model: args.model,
|
|
103
|
+
requestedModel: args.model,
|
|
104
|
+
pricingModel: args.pricingModel,
|
|
105
|
+
sessionId: args.sessionId,
|
|
106
|
+
sourceType: args.sourceType,
|
|
107
|
+
inputTokens: usage.inputTokens,
|
|
108
|
+
outputTokens: usage.outputTokens,
|
|
109
|
+
cacheReadTokens: usage.cachedTokens,
|
|
110
|
+
costUsd: usage.costUsd,
|
|
111
|
+
media: {
|
|
112
|
+
reportedCostUsd: usage.costUsd,
|
|
113
|
+
outputImageTokens: usage.outputImageTokens,
|
|
114
|
+
outputVideoTokens: usage.outputVideoTokens,
|
|
115
|
+
images: usage.images,
|
|
116
|
+
seconds: usage.seconds,
|
|
117
|
+
resolution: usage.resolution,
|
|
118
|
+
},
|
|
119
|
+
account: ACCOUNT_PROVIDERS.includes(args.lane) ? currentProviderAccountId(args.lane) : '',
|
|
120
|
+
durationMs: args.durationMs,
|
|
121
|
+
});
|
|
122
|
+
await getLedger()?.recordQueued(row);
|
|
123
|
+
} catch (error) {
|
|
124
|
+
process.stderr.write(`[usage-ledger] RECORD NOT SAVED: ${String(error?.message || error)}\n`);
|
|
125
|
+
}
|
|
126
|
+
}
|
|
@@ -280,7 +280,7 @@ function jobView(job) {
|
|
|
280
280
|
};
|
|
281
281
|
}
|
|
282
282
|
|
|
283
|
-
async function generate(args, { cwd, signal, deps }) {
|
|
283
|
+
async function generate(args, { cwd, signal, deps, sessionId, sourceType }) {
|
|
284
284
|
const kind = clean(args.kind);
|
|
285
285
|
if (!MEDIA_KINDS.includes(kind)) throw new MediaToolError(`generate requires kind: ${MEDIA_KINDS.join(' | ')}`);
|
|
286
286
|
const prompt = clean(args.prompt);
|
|
@@ -314,7 +314,16 @@ async function generate(args, { cwd, signal, deps }) {
|
|
|
314
314
|
quality: clean(args.quality),
|
|
315
315
|
});
|
|
316
316
|
const references = await readReferences(args.references, cwd, model.controls);
|
|
317
|
-
const started = await jobs.startMediaJob({
|
|
317
|
+
const started = await jobs.startMediaJob({
|
|
318
|
+
lane: lane.id,
|
|
319
|
+
kind,
|
|
320
|
+
model: modelId,
|
|
321
|
+
prompt,
|
|
322
|
+
options,
|
|
323
|
+
references,
|
|
324
|
+
sessionId,
|
|
325
|
+
sourceType,
|
|
326
|
+
});
|
|
318
327
|
const base = { lane: lane.id, model: modelId, laneSource, options, referenceCount: references.length, prompt };
|
|
319
328
|
if (args.wait === false) {
|
|
320
329
|
return {
|
|
@@ -371,7 +380,10 @@ async function cancel(args, { deps }) {
|
|
|
371
380
|
return { ok: true, ...jobView(jobs.getMediaJob(id)), canceled: canceled === true };
|
|
372
381
|
}
|
|
373
382
|
|
|
374
|
-
export async function executeMediaTool(
|
|
383
|
+
export async function executeMediaTool(
|
|
384
|
+
args = {},
|
|
385
|
+
{ cwd = process.cwd(), signal = null, deps = null, sessionId = '', sourceType = '' } = {}
|
|
386
|
+
) {
|
|
375
387
|
const action = clean(args.action).toLowerCase();
|
|
376
388
|
try {
|
|
377
389
|
if (!MEDIA_ACTIONS.includes(action))
|
|
@@ -388,7 +400,7 @@ export async function executeMediaTool(args = {}, { cwd = process.cwd(), signal
|
|
|
388
400
|
),
|
|
389
401
|
});
|
|
390
402
|
}
|
|
391
|
-
if (action === 'generate') return mediaToolResult(await generate(args, { cwd, signal, deps }));
|
|
403
|
+
if (action === 'generate') return mediaToolResult(await generate(args, { cwd, signal, deps, sessionId, sourceType }));
|
|
392
404
|
if (action === 'status') return mediaToolResult(await status(args, { cwd, deps }));
|
|
393
405
|
return mediaToolResult(await cancel(args, { deps }));
|
|
394
406
|
} catch (error) {
|
|
@@ -39,6 +39,35 @@ export function billableInputTokensForProvider(provider, inputTokens, cacheReadT
|
|
|
39
39
|
return Math.max(input - (Number(cacheReadTokens) || 0) - (Number(cacheWriteTokens) || 0), 0);
|
|
40
40
|
}
|
|
41
41
|
|
|
42
|
+
/**
|
|
43
|
+
* Media billed by something other than the text token slots: image/video output
|
|
44
|
+
* tokens, whole images, or video seconds (per resolution when the catalog lists
|
|
45
|
+
* one). A used unit without a catalog rate is reported missing, never free.
|
|
46
|
+
*/
|
|
47
|
+
function mediaCharges(meta, media, imageOut, videoOut) {
|
|
48
|
+
const n = (value) => (Number.isFinite(Number(value)) && Number(value) > 0 ? Number(value) : 0);
|
|
49
|
+
const out = { usd: 0, charged: false, missing: [], rates: {} };
|
|
50
|
+
const bill = (amount, key, rate, scale = 1) => {
|
|
51
|
+
if (amount <= 0) return;
|
|
52
|
+
out.charged = true;
|
|
53
|
+
if (rate == null) out.missing.push(key);
|
|
54
|
+
else {
|
|
55
|
+
out.usd += (amount * rate) / scale;
|
|
56
|
+
out.rates[key] = rate;
|
|
57
|
+
}
|
|
58
|
+
};
|
|
59
|
+
bill(imageOut, 'outputImageCostPerM', meta.outputImageCostPerM, 1_000_000);
|
|
60
|
+
bill(videoOut, 'outputVideoCostPerM', meta.outputVideoCostPerM, 1_000_000);
|
|
61
|
+
bill(n(media.images), 'outputCostPerImage', meta.outputCostPerImage);
|
|
62
|
+
const resolution = String(media.resolution || '').toLowerCase();
|
|
63
|
+
bill(
|
|
64
|
+
n(media.seconds),
|
|
65
|
+
'outputCostPerSecond',
|
|
66
|
+
meta.outputCostPerSecondByResolution?.[resolution] ?? meta.outputCostPerSecond
|
|
67
|
+
);
|
|
68
|
+
return out;
|
|
69
|
+
}
|
|
70
|
+
|
|
42
71
|
/**
|
|
43
72
|
* Price normalized token slots once. Null means unknown, not a free request.
|
|
44
73
|
* Rates are returned so a durable record keeps the price applied at ingestion.
|
|
@@ -68,6 +97,15 @@ export function priceUsage(args) {
|
|
|
68
97
|
...(args.serviceTier ? { serviceTier: args.serviceTier } : {}),
|
|
69
98
|
...(written1h ? { cacheWrite1hTokens: written1h } : {}),
|
|
70
99
|
};
|
|
100
|
+
// A provider-billed figure for a media request (e.g. xAI cost_in_usd_ticks)
|
|
101
|
+
// is the price, with or without a catalog row.
|
|
102
|
+
const media = args.media || null;
|
|
103
|
+
if (typeof media?.reportedCostUsd === 'number' && Number.isFinite(media.reportedCostUsd) && media.reportedCostUsd >= 0)
|
|
104
|
+
return {
|
|
105
|
+
input,
|
|
106
|
+
costUsd: Number(media.reportedCostUsd.toFixed(6)),
|
|
107
|
+
rates: { ...provenance, pricingSource: 'provider' },
|
|
108
|
+
};
|
|
71
109
|
if (args.inputTokensKnown === false || !meta)
|
|
72
110
|
return {
|
|
73
111
|
input,
|
|
@@ -107,7 +145,11 @@ export function priceUsage(args) {
|
|
|
107
145
|
// Fast mode bills 2x standard rates on every fast-capable Opus.
|
|
108
146
|
if (anthropicFast) multiplier *= 2;
|
|
109
147
|
const keys = PRICING_RATE_KEYS;
|
|
110
|
-
|
|
148
|
+
// Image/video output tokens bill above the text output rate; the row still
|
|
149
|
+
// carries the full output count.
|
|
150
|
+
const imageOut = media ? Math.min(n(media.outputImageTokens), n(args.outputTokens)) : 0;
|
|
151
|
+
const videoOut = media ? Math.min(n(media.outputVideoTokens), n(args.outputTokens) - imageOut) : 0;
|
|
152
|
+
const tokens = [input, n(args.outputTokens) - imageOut - videoOut, cached, written - written1h];
|
|
111
153
|
const tierRates = ratesForPrompt(rateMeta, promptTokens);
|
|
112
154
|
const rates = {
|
|
113
155
|
...provenance,
|
|
@@ -116,6 +158,10 @@ export function priceUsage(args) {
|
|
|
116
158
|
if (written1h) rates.cacheWrite1hCostPerM = rates.inputCostPerM === null ? null : rates.inputCostPerM * 2;
|
|
117
159
|
const missingRates = keys.filter((key, i) => tokens[i] > 0 && rates[key] === null);
|
|
118
160
|
if (written1h && rates.cacheWrite1hCostPerM === null) missingRates.push('cacheWrite1hCostPerM');
|
|
161
|
+
const charge = media ? mediaCharges(meta, media, imageOut, videoOut) : { usd: 0, charged: false, missing: [] };
|
|
162
|
+
missingRates.push(...charge.missing);
|
|
163
|
+
if (media && !charge.charged && tokens.every((amount) => amount === 0) && !written1h)
|
|
164
|
+
return { input, costUsd: null, rates: { ...rates, unpricedReason: 'usage-not-reported' } };
|
|
119
165
|
if (missingRates.length) {
|
|
120
166
|
rates.unpricedReason = 'missing-rate';
|
|
121
167
|
rates.missingRates = missingRates;
|
|
@@ -124,8 +170,9 @@ export function priceUsage(args) {
|
|
|
124
170
|
const costUsd =
|
|
125
171
|
(tokens.reduce((sum, amount, i) => sum + amount * (rates[keys[i]] ?? 0), 0) +
|
|
126
172
|
written1h * (rates.cacheWrite1hCostPerM ?? 0)) /
|
|
127
|
-
|
|
128
|
-
|
|
173
|
+
1_000_000 +
|
|
174
|
+
charge.usd;
|
|
175
|
+
return { input, costUsd: Number(costUsd.toFixed(6)), rates: { ...rates, ...charge.rates } };
|
|
129
176
|
}
|
|
130
177
|
|
|
131
178
|
/**
|
|
@@ -124,6 +124,10 @@ const LITELLM_NUMBER_FIELDS = [
|
|
|
124
124
|
'output_cost_per_token',
|
|
125
125
|
'cache_read_input_token_cost',
|
|
126
126
|
'cache_creation_input_token_cost',
|
|
127
|
+
'output_cost_per_image',
|
|
128
|
+
'output_cost_per_image_token',
|
|
129
|
+
'output_cost_per_video_token',
|
|
130
|
+
'output_cost_per_second',
|
|
127
131
|
];
|
|
128
132
|
// _normalize tests each of these with `=== true`, so only a true value carries
|
|
129
133
|
// information; anything else is indistinguishable from absent.
|
|
@@ -148,9 +152,10 @@ function projectLitellmRow(row) {
|
|
|
148
152
|
// litellmPricing), not just the base price.
|
|
149
153
|
for (const [field, value] of Object.entries(row)) {
|
|
150
154
|
if (
|
|
151
|
-
/^(?:input_cost_per_token|output_cost_per_token|cache_read_input_token_cost|cache_creation_input_token_cost)(?:_above_\d+k_tokens)?(?:_priority)?$/.test(
|
|
155
|
+
(/^(?:input_cost_per_token|output_cost_per_token|cache_read_input_token_cost|cache_creation_input_token_cost)(?:_above_\d+k_tokens)?(?:_priority)?$/.test(
|
|
152
156
|
field
|
|
153
|
-
)
|
|
157
|
+
) ||
|
|
158
|
+
/^output_cost_per_second_\w+$/.test(field)) &&
|
|
154
159
|
typeof value === 'number' &&
|
|
155
160
|
Number.isFinite(value)
|
|
156
161
|
)
|
|
@@ -26,7 +26,7 @@ import {
|
|
|
26
26
|
cachedProviderModelListsSync,
|
|
27
27
|
providerCachedModelsSync,
|
|
28
28
|
} from './provider-catalog-cache.mjs';
|
|
29
|
-
import { litellmPricing, modelsDevPricing, PRICING_RATE_KEYS } from './model-pricing-rates.mjs';
|
|
29
|
+
import { litellmMediaPricing, litellmPricing, modelsDevPricing, PRICING_RATE_KEYS } from './model-pricing-rates.mjs';
|
|
30
30
|
// Both overlays are narrowed to their read surface before becoming resident;
|
|
31
31
|
// the disk caches below still receive the full payload.
|
|
32
32
|
import { projectLitellmCatalog, projectModelsDevCatalog } from './model-catalog-projection.mjs';
|
|
@@ -597,6 +597,7 @@ function _normalize(entry) {
|
|
|
597
597
|
contextWindow: entry.max_input_tokens || entry.max_tokens || null,
|
|
598
598
|
outputTokens: entry.max_output_tokens || null,
|
|
599
599
|
...litellmPricing(entry),
|
|
600
|
+
...litellmMediaPricing(entry),
|
|
600
601
|
...(PRICING_RATE_KEYS.some((key) => fastPricing[key] != null) ? { fastPricing } : {}),
|
|
601
602
|
...(entry.off_peak_multiplier ? { offPeakMultiplier: entry.off_peak_multiplier } : {}),
|
|
602
603
|
supportsVision: entry.supports_vision === true,
|
|
@@ -51,6 +51,27 @@ export function litellmPricing(entry, suffix = '') {
|
|
|
51
51
|
};
|
|
52
52
|
}
|
|
53
53
|
|
|
54
|
+
/**
|
|
55
|
+
* Non-token media rates published by LiteLLM: USD per generated image, USD per
|
|
56
|
+
* generated video second (optionally per resolution, `output_cost_per_second_<res>`),
|
|
57
|
+
* and USD/M for image / video output tokens, which bill above the text output rate.
|
|
58
|
+
*/
|
|
59
|
+
export function litellmMediaPricing(entry) {
|
|
60
|
+
const perM = (value) => (validRate(value) ? value * 1_000_000 : null);
|
|
61
|
+
const byResolution = {};
|
|
62
|
+
for (const [key, value] of Object.entries(entry || {})) {
|
|
63
|
+
const match = key.match(/^output_cost_per_second_(.+)$/);
|
|
64
|
+
if (match && validRate(value)) byResolution[match[1].toLowerCase()] = value;
|
|
65
|
+
}
|
|
66
|
+
return {
|
|
67
|
+
outputImageCostPerM: perM(entry?.output_cost_per_image_token),
|
|
68
|
+
outputVideoCostPerM: perM(entry?.output_cost_per_video_token),
|
|
69
|
+
outputCostPerImage: validRate(entry?.output_cost_per_image) ? entry.output_cost_per_image : null,
|
|
70
|
+
outputCostPerSecond: validRate(entry?.output_cost_per_second) ? entry.output_cost_per_second : null,
|
|
71
|
+
outputCostPerSecondByResolution: byResolution,
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
|
|
54
75
|
export function modelsDevPricing(cost) {
|
|
55
76
|
// Structured tiers supersede the older context_over_200k compatibility
|
|
56
77
|
// field; it can coexist with a tier whose actual boundary is not 200k.
|
|
@@ -4,6 +4,13 @@ import { withUsageContext } from './usage-context.mjs';
|
|
|
4
4
|
import { ACCOUNT_PROVIDERS } from '../provider-accounts.mjs';
|
|
5
5
|
import { currentProviderAccountId } from '../provider-auth-binding.mjs';
|
|
6
6
|
|
|
7
|
+
// Results and errors already written. A provider-local re-send (model
|
|
8
|
+
// fallback, catalog retry) runs its own accounting and its result or error
|
|
9
|
+
// then returns through the enclosing send, which must not write it again.
|
|
10
|
+
const recorded = new WeakSet();
|
|
11
|
+
const hasTokens = (usage) =>
|
|
12
|
+
['inputTokens', 'outputTokens', 'cachedTokens', 'cacheWriteTokens'].some((key) => Number(usage?.[key]) > 0);
|
|
13
|
+
|
|
7
14
|
/**
|
|
8
15
|
* Runs at the common provider boundary, not inside optional diagnostic IO.
|
|
9
16
|
* Provider-local retries remain owned by the provider. Accounting failure must
|
|
@@ -12,7 +19,9 @@ import { currentProviderAccountId } from '../provider-auth-binding.mjs';
|
|
|
12
19
|
export async function accountProviderSend(provider, instance, send, model, opts = {}) {
|
|
13
20
|
const requestId = randomUUID();
|
|
14
21
|
const startedAt = Date.now();
|
|
15
|
-
|
|
22
|
+
// usageSessionId: a request isolated under its own provider session id
|
|
23
|
+
// (compaction's `:compact`) whose spend belongs to the source session.
|
|
24
|
+
const sessionId = opts.usageSessionId || opts.sessionId || opts.session?.id;
|
|
16
25
|
const sourceType = opts.session?.sourceType || opts.sourceType || opts.requestKind || '';
|
|
17
26
|
const inputTokensInclusive = instance.constructor?.inputExcludesCache !== true;
|
|
18
27
|
let ledger;
|
|
@@ -30,16 +39,18 @@ export async function accountProviderSend(provider, instance, send, model, opts
|
|
|
30
39
|
sourceType,
|
|
31
40
|
inputTokensInclusive,
|
|
32
41
|
};
|
|
33
|
-
const record = async (result) => {
|
|
42
|
+
const record = async (result, id, owner) => {
|
|
34
43
|
if (!result?.usage) return;
|
|
35
44
|
// A nested send (e.g. a fallback model re-send) already stamped its own
|
|
36
45
|
// final attempt's tier; the outer context only saw the abandoned attempt.
|
|
37
46
|
result.requestServiceTier ??= identity.requestServiceTier || '';
|
|
47
|
+
if (recorded.has(owner)) return;
|
|
38
48
|
if (openingError) throw openingError;
|
|
39
49
|
if (!ledger) return;
|
|
50
|
+
recorded.add(owner);
|
|
40
51
|
const usage = result.usage;
|
|
41
52
|
const row = makeUsageRecord({
|
|
42
|
-
id: result.responseId ? undefined :
|
|
53
|
+
id: result.responseId ? undefined : id,
|
|
43
54
|
ts: Date.now(),
|
|
44
55
|
provider,
|
|
45
56
|
model: result.model || model,
|
|
@@ -68,23 +79,32 @@ export async function accountProviderSend(provider, instance, send, model, opts
|
|
|
68
79
|
// SQLite write no longer runs on the event loop.
|
|
69
80
|
await ledger.recordQueued(row);
|
|
70
81
|
};
|
|
71
|
-
const save = async (result) => {
|
|
82
|
+
const save = async (result, id = requestId, owner = result) => {
|
|
72
83
|
try {
|
|
73
|
-
await record(result);
|
|
84
|
+
await record(result, id, owner);
|
|
74
85
|
} catch (error) {
|
|
75
86
|
result.usageAccountingError = String(error?.message || error);
|
|
76
87
|
process.stderr.write(`[usage-ledger] RECORD NOT SAVED: ${result.usageAccountingError}\n`);
|
|
77
88
|
}
|
|
78
89
|
};
|
|
90
|
+
const saveAbandoned = async () => {
|
|
91
|
+
for (const [index, attempt] of (identity.abandonedUsage || []).entries()) {
|
|
92
|
+
if (hasTokens(attempt.usage)) await save(attempt, `${requestId}:abandoned:${index}`);
|
|
93
|
+
}
|
|
94
|
+
};
|
|
79
95
|
let result;
|
|
80
96
|
try {
|
|
81
97
|
result = await withUsageContext(identity, send);
|
|
82
98
|
} catch (error) {
|
|
99
|
+
await saveAbandoned();
|
|
83
100
|
// Only provider-reported partial usage is recordable; never invent
|
|
84
101
|
// tokens for a failed request or reinterpret an error as a success.
|
|
85
102
|
if (error?.usage) await save(error);
|
|
103
|
+
else if (hasTokens(error?.partialUsage))
|
|
104
|
+
await save({ usage: error.partialUsage, model: error.partialModel }, requestId, error);
|
|
86
105
|
throw error;
|
|
87
106
|
}
|
|
107
|
+
await saveAbandoned();
|
|
88
108
|
await save(result);
|
|
89
109
|
return result;
|
|
90
110
|
}
|
|
@@ -11,3 +11,9 @@ export const noteRequestServiceTier = (tier) => {
|
|
|
11
11
|
const identity = context.getStore();
|
|
12
12
|
if (identity) identity.requestServiceTier = tier || '';
|
|
13
13
|
};
|
|
14
|
+
// A provider-local retry abandons an attempt the provider still billed; the
|
|
15
|
+
// enclosing send records it alongside its own final usage.
|
|
16
|
+
export const noteAbandonedUsage = (usage, model) => {
|
|
17
|
+
const identity = context.getStore();
|
|
18
|
+
if (identity && usage) (identity.abandonedUsage ||= []).push({ usage, model });
|
|
19
|
+
};
|
|
@@ -43,6 +43,7 @@ export function importTraceRow(row) {
|
|
|
43
43
|
uncachedInputTokens: raw ? (row.uncached_input_tokens ?? payload.uncached_input_tokens) : undefined,
|
|
44
44
|
cacheReadTokens: raw ? row.cached_tokens : row.cacheReadTokens,
|
|
45
45
|
cacheWriteTokens: raw ? row.cache_write_tokens : row.cacheWriteTokens,
|
|
46
|
+
cacheWrite1hTokens: raw ? payload.raw_usage?.cache_creation?.ephemeral_1h_input_tokens : undefined,
|
|
46
47
|
costUsd: raw ? undefined : row.costUsd,
|
|
47
48
|
sessionId: row.session_id || row.sessionId,
|
|
48
49
|
sourceType: row.sourceType || row.source_type || payload.sourceType || payload.source_type,
|
|
@@ -27,7 +27,15 @@ export function createFeatureToolHandlers({ rt, setupTool, officeToolsEnabled, m
|
|
|
27
27
|
media: async (args, { callerCtx, callerCwd }) => {
|
|
28
28
|
requireEnabled(callerCtx, mediaToolEnabled, 'media');
|
|
29
29
|
const { executeMediaTool } = await import('../../runtime/media/tool.mjs');
|
|
30
|
-
|
|
30
|
+
const { getSession } = await import('../../runtime/agent/orchestrator/session/manager/session-crud.mjs');
|
|
31
|
+
// Generation spend is attributed to the calling session in the usage ledger.
|
|
32
|
+
const sessionId = sessionIdFor(callerCtx) || '';
|
|
33
|
+
return await executeMediaTool(args, {
|
|
34
|
+
cwd: callerCwd,
|
|
35
|
+
signal: signalFor(callerCtx),
|
|
36
|
+
sessionId,
|
|
37
|
+
sourceType: (sessionId && getSession(sessionId)?.sourceType) || '',
|
|
38
|
+
});
|
|
31
39
|
},
|
|
32
40
|
tidy: async (args, { callerCtx, callerCwd }) => {
|
|
33
41
|
requireEnabled(callerCtx, tidyToolEnabled, 'tidy');
|
|
@@ -51,6 +51,38 @@ function recordOps(args, run) {
|
|
|
51
51
|
});
|
|
52
52
|
}
|
|
53
53
|
|
|
54
|
+
/**
|
|
55
|
+
* Collapse a sequence of op sets into one op set per argument slot with the
|
|
56
|
+
* same sequential outcome: every key ever deleted is deleted (so a foreign
|
|
57
|
+
* writer's key is still removed), then each surviving key is set once with its
|
|
58
|
+
* final value, ordered as sequential insertion would leave it (an overwrite
|
|
59
|
+
* keeps its position, a delete/reinsert moves to the end). Set adds are unioned.
|
|
60
|
+
*/
|
|
61
|
+
function compactOps(batch) {
|
|
62
|
+
const slots = [];
|
|
63
|
+
for (const ops of batch) {
|
|
64
|
+
ops.forEach((op, index) => {
|
|
65
|
+
if (!op) return;
|
|
66
|
+
let slot = slots[index];
|
|
67
|
+
if (!slot) {
|
|
68
|
+
slot = slots[index] = op.add ? { add: new Set() } : { del: new Set(), final: new Map() };
|
|
69
|
+
}
|
|
70
|
+
if (slot.add) {
|
|
71
|
+
for (const value of op.add) slot.add.add(value);
|
|
72
|
+
return;
|
|
73
|
+
}
|
|
74
|
+
for (const key of op.del) {
|
|
75
|
+
slot.del.add(key);
|
|
76
|
+
slot.final.delete(key);
|
|
77
|
+
}
|
|
78
|
+
for (const [key, json] of op.set) slot.final.set(key, json);
|
|
79
|
+
});
|
|
80
|
+
}
|
|
81
|
+
return slots.map((slot) =>
|
|
82
|
+
!slot ? null : slot.add ? { add: slot.add } : { del: slot.del, set: slot.final }
|
|
83
|
+
);
|
|
84
|
+
}
|
|
85
|
+
|
|
54
86
|
function replayOps(ops, args) {
|
|
55
87
|
args.forEach((arg, index) => {
|
|
56
88
|
const op = ops[index];
|
|
@@ -64,6 +96,10 @@ function replayOps(ops, args) {
|
|
|
64
96
|
});
|
|
65
97
|
}
|
|
66
98
|
|
|
99
|
+
function replayBatch(batch, args) {
|
|
100
|
+
replayOps(compactOps(batch), args);
|
|
101
|
+
}
|
|
102
|
+
|
|
67
103
|
/**
|
|
68
104
|
* @param {object} options
|
|
69
105
|
* @param {string|null} options.file index file; null → every write is a no-op
|
|
@@ -99,9 +135,7 @@ export function createIndexWriteQueue({ file, rewrite, readDoc, onPersisted = ()
|
|
|
99
135
|
await updateJsonAtomic(
|
|
100
136
|
file,
|
|
101
137
|
(cur) =>
|
|
102
|
-
rewrite(cur, (...args) =>
|
|
103
|
-
for (const ops of batch) replayOps(ops, args);
|
|
104
|
-
}),
|
|
138
|
+
rewrite(cur, (...args) => replayBatch(batch, args)),
|
|
105
139
|
{ lock: true }
|
|
106
140
|
);
|
|
107
141
|
return;
|
|
@@ -153,9 +187,7 @@ export function createIndexWriteQueue({ file, rewrite, readDoc, onPersisted = ()
|
|
|
153
187
|
updateJsonAtomicSync(
|
|
154
188
|
file,
|
|
155
189
|
(cur) =>
|
|
156
|
-
rewrite(cur, (...args) =>
|
|
157
|
-
for (const set of ops) replayOps(set, args);
|
|
158
|
-
}),
|
|
190
|
+
rewrite(cur, (...args) => replayBatch(ops, args)),
|
|
159
191
|
{ lock: true, timeoutMs: EXIT_LOCK_TIMEOUT_MS }
|
|
160
192
|
);
|
|
161
193
|
} catch {
|
|
@@ -101,9 +101,15 @@ export function createWorkerRowStore(file) {
|
|
|
101
101
|
return rows;
|
|
102
102
|
}
|
|
103
103
|
|
|
104
|
+
// Rows and tombstones must come from the same read/projection. In
|
|
105
|
+
// particular, do not stat and reload the file between the two collections.
|
|
106
|
+
function readSnapshot() {
|
|
107
|
+
const rows = readAll();
|
|
108
|
+
return { rows, tombstones: (projectedView() || cache)?.tombstones || [] };
|
|
109
|
+
}
|
|
110
|
+
|
|
104
111
|
function readTombstones() {
|
|
105
|
-
|
|
106
|
-
return (projectedView() || cache)?.tombstones || [];
|
|
112
|
+
return readSnapshot().tombstones;
|
|
107
113
|
}
|
|
108
114
|
|
|
109
115
|
// Single writer path: the mutator runs now over keyed maps against the
|
|
@@ -113,6 +119,7 @@ export function createWorkerRowStore(file) {
|
|
|
113
119
|
|
|
114
120
|
return {
|
|
115
121
|
readAll,
|
|
122
|
+
readSnapshot,
|
|
116
123
|
readTombstones,
|
|
117
124
|
write,
|
|
118
125
|
/** Resolves once every write so far is on disk. */
|
|
@@ -27,16 +27,19 @@ export function createWorkerIndex({ dataDir, cfgMod, mgr, tags, tagAgents, tagCw
|
|
|
27
27
|
const activeWorkerKeys = new Set();
|
|
28
28
|
|
|
29
29
|
function readWorkerRows(context = {}) {
|
|
30
|
-
const rows = store.
|
|
30
|
+
const { rows, tombstones: savedTombstones } = store.readSnapshot();
|
|
31
31
|
if (rows.length === 0) return rows;
|
|
32
|
-
const tombstones = new Map(
|
|
32
|
+
const tombstones = new Map(recoverTombstoneOwners(savedTombstones).map((row) => [tagTombstoneKey(row), row]));
|
|
33
33
|
return rows.filter(
|
|
34
34
|
(row) => rowMatchesContext(row, context) && !tombstoneBlocksWork(row, findTagTombstone(row, tombstones))
|
|
35
35
|
);
|
|
36
36
|
}
|
|
37
37
|
|
|
38
38
|
function readAllTagTombstones() {
|
|
39
|
-
|
|
39
|
+
return recoverTombstoneOwners(store.readTombstones());
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
function recoverTombstoneOwners(rows) {
|
|
40
43
|
if (!rows.some((row) => !clean(row.parentSessionId || row.ownerSessionId))) return rows;
|
|
41
44
|
const sessions = typeof mgr.listSessions === 'function' ? mgr.listSessions({ includeClosed: true }) : [];
|
|
42
45
|
// Resolve legacy ownership against all sessions, never just the caller's
|