mixdog 1.0.1 → 1.0.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -0
- package/package.json +2 -2
- package/src/defaults/skills/browser-use/SKILL.md +14 -6
- package/src/defaults/skills/setup/references/actions.md +6 -3
- package/src/defaults/skills/setup/references/surfaces.md +8 -4
- package/src/runtime/agent/orchestrator/context/role-instructions.mjs +0 -1
- package/src/runtime/agent/orchestrator/providers/anthropic-midstream-recovery.mjs +10 -1
- package/src/runtime/agent/orchestrator/session/cache/scoped-cache.mjs +40 -99
- package/src/runtime/agent/orchestrator/session/compact/runner.mjs +6 -1
- package/src/runtime/agent/orchestrator/session/loop/fresh-context.mjs +2 -0
- package/src/runtime/agent/orchestrator/session/loop/tool-exec/before-hook.mjs +7 -3
- package/src/runtime/agent/orchestrator/session/manager/session-prompt-composition.mjs +1 -1
- package/src/runtime/agent/orchestrator/tools/builtin/cache-layers.mjs +18 -38
- package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-path-fanout.mjs +59 -32
- package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-pattern-fanout.mjs +15 -10
- package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-request.mjs +1 -1
- package/src/runtime/agent/orchestrator/tools/builtin/lib/shared-native-scan.mjs +12 -0
- package/src/runtime/agent/orchestrator/tools/builtin/read-source-windows.mjs +14 -0
- package/src/runtime/agent/orchestrator/tools/graph-manifest.json +13 -13
- package/src/runtime/agent/orchestrator/tools/patch/apply-patch/codex-batch.mjs +3 -1
- package/src/runtime/agent/orchestrator/tools/patch/dispatch.mjs +17 -9
- package/src/runtime/agent/orchestrator/tools/patch/orchestrator.mjs +1 -0
- package/src/runtime/agent/orchestrator/tools/patch/wave.mjs +4 -0
- package/src/runtime/browser-bridge/input-fields.mjs +1 -1
- package/src/runtime/browser-bridge/tool-defs.mjs +1 -1
- package/src/runtime/computer-bridge/actions.mjs +0 -13
- package/src/runtime/media/adapters/antigravity-image.mjs +7 -3
- package/src/runtime/media/adapters/codex-image.mjs +12 -1
- package/src/runtime/media/adapters/gemini-image.mjs +17 -6
- package/src/runtime/media/adapters/gemini-video.mjs +19 -9
- package/src/runtime/media/adapters/xai-media.mjs +20 -7
- package/src/runtime/media/jobs.mjs +35 -3
- package/src/runtime/media/media-usage.mjs +126 -0
- package/src/runtime/media/tool.mjs +16 -4
- package/src/runtime/shared/llm/cost.mjs +50 -3
- package/src/runtime/shared/llm/model-catalog-projection.mjs +7 -2
- package/src/runtime/shared/llm/model-catalog.mjs +2 -1
- package/src/runtime/shared/llm/model-pricing-rates.mjs +21 -0
- package/src/runtime/shared/llm/usage-accounting.mjs +25 -5
- package/src/runtime/shared/llm/usage-context.mjs +6 -0
- package/src/runtime/shared/llm/usage-ledger-import.mjs +1 -0
- package/src/session-runtime/cwd-plugins/plugin-status.mjs +24 -4
- package/src/session-runtime/internal-tool-executor/feature-tools.mjs +9 -1
- package/src/session-runtime/services/agent-tool/index-write-queue.mjs +38 -6
- package/src/session-runtime/services/agent-tool/worker-index/row-store.mjs +9 -2
- package/src/session-runtime/services/agent-tool/worker-index.mjs +6 -3
- package/src/session-runtime/services/hook-bus/event-runner.mjs +10 -2
- package/src/session-runtime/services/hook-bus/tool-gate.mjs +5 -2
- package/src/session-runtime/setup-tool/tool-defs.mjs +1 -1
- package/src/session-runtime/workflow-agents-api/agent-route.mjs +8 -5
- package/src/tui/app/doctor.mjs +78 -2
- package/src/tui/dist/index.mjs +7 -3
- package/scripts/lib/bench-sandbox.test.mjs +0 -54
- package/scripts/lib/cli-args.test.mjs +0 -64
- package/scripts/lib/embedding-model-bench-core.test.mjs +0 -47
- package/scripts/lib/smoke-loop-shared.test.mjs +0 -22
- package/scripts/lib/trace-row.test.mjs +0 -47
- package/scripts/lib/trace-stats.test.mjs +0 -62
|
@@ -12,7 +12,9 @@ import { boundedSignal } from '../bounded-signal.mjs';
|
|
|
12
12
|
import { decodeBase64Media, downloadGeminiMedia } from '../download.mjs';
|
|
13
13
|
import { mediaError } from '../lanes.mjs';
|
|
14
14
|
import { upstreamError } from '../upstream-error.mjs';
|
|
15
|
+
import { interactionsUsage, withReportedUsage } from '../media-usage.mjs';
|
|
15
16
|
|
|
17
|
+
const VEO_DEFAULT_SECONDS = 8;
|
|
16
18
|
const BASE_URL = 'https://generativelanguage.googleapis.com/v1beta';
|
|
17
19
|
const OMNI_TIMEOUT_MS = 600_000;
|
|
18
20
|
const POLL_INTERVAL_MS = 8_000;
|
|
@@ -54,14 +56,15 @@ async function generateViaOmni({ model, prompt, options, references = [], signal
|
|
|
54
56
|
});
|
|
55
57
|
if (!res.ok) throw upstreamError('Gemini Omni video', res.status, await res.text().catch(() => ''));
|
|
56
58
|
const data = await res.json();
|
|
59
|
+
const usage = interactionsUsage(data?.usage);
|
|
57
60
|
for (const step of data?.steps || []) {
|
|
58
61
|
const content = Array.isArray(step?.content) ? step.content : [step?.content].filter(Boolean);
|
|
59
62
|
const video = content.find((item) => item?.type === 'video' && typeof item?.data === 'string');
|
|
60
63
|
if (video) {
|
|
61
|
-
return { bytes: decodeBase64Media(video.data, 'Gemini Omni video'), mime: video.mime_type || 'video/mp4' };
|
|
64
|
+
return { bytes: decodeBase64Media(video.data, 'Gemini Omni video'), mime: video.mime_type || 'video/mp4', usage };
|
|
62
65
|
}
|
|
63
66
|
}
|
|
64
|
-
throw mediaError('Gemini Omni returned no video data', 'MEDIA_EMPTY_RESULT', 502);
|
|
67
|
+
throw withReportedUsage(mediaError('Gemini Omni returned no video data', 'MEDIA_EMPTY_RESULT', 502), usage);
|
|
65
68
|
}
|
|
66
69
|
|
|
67
70
|
async function generateViaVeo({ model, prompt, options, references = [], signal, onProgress, key }) {
|
|
@@ -114,13 +117,20 @@ async function generateViaVeo({ model, prompt, options, references = [], signal,
|
|
|
114
117
|
const sample = data?.response?.generateVideoResponse?.generatedSamples?.[0] || data?.response?.generatedVideos?.[0];
|
|
115
118
|
const uri = sample?.video?.uri || sample?.video?.fileUri;
|
|
116
119
|
if (!uri) throw mediaError('Veo finished without a video URI', 'MEDIA_EMPTY_RESULT', 502);
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
120
|
+
// Veo reports no usage; it bills per generated second. The response does
|
|
121
|
+
// not echo the length, so the requested one (default 8s) is the billed one.
|
|
122
|
+
// A failed download still leaves the generated seconds billed.
|
|
123
|
+
const usage = { seconds: parameters.durationSeconds || VEO_DEFAULT_SECONDS, resolution: parameters.resolution || '720p' };
|
|
124
|
+
// The download is part of the generation budget: unbounded, it held an
|
|
125
|
+
// active job slot (and the upstream connection) open indefinitely after
|
|
126
|
+
// the poll loop finished.
|
|
127
|
+
const bytes = await downloadGeminiMedia(uri, {
|
|
128
|
+
key,
|
|
129
|
+
signal: boundedSignal(signal, deadline, TOTAL_TIMEOUT_MS),
|
|
130
|
+
}).catch((error) => {
|
|
131
|
+
throw withReportedUsage(error, usage);
|
|
132
|
+
});
|
|
133
|
+
return { bytes, mime: 'video/mp4', usage };
|
|
124
134
|
}
|
|
125
135
|
}
|
|
126
136
|
|
|
@@ -10,6 +10,7 @@ import { boundedSignal, timeoutSignal } from '../bounded-signal.mjs';
|
|
|
10
10
|
import { decodeBase64Media, downloadPublicMedia } from '../download.mjs';
|
|
11
11
|
import { mediaError } from '../lanes.mjs';
|
|
12
12
|
import { upstreamError } from '../upstream-error.mjs';
|
|
13
|
+
import { withReportedUsage, xaiUsage } from '../media-usage.mjs';
|
|
13
14
|
|
|
14
15
|
const POLL_INTERVAL_MS = 4_000;
|
|
15
16
|
const START_TIMEOUT_MS = 60_000;
|
|
@@ -72,6 +73,7 @@ export async function generateImage({ lane, model, prompt, options = {}, referen
|
|
|
72
73
|
bytes: decodeBase64Media(entry.b64_json, 'xAI image'),
|
|
73
74
|
mime: entry.mime_type || 'image/png',
|
|
74
75
|
revisedPrompt: entry.revised_prompt || null,
|
|
76
|
+
usage: xaiUsage(data?.usage),
|
|
75
77
|
};
|
|
76
78
|
}
|
|
77
79
|
|
|
@@ -114,18 +116,29 @@ export async function generateVideo({ lane, model, prompt, options = {}, referen
|
|
|
114
116
|
if (typeof data?.progress === 'number' && typeof onProgress === 'function') onProgress(data.progress);
|
|
115
117
|
if (data?.status === 'done') {
|
|
116
118
|
const url = data?.video?.url;
|
|
117
|
-
|
|
119
|
+
// The billed cost (when reported) is the price; seconds + resolution
|
|
120
|
+
// let the catalog's per-second rate price it otherwise.
|
|
121
|
+
const seconds = Number(data?.video?.duration) || duration;
|
|
122
|
+
const usage = { ...xaiUsage(data?.usage), seconds, resolution };
|
|
123
|
+
if (!url) throw withReportedUsage(mediaError('xAI video finished without a URL', 'MEDIA_EMPTY_RESULT', 502), usage);
|
|
124
|
+
const bytes = await downloadPublicMedia(url, { signal, label: 'xAI video' }).catch((error) => {
|
|
125
|
+
throw withReportedUsage(error, usage);
|
|
126
|
+
});
|
|
118
127
|
return {
|
|
119
|
-
bytes
|
|
128
|
+
bytes,
|
|
120
129
|
mime: 'video/mp4',
|
|
121
|
-
durationSeconds:
|
|
130
|
+
durationSeconds: seconds,
|
|
131
|
+
usage,
|
|
122
132
|
};
|
|
123
133
|
}
|
|
124
134
|
if (data?.status === 'failed' || data?.status === 'expired') {
|
|
125
|
-
throw
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
135
|
+
throw withReportedUsage(
|
|
136
|
+
mediaError(
|
|
137
|
+
`xAI video ${data.status}${data?.error?.code ? `: ${data.error.code}` : ''}`,
|
|
138
|
+
'MEDIA_UPSTREAM_FAILED',
|
|
139
|
+
502
|
|
140
|
+
),
|
|
141
|
+
xaiUsage(data?.usage)
|
|
129
142
|
);
|
|
130
143
|
}
|
|
131
144
|
}
|
|
@@ -11,6 +11,7 @@ import { MAX_GENERATED_MEDIA_BYTES } from './download.mjs';
|
|
|
11
11
|
import { mediaError, resolveMediaRequest } from './lanes.mjs';
|
|
12
12
|
import { saveMediaAsset } from './store.mjs';
|
|
13
13
|
import { setMediaDefault } from './defaults.mjs';
|
|
14
|
+
import { recordMediaUsage } from './media-usage.mjs';
|
|
14
15
|
|
|
15
16
|
const JOBS = new Map();
|
|
16
17
|
// Finished jobs stay readable for a while so a slow poller still sees the
|
|
@@ -77,7 +78,16 @@ async function runAdapter({ lane, kind, model, requestModel, prompt, options, re
|
|
|
77
78
|
* Validate + start one generation. Returns the initial snapshot immediately;
|
|
78
79
|
* the caller polls getMediaJob for progress and the finished asset id.
|
|
79
80
|
*/
|
|
80
|
-
export async function startMediaJob({
|
|
81
|
+
export async function startMediaJob({
|
|
82
|
+
lane: laneId,
|
|
83
|
+
kind,
|
|
84
|
+
model,
|
|
85
|
+
prompt,
|
|
86
|
+
options = {},
|
|
87
|
+
references = [],
|
|
88
|
+
sessionId = '',
|
|
89
|
+
sourceType = '',
|
|
90
|
+
} = {}) {
|
|
81
91
|
const text = String(prompt || '').trim();
|
|
82
92
|
if (!text) throw mediaError('prompt is required', 'MEDIA_PROMPT_REQUIRED');
|
|
83
93
|
if (text.length > MAX_PROMPT_CHARS) throw mediaError('prompt is too long', 'MEDIA_PROMPT_TOO_LONG');
|
|
@@ -128,13 +138,31 @@ export async function startMediaJob({ lane: laneId, kind, model, prompt, options
|
|
|
128
138
|
controller,
|
|
129
139
|
};
|
|
130
140
|
JOBS.set(job.id, job);
|
|
131
|
-
void runJob(job, resolved, {
|
|
141
|
+
void runJob(job, resolved, {
|
|
142
|
+
requestModel: modelEntry?.requestModel,
|
|
143
|
+
options,
|
|
144
|
+
references: refs,
|
|
145
|
+
sessionId,
|
|
146
|
+
sourceType,
|
|
147
|
+
});
|
|
132
148
|
return snapshot(job);
|
|
133
149
|
}
|
|
134
150
|
|
|
135
151
|
/** Drive one started job to a terminal state; never rejects. */
|
|
136
|
-
async function runJob(job, resolved, { requestModel, options, references }) {
|
|
152
|
+
async function runJob(job, resolved, { requestModel, options, references, sessionId, sourceType }) {
|
|
137
153
|
const { controller } = job;
|
|
154
|
+
// One ledger row per generation. The ChatGPT lane bills the orchestrator
|
|
155
|
+
// model that ran the hosted tool, not the "auto" image route.
|
|
156
|
+
const record = (usage) =>
|
|
157
|
+
recordMediaUsage({
|
|
158
|
+
lane: job.lane,
|
|
159
|
+
model: job.model,
|
|
160
|
+
pricingModel: job.lane === 'openai-oauth' ? requestModel : undefined,
|
|
161
|
+
usage,
|
|
162
|
+
sessionId,
|
|
163
|
+
sourceType,
|
|
164
|
+
durationMs: Date.now() - job.startedAt,
|
|
165
|
+
});
|
|
138
166
|
try {
|
|
139
167
|
const result = await runAdapter({
|
|
140
168
|
lane: resolved.lane,
|
|
@@ -154,6 +182,8 @@ async function runJob(job, resolved, { requestModel, options, references }) {
|
|
|
154
182
|
if (next > job.progress) job.progress = next;
|
|
155
183
|
},
|
|
156
184
|
});
|
|
185
|
+
// Billed once the provider returned a result, whatever happens to the bytes next.
|
|
186
|
+
await record(result?.usage);
|
|
157
187
|
if (!Buffer.isBuffer(result?.bytes) || !result.bytes.length || result.bytes.length > MAX_GENERATED_MEDIA_BYTES) {
|
|
158
188
|
throw mediaError('generated media exceeds the media size limit', 'MEDIA_RESULT_TOO_LARGE', 502);
|
|
159
189
|
}
|
|
@@ -175,6 +205,8 @@ async function runJob(job, resolved, { requestModel, options, references }) {
|
|
|
175
205
|
job.status = 'done';
|
|
176
206
|
} catch (err) {
|
|
177
207
|
const canceled = controller.signal.aborted || err?.code === 'MEDIA_CANCELED' || err?.name === 'AbortError';
|
|
208
|
+
// A failure is recorded only when the provider reported usage for it.
|
|
209
|
+
if (err?.usage) await record(err.usage);
|
|
178
210
|
job.status = canceled ? 'canceled' : 'failed';
|
|
179
211
|
job.error = canceled ? 'canceled' : String(err?.message || err).slice(0, 500);
|
|
180
212
|
job.errorCode = err?.code || null;
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Usage ledger rows for media generations.
|
|
3
|
+
*
|
|
4
|
+
* Adapters return (or attach to their error) a normalized `usage` object read
|
|
5
|
+
* from what the upstream response actually reported:
|
|
6
|
+
* { inputTokens, outputTokens, cachedTokens, outputImageTokens, outputVideoTokens,
|
|
7
|
+
* costUsd, images, seconds, resolution }
|
|
8
|
+
* `jobs.mjs` records exactly one row per generation through recordMediaUsage.
|
|
9
|
+
* Nothing here estimates tokens: a field the provider did not report stays absent.
|
|
10
|
+
*/
|
|
11
|
+
import { getUsageLedger, makeUsageRecord } from '../shared/llm/usage-ledger.mjs';
|
|
12
|
+
import { ACCOUNT_PROVIDERS } from '../shared/provider-accounts.mjs';
|
|
13
|
+
import { currentProviderAccountId } from '../shared/provider-auth-binding.mjs';
|
|
14
|
+
|
|
15
|
+
const count = (value) => (Number.isFinite(Number(value)) && Number(value) > 0 ? Math.trunc(Number(value)) : 0);
|
|
16
|
+
const pick = (source, ...keys) => {
|
|
17
|
+
for (const key of keys) if (source?.[key] != null) return source[key];
|
|
18
|
+
return undefined;
|
|
19
|
+
};
|
|
20
|
+
|
|
21
|
+
const TICKS_PER_USD = 1e10;
|
|
22
|
+
|
|
23
|
+
/** Token usage when the provider reported any; null otherwise. */
|
|
24
|
+
function tokenUsage(fields) {
|
|
25
|
+
const usage = Object.fromEntries(Object.entries(fields).filter(([, value]) => value > 0));
|
|
26
|
+
return Object.keys(usage).length ? usage : null;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** Gemini `generateContent` / Antigravity `usageMetadata` (camelCase or snake_case). */
|
|
30
|
+
export function geminiUsage(meta) {
|
|
31
|
+
if (!meta || typeof meta !== 'object') return null;
|
|
32
|
+
const details = pick(meta, 'candidatesTokensDetails', 'candidates_tokens_details');
|
|
33
|
+
const imageTokens = (Array.isArray(details) ? details : [])
|
|
34
|
+
.filter((entry) => String(entry?.modality || '').toUpperCase() === 'IMAGE')
|
|
35
|
+
.reduce((sum, entry) => sum + count(pick(entry, 'tokenCount', 'token_count')), 0);
|
|
36
|
+
return tokenUsage({
|
|
37
|
+
inputTokens: count(pick(meta, 'promptTokenCount', 'prompt_token_count')),
|
|
38
|
+
cachedTokens: count(pick(meta, 'cachedContentTokenCount', 'cached_content_token_count')),
|
|
39
|
+
outputTokens:
|
|
40
|
+
count(pick(meta, 'candidatesTokenCount', 'candidates_token_count')) +
|
|
41
|
+
count(pick(meta, 'thoughtsTokenCount', 'thoughts_token_count')),
|
|
42
|
+
outputImageTokens: imageTokens,
|
|
43
|
+
});
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** Gemini Interactions API `usage` (omni video). */
|
|
47
|
+
export function interactionsUsage(usage) {
|
|
48
|
+
if (!usage || typeof usage !== 'object') return null;
|
|
49
|
+
const videoTokens = (Array.isArray(usage.output_tokens_by_modality) ? usage.output_tokens_by_modality : [])
|
|
50
|
+
.filter((entry) => String(entry?.modality || '').toLowerCase() === 'video')
|
|
51
|
+
.reduce((sum, entry) => sum + count(entry?.tokens), 0);
|
|
52
|
+
return tokenUsage({
|
|
53
|
+
inputTokens: count(usage.total_input_tokens),
|
|
54
|
+
cachedTokens: count(usage.total_cached_tokens),
|
|
55
|
+
outputTokens: count(usage.total_output_tokens) + count(usage.total_thought_tokens),
|
|
56
|
+
outputVideoTokens: videoTokens,
|
|
57
|
+
});
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/** OpenAI / Codex Responses `usage` from response.completed. */
|
|
61
|
+
export function responsesUsage(usage) {
|
|
62
|
+
if (!usage || typeof usage !== 'object') return null;
|
|
63
|
+
return tokenUsage({
|
|
64
|
+
inputTokens: count(usage.input_tokens),
|
|
65
|
+
cachedTokens: count(usage.input_tokens_details?.cached_tokens),
|
|
66
|
+
outputTokens: count(usage.output_tokens),
|
|
67
|
+
});
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** xAI image/video `usage.cost_in_usd_ticks` (1 USD = 1e10 ticks); the billed cost. */
|
|
71
|
+
export function xaiUsage(usage) {
|
|
72
|
+
const ticks = usage?.cost_in_usd_ticks;
|
|
73
|
+
return typeof ticks === 'number' && Number.isFinite(ticks) && ticks >= 0 ? { costUsd: ticks / TICKS_PER_USD } : null;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** True when the provider reported something billable for the request. */
|
|
77
|
+
export function hasReportedUsage(usage) {
|
|
78
|
+
if (!usage) return false;
|
|
79
|
+
// Seconds are only attached once a video generation completed (billed per second).
|
|
80
|
+
return (
|
|
81
|
+
typeof usage.costUsd === 'number' ||
|
|
82
|
+
['inputTokens', 'outputTokens', 'cachedTokens', 'seconds'].some((k) => usage[k] > 0)
|
|
83
|
+
);
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** Attach provider-reported usage to a failure so jobs.mjs can still record it. */
|
|
87
|
+
export function withReportedUsage(error, usage) {
|
|
88
|
+
if (hasReportedUsage(usage)) error.usage = usage;
|
|
89
|
+
return error;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Append one ledger row for a generation. A ledger failure never fails the
|
|
94
|
+
* generation; it is logged the way usage-accounting does.
|
|
95
|
+
*/
|
|
96
|
+
export async function recordMediaUsage(args, getLedger = getUsageLedger) {
|
|
97
|
+
try {
|
|
98
|
+
const usage = args.usage || {};
|
|
99
|
+
const row = makeUsageRecord({
|
|
100
|
+
ts: Date.now(),
|
|
101
|
+
provider: args.lane,
|
|
102
|
+
model: args.model,
|
|
103
|
+
requestedModel: args.model,
|
|
104
|
+
pricingModel: args.pricingModel,
|
|
105
|
+
sessionId: args.sessionId,
|
|
106
|
+
sourceType: args.sourceType,
|
|
107
|
+
inputTokens: usage.inputTokens,
|
|
108
|
+
outputTokens: usage.outputTokens,
|
|
109
|
+
cacheReadTokens: usage.cachedTokens,
|
|
110
|
+
costUsd: usage.costUsd,
|
|
111
|
+
media: {
|
|
112
|
+
reportedCostUsd: usage.costUsd,
|
|
113
|
+
outputImageTokens: usage.outputImageTokens,
|
|
114
|
+
outputVideoTokens: usage.outputVideoTokens,
|
|
115
|
+
images: usage.images,
|
|
116
|
+
seconds: usage.seconds,
|
|
117
|
+
resolution: usage.resolution,
|
|
118
|
+
},
|
|
119
|
+
account: ACCOUNT_PROVIDERS.includes(args.lane) ? currentProviderAccountId(args.lane) : '',
|
|
120
|
+
durationMs: args.durationMs,
|
|
121
|
+
});
|
|
122
|
+
await getLedger()?.recordQueued(row);
|
|
123
|
+
} catch (error) {
|
|
124
|
+
process.stderr.write(`[usage-ledger] RECORD NOT SAVED: ${String(error?.message || error)}\n`);
|
|
125
|
+
}
|
|
126
|
+
}
|
|
@@ -280,7 +280,7 @@ function jobView(job) {
|
|
|
280
280
|
};
|
|
281
281
|
}
|
|
282
282
|
|
|
283
|
-
async function generate(args, { cwd, signal, deps }) {
|
|
283
|
+
async function generate(args, { cwd, signal, deps, sessionId, sourceType }) {
|
|
284
284
|
const kind = clean(args.kind);
|
|
285
285
|
if (!MEDIA_KINDS.includes(kind)) throw new MediaToolError(`generate requires kind: ${MEDIA_KINDS.join(' | ')}`);
|
|
286
286
|
const prompt = clean(args.prompt);
|
|
@@ -314,7 +314,16 @@ async function generate(args, { cwd, signal, deps }) {
|
|
|
314
314
|
quality: clean(args.quality),
|
|
315
315
|
});
|
|
316
316
|
const references = await readReferences(args.references, cwd, model.controls);
|
|
317
|
-
const started = await jobs.startMediaJob({
|
|
317
|
+
const started = await jobs.startMediaJob({
|
|
318
|
+
lane: lane.id,
|
|
319
|
+
kind,
|
|
320
|
+
model: modelId,
|
|
321
|
+
prompt,
|
|
322
|
+
options,
|
|
323
|
+
references,
|
|
324
|
+
sessionId,
|
|
325
|
+
sourceType,
|
|
326
|
+
});
|
|
318
327
|
const base = { lane: lane.id, model: modelId, laneSource, options, referenceCount: references.length, prompt };
|
|
319
328
|
if (args.wait === false) {
|
|
320
329
|
return {
|
|
@@ -371,7 +380,10 @@ async function cancel(args, { deps }) {
|
|
|
371
380
|
return { ok: true, ...jobView(jobs.getMediaJob(id)), canceled: canceled === true };
|
|
372
381
|
}
|
|
373
382
|
|
|
374
|
-
export async function executeMediaTool(
|
|
383
|
+
export async function executeMediaTool(
|
|
384
|
+
args = {},
|
|
385
|
+
{ cwd = process.cwd(), signal = null, deps = null, sessionId = '', sourceType = '' } = {}
|
|
386
|
+
) {
|
|
375
387
|
const action = clean(args.action).toLowerCase();
|
|
376
388
|
try {
|
|
377
389
|
if (!MEDIA_ACTIONS.includes(action))
|
|
@@ -388,7 +400,7 @@ export async function executeMediaTool(args = {}, { cwd = process.cwd(), signal
|
|
|
388
400
|
),
|
|
389
401
|
});
|
|
390
402
|
}
|
|
391
|
-
if (action === 'generate') return mediaToolResult(await generate(args, { cwd, signal, deps }));
|
|
403
|
+
if (action === 'generate') return mediaToolResult(await generate(args, { cwd, signal, deps, sessionId, sourceType }));
|
|
392
404
|
if (action === 'status') return mediaToolResult(await status(args, { cwd, deps }));
|
|
393
405
|
return mediaToolResult(await cancel(args, { deps }));
|
|
394
406
|
} catch (error) {
|
|
@@ -39,6 +39,35 @@ export function billableInputTokensForProvider(provider, inputTokens, cacheReadT
|
|
|
39
39
|
return Math.max(input - (Number(cacheReadTokens) || 0) - (Number(cacheWriteTokens) || 0), 0);
|
|
40
40
|
}
|
|
41
41
|
|
|
42
|
+
/**
|
|
43
|
+
* Media billed by something other than the text token slots: image/video output
|
|
44
|
+
* tokens, whole images, or video seconds (per resolution when the catalog lists
|
|
45
|
+
* one). A used unit without a catalog rate is reported missing, never free.
|
|
46
|
+
*/
|
|
47
|
+
function mediaCharges(meta, media, imageOut, videoOut) {
|
|
48
|
+
const n = (value) => (Number.isFinite(Number(value)) && Number(value) > 0 ? Number(value) : 0);
|
|
49
|
+
const out = { usd: 0, charged: false, missing: [], rates: {} };
|
|
50
|
+
const bill = (amount, key, rate, scale = 1) => {
|
|
51
|
+
if (amount <= 0) return;
|
|
52
|
+
out.charged = true;
|
|
53
|
+
if (rate == null) out.missing.push(key);
|
|
54
|
+
else {
|
|
55
|
+
out.usd += (amount * rate) / scale;
|
|
56
|
+
out.rates[key] = rate;
|
|
57
|
+
}
|
|
58
|
+
};
|
|
59
|
+
bill(imageOut, 'outputImageCostPerM', meta.outputImageCostPerM, 1_000_000);
|
|
60
|
+
bill(videoOut, 'outputVideoCostPerM', meta.outputVideoCostPerM, 1_000_000);
|
|
61
|
+
bill(n(media.images), 'outputCostPerImage', meta.outputCostPerImage);
|
|
62
|
+
const resolution = String(media.resolution || '').toLowerCase();
|
|
63
|
+
bill(
|
|
64
|
+
n(media.seconds),
|
|
65
|
+
'outputCostPerSecond',
|
|
66
|
+
meta.outputCostPerSecondByResolution?.[resolution] ?? meta.outputCostPerSecond
|
|
67
|
+
);
|
|
68
|
+
return out;
|
|
69
|
+
}
|
|
70
|
+
|
|
42
71
|
/**
|
|
43
72
|
* Price normalized token slots once. Null means unknown, not a free request.
|
|
44
73
|
* Rates are returned so a durable record keeps the price applied at ingestion.
|
|
@@ -68,6 +97,15 @@ export function priceUsage(args) {
|
|
|
68
97
|
...(args.serviceTier ? { serviceTier: args.serviceTier } : {}),
|
|
69
98
|
...(written1h ? { cacheWrite1hTokens: written1h } : {}),
|
|
70
99
|
};
|
|
100
|
+
// A provider-billed figure for a media request (e.g. xAI cost_in_usd_ticks)
|
|
101
|
+
// is the price, with or without a catalog row.
|
|
102
|
+
const media = args.media || null;
|
|
103
|
+
if (typeof media?.reportedCostUsd === 'number' && Number.isFinite(media.reportedCostUsd) && media.reportedCostUsd >= 0)
|
|
104
|
+
return {
|
|
105
|
+
input,
|
|
106
|
+
costUsd: Number(media.reportedCostUsd.toFixed(6)),
|
|
107
|
+
rates: { ...provenance, pricingSource: 'provider' },
|
|
108
|
+
};
|
|
71
109
|
if (args.inputTokensKnown === false || !meta)
|
|
72
110
|
return {
|
|
73
111
|
input,
|
|
@@ -107,7 +145,11 @@ export function priceUsage(args) {
|
|
|
107
145
|
// Fast mode bills 2x standard rates on every fast-capable Opus.
|
|
108
146
|
if (anthropicFast) multiplier *= 2;
|
|
109
147
|
const keys = PRICING_RATE_KEYS;
|
|
110
|
-
|
|
148
|
+
// Image/video output tokens bill above the text output rate; the row still
|
|
149
|
+
// carries the full output count.
|
|
150
|
+
const imageOut = media ? Math.min(n(media.outputImageTokens), n(args.outputTokens)) : 0;
|
|
151
|
+
const videoOut = media ? Math.min(n(media.outputVideoTokens), n(args.outputTokens) - imageOut) : 0;
|
|
152
|
+
const tokens = [input, n(args.outputTokens) - imageOut - videoOut, cached, written - written1h];
|
|
111
153
|
const tierRates = ratesForPrompt(rateMeta, promptTokens);
|
|
112
154
|
const rates = {
|
|
113
155
|
...provenance,
|
|
@@ -116,6 +158,10 @@ export function priceUsage(args) {
|
|
|
116
158
|
if (written1h) rates.cacheWrite1hCostPerM = rates.inputCostPerM === null ? null : rates.inputCostPerM * 2;
|
|
117
159
|
const missingRates = keys.filter((key, i) => tokens[i] > 0 && rates[key] === null);
|
|
118
160
|
if (written1h && rates.cacheWrite1hCostPerM === null) missingRates.push('cacheWrite1hCostPerM');
|
|
161
|
+
const charge = media ? mediaCharges(meta, media, imageOut, videoOut) : { usd: 0, charged: false, missing: [] };
|
|
162
|
+
missingRates.push(...charge.missing);
|
|
163
|
+
if (media && !charge.charged && tokens.every((amount) => amount === 0) && !written1h)
|
|
164
|
+
return { input, costUsd: null, rates: { ...rates, unpricedReason: 'usage-not-reported' } };
|
|
119
165
|
if (missingRates.length) {
|
|
120
166
|
rates.unpricedReason = 'missing-rate';
|
|
121
167
|
rates.missingRates = missingRates;
|
|
@@ -124,8 +170,9 @@ export function priceUsage(args) {
|
|
|
124
170
|
const costUsd =
|
|
125
171
|
(tokens.reduce((sum, amount, i) => sum + amount * (rates[keys[i]] ?? 0), 0) +
|
|
126
172
|
written1h * (rates.cacheWrite1hCostPerM ?? 0)) /
|
|
127
|
-
|
|
128
|
-
|
|
173
|
+
1_000_000 +
|
|
174
|
+
charge.usd;
|
|
175
|
+
return { input, costUsd: Number(costUsd.toFixed(6)), rates: { ...rates, ...charge.rates } };
|
|
129
176
|
}
|
|
130
177
|
|
|
131
178
|
/**
|
|
@@ -124,6 +124,10 @@ const LITELLM_NUMBER_FIELDS = [
|
|
|
124
124
|
'output_cost_per_token',
|
|
125
125
|
'cache_read_input_token_cost',
|
|
126
126
|
'cache_creation_input_token_cost',
|
|
127
|
+
'output_cost_per_image',
|
|
128
|
+
'output_cost_per_image_token',
|
|
129
|
+
'output_cost_per_video_token',
|
|
130
|
+
'output_cost_per_second',
|
|
127
131
|
];
|
|
128
132
|
// _normalize tests each of these with `=== true`, so only a true value carries
|
|
129
133
|
// information; anything else is indistinguishable from absent.
|
|
@@ -148,9 +152,10 @@ function projectLitellmRow(row) {
|
|
|
148
152
|
// litellmPricing), not just the base price.
|
|
149
153
|
for (const [field, value] of Object.entries(row)) {
|
|
150
154
|
if (
|
|
151
|
-
/^(?:input_cost_per_token|output_cost_per_token|cache_read_input_token_cost|cache_creation_input_token_cost)(?:_above_\d+k_tokens)?(?:_priority)?$/.test(
|
|
155
|
+
(/^(?:input_cost_per_token|output_cost_per_token|cache_read_input_token_cost|cache_creation_input_token_cost)(?:_above_\d+k_tokens)?(?:_priority)?$/.test(
|
|
152
156
|
field
|
|
153
|
-
)
|
|
157
|
+
) ||
|
|
158
|
+
/^output_cost_per_second_\w+$/.test(field)) &&
|
|
154
159
|
typeof value === 'number' &&
|
|
155
160
|
Number.isFinite(value)
|
|
156
161
|
)
|
|
@@ -26,7 +26,7 @@ import {
|
|
|
26
26
|
cachedProviderModelListsSync,
|
|
27
27
|
providerCachedModelsSync,
|
|
28
28
|
} from './provider-catalog-cache.mjs';
|
|
29
|
-
import { litellmPricing, modelsDevPricing, PRICING_RATE_KEYS } from './model-pricing-rates.mjs';
|
|
29
|
+
import { litellmMediaPricing, litellmPricing, modelsDevPricing, PRICING_RATE_KEYS } from './model-pricing-rates.mjs';
|
|
30
30
|
// Both overlays are narrowed to their read surface before becoming resident;
|
|
31
31
|
// the disk caches below still receive the full payload.
|
|
32
32
|
import { projectLitellmCatalog, projectModelsDevCatalog } from './model-catalog-projection.mjs';
|
|
@@ -597,6 +597,7 @@ function _normalize(entry) {
|
|
|
597
597
|
contextWindow: entry.max_input_tokens || entry.max_tokens || null,
|
|
598
598
|
outputTokens: entry.max_output_tokens || null,
|
|
599
599
|
...litellmPricing(entry),
|
|
600
|
+
...litellmMediaPricing(entry),
|
|
600
601
|
...(PRICING_RATE_KEYS.some((key) => fastPricing[key] != null) ? { fastPricing } : {}),
|
|
601
602
|
...(entry.off_peak_multiplier ? { offPeakMultiplier: entry.off_peak_multiplier } : {}),
|
|
602
603
|
supportsVision: entry.supports_vision === true,
|
|
@@ -51,6 +51,27 @@ export function litellmPricing(entry, suffix = '') {
|
|
|
51
51
|
};
|
|
52
52
|
}
|
|
53
53
|
|
|
54
|
+
/**
|
|
55
|
+
* Non-token media rates published by LiteLLM: USD per generated image, USD per
|
|
56
|
+
* generated video second (optionally per resolution, `output_cost_per_second_<res>`),
|
|
57
|
+
* and USD/M for image / video output tokens, which bill above the text output rate.
|
|
58
|
+
*/
|
|
59
|
+
export function litellmMediaPricing(entry) {
|
|
60
|
+
const perM = (value) => (validRate(value) ? value * 1_000_000 : null);
|
|
61
|
+
const byResolution = {};
|
|
62
|
+
for (const [key, value] of Object.entries(entry || {})) {
|
|
63
|
+
const match = key.match(/^output_cost_per_second_(.+)$/);
|
|
64
|
+
if (match && validRate(value)) byResolution[match[1].toLowerCase()] = value;
|
|
65
|
+
}
|
|
66
|
+
return {
|
|
67
|
+
outputImageCostPerM: perM(entry?.output_cost_per_image_token),
|
|
68
|
+
outputVideoCostPerM: perM(entry?.output_cost_per_video_token),
|
|
69
|
+
outputCostPerImage: validRate(entry?.output_cost_per_image) ? entry.output_cost_per_image : null,
|
|
70
|
+
outputCostPerSecond: validRate(entry?.output_cost_per_second) ? entry.output_cost_per_second : null,
|
|
71
|
+
outputCostPerSecondByResolution: byResolution,
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
|
|
54
75
|
export function modelsDevPricing(cost) {
|
|
55
76
|
// Structured tiers supersede the older context_over_200k compatibility
|
|
56
77
|
// field; it can coexist with a tier whose actual boundary is not 200k.
|
|
@@ -4,6 +4,13 @@ import { withUsageContext } from './usage-context.mjs';
|
|
|
4
4
|
import { ACCOUNT_PROVIDERS } from '../provider-accounts.mjs';
|
|
5
5
|
import { currentProviderAccountId } from '../provider-auth-binding.mjs';
|
|
6
6
|
|
|
7
|
+
// Results and errors already written. A provider-local re-send (model
|
|
8
|
+
// fallback, catalog retry) runs its own accounting and its result or error
|
|
9
|
+
// then returns through the enclosing send, which must not write it again.
|
|
10
|
+
const recorded = new WeakSet();
|
|
11
|
+
const hasTokens = (usage) =>
|
|
12
|
+
['inputTokens', 'outputTokens', 'cachedTokens', 'cacheWriteTokens'].some((key) => Number(usage?.[key]) > 0);
|
|
13
|
+
|
|
7
14
|
/**
|
|
8
15
|
* Runs at the common provider boundary, not inside optional diagnostic IO.
|
|
9
16
|
* Provider-local retries remain owned by the provider. Accounting failure must
|
|
@@ -12,7 +19,9 @@ import { currentProviderAccountId } from '../provider-auth-binding.mjs';
|
|
|
12
19
|
export async function accountProviderSend(provider, instance, send, model, opts = {}) {
|
|
13
20
|
const requestId = randomUUID();
|
|
14
21
|
const startedAt = Date.now();
|
|
15
|
-
|
|
22
|
+
// usageSessionId: a request isolated under its own provider session id
|
|
23
|
+
// (compaction's `:compact`) whose spend belongs to the source session.
|
|
24
|
+
const sessionId = opts.usageSessionId || opts.sessionId || opts.session?.id;
|
|
16
25
|
const sourceType = opts.session?.sourceType || opts.sourceType || opts.requestKind || '';
|
|
17
26
|
const inputTokensInclusive = instance.constructor?.inputExcludesCache !== true;
|
|
18
27
|
let ledger;
|
|
@@ -30,16 +39,18 @@ export async function accountProviderSend(provider, instance, send, model, opts
|
|
|
30
39
|
sourceType,
|
|
31
40
|
inputTokensInclusive,
|
|
32
41
|
};
|
|
33
|
-
const record = async (result) => {
|
|
42
|
+
const record = async (result, id, owner) => {
|
|
34
43
|
if (!result?.usage) return;
|
|
35
44
|
// A nested send (e.g. a fallback model re-send) already stamped its own
|
|
36
45
|
// final attempt's tier; the outer context only saw the abandoned attempt.
|
|
37
46
|
result.requestServiceTier ??= identity.requestServiceTier || '';
|
|
47
|
+
if (recorded.has(owner)) return;
|
|
38
48
|
if (openingError) throw openingError;
|
|
39
49
|
if (!ledger) return;
|
|
50
|
+
recorded.add(owner);
|
|
40
51
|
const usage = result.usage;
|
|
41
52
|
const row = makeUsageRecord({
|
|
42
|
-
id: result.responseId ? undefined :
|
|
53
|
+
id: result.responseId ? undefined : id,
|
|
43
54
|
ts: Date.now(),
|
|
44
55
|
provider,
|
|
45
56
|
model: result.model || model,
|
|
@@ -68,23 +79,32 @@ export async function accountProviderSend(provider, instance, send, model, opts
|
|
|
68
79
|
// SQLite write no longer runs on the event loop.
|
|
69
80
|
await ledger.recordQueued(row);
|
|
70
81
|
};
|
|
71
|
-
const save = async (result) => {
|
|
82
|
+
const save = async (result, id = requestId, owner = result) => {
|
|
72
83
|
try {
|
|
73
|
-
await record(result);
|
|
84
|
+
await record(result, id, owner);
|
|
74
85
|
} catch (error) {
|
|
75
86
|
result.usageAccountingError = String(error?.message || error);
|
|
76
87
|
process.stderr.write(`[usage-ledger] RECORD NOT SAVED: ${result.usageAccountingError}\n`);
|
|
77
88
|
}
|
|
78
89
|
};
|
|
90
|
+
const saveAbandoned = async () => {
|
|
91
|
+
for (const [index, attempt] of (identity.abandonedUsage || []).entries()) {
|
|
92
|
+
if (hasTokens(attempt.usage)) await save(attempt, `${requestId}:abandoned:${index}`);
|
|
93
|
+
}
|
|
94
|
+
};
|
|
79
95
|
let result;
|
|
80
96
|
try {
|
|
81
97
|
result = await withUsageContext(identity, send);
|
|
82
98
|
} catch (error) {
|
|
99
|
+
await saveAbandoned();
|
|
83
100
|
// Only provider-reported partial usage is recordable; never invent
|
|
84
101
|
// tokens for a failed request or reinterpret an error as a success.
|
|
85
102
|
if (error?.usage) await save(error);
|
|
103
|
+
else if (hasTokens(error?.partialUsage))
|
|
104
|
+
await save({ usage: error.partialUsage, model: error.partialModel }, requestId, error);
|
|
86
105
|
throw error;
|
|
87
106
|
}
|
|
107
|
+
await saveAbandoned();
|
|
88
108
|
await save(result);
|
|
89
109
|
return result;
|
|
90
110
|
}
|
|
@@ -11,3 +11,9 @@ export const noteRequestServiceTier = (tier) => {
|
|
|
11
11
|
const identity = context.getStore();
|
|
12
12
|
if (identity) identity.requestServiceTier = tier || '';
|
|
13
13
|
};
|
|
14
|
+
// A provider-local retry abandons an attempt the provider still billed; the
|
|
15
|
+
// enclosing send records it alongside its own final usage.
|
|
16
|
+
export const noteAbandonedUsage = (usage, model) => {
|
|
17
|
+
const identity = context.getStore();
|
|
18
|
+
if (identity && usage) (identity.abandonedUsage ||= []).push({ usage, model });
|
|
19
|
+
};
|
|
@@ -43,6 +43,7 @@ export function importTraceRow(row) {
|
|
|
43
43
|
uncachedInputTokens: raw ? (row.uncached_input_tokens ?? payload.uncached_input_tokens) : undefined,
|
|
44
44
|
cacheReadTokens: raw ? row.cached_tokens : row.cacheReadTokens,
|
|
45
45
|
cacheWriteTokens: raw ? row.cache_write_tokens : row.cacheWriteTokens,
|
|
46
|
+
cacheWrite1hTokens: raw ? payload.raw_usage?.cache_creation?.ephemeral_1h_input_tokens : undefined,
|
|
46
47
|
costUsd: raw ? undefined : row.costUsd,
|
|
47
48
|
sessionId: row.session_id || row.sessionId,
|
|
48
49
|
sourceType: row.sourceType || row.source_type || payload.sourceType || payload.source_type,
|