@ossclip/core 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,56 @@
1
+ import type { z } from "zod/v4";
2
+ import type { CallTier, LlmProvider } from "./provider";
3
+ import type { LlmUsage } from "./usage";
4
+
5
+ /**
6
+ * Routes each call to a model sized for the thinking it needs (FINDINGS §37).
7
+ *
8
+ * Producing a video is one editorial call — the beat sheet, where the hook and
9
+ * the segmentation are decided — and a handful of mechanical ones that fill a
10
+ * schema or repair a mishearing. Measured on the CLI path, a call costs ~$0.26
11
+ * on the top model against ~$0.04 on the small one, and since the harness
12
+ * prefix dominates the token count either way, the model is the only per-call
13
+ * lever left once the calls themselves have been batched down.
14
+ *
15
+ * A wrapper rather than a flag inside each provider: the providers stay dumb
16
+ * about policy, and any two of them compose — including two of DIFFERENT kinds,
17
+ * which is what lets a subscription CLI do the editorial call while a metered
18
+ * flash model does the mechanical ones.
19
+ */
20
+ export class TieredProvider implements LlmProvider {
21
+ readonly name: string;
22
+ private readonly records: LlmUsage[] = [];
23
+
24
+ constructor(
25
+ private readonly editorial: LlmProvider,
26
+ private readonly mechanical: LlmProvider,
27
+ ) {
28
+ this.name =
29
+ editorial === mechanical ? editorial.name : `${editorial.name}+${mechanical.name}`;
30
+ }
31
+
32
+ /** Merged in call order — the sub-providers each keep their own log. */
33
+ get usage(): readonly LlmUsage[] {
34
+ return this.records;
35
+ }
36
+
37
+ async complete<T>(req: {
38
+ system: string;
39
+ user: string;
40
+ schema: z.ZodType<T>;
41
+ schemaName: string;
42
+ maxTokens?: number;
43
+ tier?: CallTier;
44
+ }): Promise<T> {
45
+ const target = req.tier === "mechanical" ? this.mechanical : this.editorial;
46
+ const before = target.usage.length;
47
+ try {
48
+ return await target.complete(req);
49
+ } finally {
50
+ // Drain whatever the sub-provider logged, including for a call that
51
+ // threw: a failed attempt still spent the tokens, and §37's whole point
52
+ // is that the bill is visible rather than quietly absorbed.
53
+ this.records.push(...target.usage.slice(before));
54
+ }
55
+ }
56
+ }
@@ -0,0 +1,426 @@
1
+ /**
2
+ * Token and cost accounting for the producer's LLM calls.
3
+ *
4
+ * Producing a video is many calls — a repair pass, a beat sheet, and one per
5
+ * scene — so "what did that cost" is a real question with a non-obvious
6
+ * answer. Every provider records one entry per call into its own `usage`
7
+ * array; the CLI prices them at the end and prints a per-call-type breakdown
8
+ * plus a total.
9
+ *
10
+ * Providers record TOKENS, not money. Pricing lives here alone, so a price
11
+ * change is one table edit rather than four provider edits, and so a provider
12
+ * that reports an authoritative cost of its own (the Claude CLI does) can
13
+ * hand it over without every other provider growing a pricing dependency.
14
+ *
15
+ * Two honesty rules, because a wrong number is worse than no number:
16
+ * - Tokens a provider actually reports are marked exact; anything derived
17
+ * from text length is marked an estimate and says so in the output.
18
+ * - Prices change and vary by model. The table below is the DEFAULT
19
+ * assumption by model family, overridable in `~/.ossclip/config.json`;
20
+ * a model that matches nothing is reported with tokens and no cost
21
+ * rather than a guess.
22
+ */
23
+
24
+ export interface LlmUsage {
25
+ provider: string;
26
+ model?: string;
27
+ /** Which call this was — `beat_sheet`, `transcript_repair`, `StatCard_props`… */
28
+ schemaName: string;
29
+ inputTokens: number;
30
+ outputTokens: number;
31
+ /** Input tokens served from cache, when the provider distinguishes them. */
32
+ cachedInputTokens?: number;
33
+ /**
34
+ * A cost the provider stated itself, in USD. Authoritative when present —
35
+ * it beats the local price table, which is only ever an assumption.
36
+ */
37
+ reportedCostUsd?: number;
38
+ /** False when tokens were derived from text length rather than reported. */
39
+ exact: boolean;
40
+ /**
41
+ * False when the call does not bill per token — a subscription CLI or the
42
+ * offline mock. The cost is still worth showing (it says what the same work
43
+ * would cost on the API), it just isn't a charge.
44
+ */
45
+ billed: boolean;
46
+ /** Wall-clock milliseconds, so a slow call is visible beside a costly one. */
47
+ ms?: number;
48
+ }
49
+
50
+ export interface ModelPrice {
51
+ inputPerMTok: number;
52
+ outputPerMTok: number;
53
+ }
54
+
55
+ /**
56
+ * Default price assumptions, in USD per million tokens, keyed by the model
57
+ * FAMILY found in the model id. Family rather than exact id because ids carry
58
+ * dates and revisions that change more often than the tier's pricing does.
59
+ *
60
+ * These are assumptions, not quotes: override them per model or per family in
61
+ * `~/.ossclip/config.json` under `pricing` if your account differs.
62
+ */
63
+ export const DEFAULT_PRICING: Record<string, ModelPrice> = {
64
+ // Opus repriced at 4.5: $5/$25 (was $15/$75 through 4.1), unchanged through
65
+ // Opus 5 (verified 2026-07, R21 §104). An account still running a pre-4.5
66
+ // opus should override this in config.json.
67
+ opus: { inputPerMTok: 5, outputPerMTok: 25 },
68
+ sonnet: { inputPerMTok: 3, outputPerMTok: 15 },
69
+ haiku: { inputPerMTok: 1, outputPerMTok: 5 },
70
+ "gemini-2.5-pro": { inputPerMTok: 1.25, outputPerMTok: 10 },
71
+ "gemini-2.5-flash": { inputPerMTok: 0.3, outputPerMTok: 2.5 },
72
+ "gemini-3.6-flash": { inputPerMTok: 1.5, outputPerMTok: 7.5 },
73
+ "gemini-3.5-flash-lite": { inputPerMTok: 0.3, outputPerMTok: 2.5 },
74
+ };
75
+
76
+ /** Resolve a price for a model id: exact override, override family, then default family. */
77
+ export function priceFor(
78
+ model: string | undefined,
79
+ overrides: Record<string, ModelPrice> = {},
80
+ ): ModelPrice | null {
81
+ if (!model) return null;
82
+ if (overrides[model]) return overrides[model]!;
83
+ const id = model.toLowerCase();
84
+ // Longest key first so `gemini-2.5-flash` wins over a hypothetical `gemini`.
85
+ const byLength = (t: Record<string, ModelPrice>): Array<[string, ModelPrice]> =>
86
+ Object.entries(t).sort((a, b) => b[0].length - a[0].length);
87
+ for (const [family, price] of byLength(overrides)) {
88
+ if (id.includes(family.toLowerCase())) return price;
89
+ }
90
+ for (const [family, price] of byLength(DEFAULT_PRICING)) {
91
+ if (id.includes(family)) return price;
92
+ }
93
+ return null;
94
+ }
95
+
96
+ /**
97
+ * What one call cost, and how confidently.
98
+ *
99
+ * `reported` is the provider's own number; `table` is priced locally from
100
+ * tokens; `unknown` means the model matches no price and we decline to guess.
101
+ *
102
+ * `inputTokens` counts every input token however it was served, including
103
+ * cached ones, and all of them are priced at the full input rate. Cache reads
104
+ * really cost a fraction of that, so a run with cache hits is an OVER-estimate
105
+ * — the safe direction for a number a user might budget against, and the
106
+ * report says so whenever cached tokens are non-zero.
107
+ */
108
+ export function costOf(
109
+ record: LlmUsage,
110
+ pricing: Record<string, ModelPrice> = {},
111
+ ): { usd: number | null; source: "reported" | "table" | "unknown" } {
112
+ if (record.reportedCostUsd !== undefined) {
113
+ return { usd: record.reportedCostUsd, source: "reported" };
114
+ }
115
+ const price = priceFor(record.model, pricing);
116
+ if (!price) return { usd: null, source: "unknown" };
117
+ return {
118
+ usd:
119
+ (record.inputTokens * price.inputPerMTok + record.outputTokens * price.outputPerMTok) /
120
+ 1_000_000,
121
+ source: "table",
122
+ };
123
+ }
124
+
125
+ /**
126
+ * Rough token count for providers that do not report one. Deliberately crude —
127
+ * ~4 characters per token is the usual English approximation — and always
128
+ * surfaced as an estimate rather than passed off as measured.
129
+ */
130
+ export function estimateTokens(text: string): number {
131
+ return Math.max(1, Math.ceil(text.length / 4));
132
+ }
133
+
134
+ export interface UsageTotals {
135
+ calls: number;
136
+ inputTokens: number;
137
+ outputTokens: number;
138
+ cachedInputTokens: number;
139
+ ms: number;
140
+ /** USD actually charged. Null when a billed call had no known price. */
141
+ billedUsd: number | null;
142
+ /**
143
+ * USD the same tokens would cost at API rates, INCLUDING subscription calls.
144
+ * This is the number that makes a Claude Max run legible: the plan covers
145
+ * it, but it still says how much work the run represents.
146
+ */
147
+ equivalentUsd: number | null;
148
+ /** True when any record's tokens were estimated rather than reported. */
149
+ anyEstimated: boolean;
150
+ /** True when no call billed per token (subscription or offline). */
151
+ allUnbilled: boolean;
152
+ /** Models seen with no price, so the report can name them. */
153
+ unpricedModels: string[];
154
+ }
155
+
156
+ export function summarizeUsage(
157
+ records: readonly LlmUsage[],
158
+ pricing: Record<string, ModelPrice> = {},
159
+ ): UsageTotals {
160
+ const unpriced = new Set<string>();
161
+ let billed: number | null = 0;
162
+ let equivalent: number | null = 0;
163
+ let input = 0;
164
+ let output = 0;
165
+ let cached = 0;
166
+ let ms = 0;
167
+ let estimated = false;
168
+ let billedAny = false;
169
+
170
+ for (const r of records) {
171
+ input += r.inputTokens;
172
+ output += r.outputTokens;
173
+ cached += r.cachedInputTokens ?? 0;
174
+ ms += r.ms ?? 0;
175
+ if (!r.exact) estimated = true;
176
+ const { usd } = costOf(r, pricing);
177
+ if (usd === null) {
178
+ unpriced.add(r.model ?? r.provider);
179
+ equivalent = null;
180
+ if (r.billed) billed = null;
181
+ } else {
182
+ if (equivalent !== null) equivalent += usd;
183
+ if (r.billed && billed !== null) billed += usd;
184
+ }
185
+ if (r.billed) billedAny = true;
186
+ }
187
+
188
+ return {
189
+ calls: records.length,
190
+ inputTokens: input,
191
+ outputTokens: output,
192
+ cachedInputTokens: cached,
193
+ ms,
194
+ billedUsd: billedAny ? billed : 0,
195
+ equivalentUsd: equivalent,
196
+ anyEstimated: estimated,
197
+ allUnbilled: !billedAny,
198
+ unpricedModels: [...unpriced],
199
+ };
200
+ }
201
+
202
+ const n = (v: number): string => v.toLocaleString("en-US");
203
+ const usd = (v: number): string => (v < 0.01 ? `$${v.toFixed(4)}` : `$${v.toFixed(2)}`);
204
+ const secs = (msTotal: number): string =>
205
+ msTotal >= 1000 ? `${Math.round(msTotal / 1000)}s` : `${msTotal}ms`;
206
+
207
+ /** One line for the console: the answer to "what did that cost". */
208
+ export function formatUsageLine(
209
+ records: readonly LlmUsage[],
210
+ pricing: Record<string, ModelPrice> = {},
211
+ ): string {
212
+ const t = summarizeUsage(records, pricing);
213
+ if (t.calls === 0) return "▸ llm: no calls";
214
+ // A bare input total invites the wrong conclusion. On the CLI path ~99% of
215
+ // it is the harness prefix re-sent per call, not anything ossclip wrote —
216
+ // "270k in" reads like a runaway prompt when the real lever is call count.
217
+ const cachedShare = t.inputTokens > 0 ? t.cachedInputTokens / t.inputTokens : 0;
218
+ const inPart =
219
+ cachedShare >= 0.5
220
+ ? `${n(t.inputTokens)} in (${Math.round(cachedShare * 100)}% cached prefix) / ${n(t.outputTokens)} out tokens`
221
+ : `${n(t.inputTokens)} in / ${n(t.outputTokens)} out tokens`;
222
+ const parts = [`${t.calls} calls`, inPart];
223
+ if (t.equivalentUsd === null) {
224
+ parts.push(
225
+ `cost unknown for ${t.unpricedModels.join(", ")} — set \`pricing\` in ~/.ossclip/config.json`,
226
+ );
227
+ } else if (t.allUnbilled) {
228
+ // Zero equivalent means an offline provider, not a generous plan — saying
229
+ // "covered by the subscription" there would be nonsense.
230
+ parts.push(
231
+ t.equivalentUsd === 0
232
+ ? "no charge (offline provider)"
233
+ : `~${usd(t.equivalentUsd)} of API-rate work, covered by the subscription`,
234
+ );
235
+ } else {
236
+ parts.push(`~${usd(t.billedUsd ?? 0)}`);
237
+ }
238
+ if (t.ms > 0) parts.push(secs(t.ms));
239
+ return `▸ llm: ${parts.join(" · ")}${t.anyEstimated ? " (tokens partly estimated)" : ""}`;
240
+ }
241
+
242
+ interface Grouped {
243
+ schemaName: string;
244
+ calls: number;
245
+ inputTokens: number;
246
+ outputTokens: number;
247
+ usd: number | null;
248
+ exact: boolean;
249
+ ms: number;
250
+ }
251
+
252
+ /** Collapse the per-scene calls into one row per call TYPE, in first-seen order. */
253
+ export function groupUsage(
254
+ records: readonly LlmUsage[],
255
+ pricing: Record<string, ModelPrice> = {},
256
+ ): Grouped[] {
257
+ const rows = new Map<string, Grouped>();
258
+ for (const r of records) {
259
+ const row = rows.get(r.schemaName) ?? {
260
+ schemaName: r.schemaName,
261
+ calls: 0,
262
+ inputTokens: 0,
263
+ outputTokens: 0,
264
+ usd: 0 as number | null,
265
+ exact: true,
266
+ ms: 0,
267
+ };
268
+ const { usd: cost } = costOf(r, pricing);
269
+ row.calls += 1;
270
+ row.inputTokens += r.inputTokens;
271
+ row.outputTokens += r.outputTokens;
272
+ row.usd = cost === null || row.usd === null ? null : row.usd + cost;
273
+ row.exact &&= r.exact;
274
+ row.ms += r.ms ?? 0;
275
+ rows.set(r.schemaName, row);
276
+ }
277
+ return [...rows.values()];
278
+ }
279
+
280
+ /** Per-call-type breakdown for report.txt — where the cost actually went. */
281
+ export function formatUsageReport(
282
+ records: readonly LlmUsage[],
283
+ pricing: Record<string, ModelPrice> = {},
284
+ ): string {
285
+ if (records.length === 0) return "";
286
+ const t = summarizeUsage(records, pricing);
287
+ const first = records[0]!;
288
+ // A column of $0.0000 says nothing; an offline run's cost column is a dash.
289
+ const free = t.allUnbilled && t.equivalentUsd === 0;
290
+ const money = (v: number | null): string => (free ? "—" : v === null ? "?" : usd(v));
291
+ const rows = groupUsage(records, pricing).map((g) => {
292
+ const name = g.calls > 1 ? `${g.schemaName} ×${g.calls}` : g.schemaName;
293
+ return (
294
+ ` ${name.padEnd(26)}${n(g.inputTokens).padStart(9)} in ${n(g.outputTokens).padStart(8)} out` +
295
+ `${money(g.usd).padStart(10)}${g.exact ? "" : " (est)"}`
296
+ );
297
+ });
298
+ const total = money(t.equivalentUsd);
299
+ const notes: string[] = [];
300
+ if (t.cachedInputTokens > 0) {
301
+ notes.push(
302
+ ` ${n(t.cachedInputTokens)} input tokens came from cache and are priced here at the\n` +
303
+ " full input rate, so the total above is an over-estimate.",
304
+ );
305
+ }
306
+ if (t.allUnbilled) {
307
+ notes.push(
308
+ t.equivalentUsd === 0
309
+ ? " Nothing was charged — this ran entirely offline."
310
+ : " Not billed per token — this ran on a subscription. The figures say what\n" +
311
+ " the same tokens would have cost at API rates.",
312
+ );
313
+ }
314
+ if (t.anyEstimated) {
315
+ notes.push(" Rows marked (est) had their tokens estimated — that provider reports none.");
316
+ }
317
+ if (t.unpricedModels.length > 0) {
318
+ notes.push(
319
+ ` No price known for ${t.unpricedModels.join(", ")} — set \`pricing\` in\n` +
320
+ " ~/.ossclip/config.json to price it.",
321
+ );
322
+ } else if (t.equivalentUsd !== 0) {
323
+ notes.push(" Prices are the built-in per-family assumption; override in ~/.ossclip/config.json.");
324
+ }
325
+ return (
326
+ `\nllm usage (${first.provider}${first.model ? ` · ${first.model}` : ""}` +
327
+ `${t.ms > 0 ? `, ${secs(t.ms)}` : ""}):\n` +
328
+ rows.join("\n") +
329
+ "\n" +
330
+ ` ${"TOTAL".padEnd(26)}${n(t.inputTokens).padStart(9)} in ${n(t.outputTokens).padStart(8)} out${total.padStart(10)}\n` +
331
+ notes.join("\n") +
332
+ "\n"
333
+ );
334
+ }
335
+
336
+ /** One produce run's entry in the workdir's usage log. */
337
+ export interface UsageRun {
338
+ /** ISO timestamp, supplied by the caller so this stays pure. */
339
+ at: string;
340
+ /** Resolved provider, or null when a run made no calls and none is known. */
341
+ provider: string | null;
342
+ /** Models seen this run, first-seen order — the tiering, visible. */
343
+ models: string[];
344
+ /** True when the run made no calls: everything came from the workdir cache. */
345
+ cached: boolean;
346
+ records: LlmUsage[];
347
+ totals: UsageTotals;
348
+ }
349
+
350
+ /**
351
+ * The workdir's usage log (R16 §78).
352
+ *
353
+ * `records`/`totals` describe ONE run, and every run rewrote them — so a
354
+ * fully-cached re-run (zero calls, `records: []`) erased the provenance of
355
+ * the run that actually did the planning. Two real workdirs were left saying
356
+ * nothing about which provider chose their scenes.
357
+ *
358
+ * So the file grows a `runs` history, and the top-level `records`/`totals`
359
+ * now hold the last run THAT MADE CALLS rather than simply the last run —
360
+ * every existing reader keeps working and stops being lied to. The whole
361
+ * shape is optional on read: a pre-§78 file is a valid input.
362
+ */
363
+ export interface UsageLog {
364
+ runs: UsageRun[];
365
+ /** Last run that made calls — NOT necessarily the last run. */
366
+ records: LlmUsage[];
367
+ totals: UsageTotals;
368
+ }
369
+
370
+ /** The provider a log knows about, newest run with calls first. */
371
+ export function providerOfLog(log: Pick<UsageLog, "runs" | "records">): string | null {
372
+ for (let i = log.runs.length - 1; i >= 0; i--) {
373
+ const run = log.runs[i]!;
374
+ if (!run.cached && run.provider) return run.provider;
375
+ }
376
+ return log.records[0]?.provider ?? null;
377
+ }
378
+
379
+ /** Models the log last saw in a run that made calls. */
380
+ function modelsOfLog(log: Pick<UsageLog, "runs" | "records">): string[] {
381
+ for (let i = log.runs.length - 1; i >= 0; i--) {
382
+ const run = log.runs[i]!;
383
+ if (!run.cached && run.models.length > 0) return [...run.models];
384
+ }
385
+ const seen: string[] = [];
386
+ for (const r of log.records) {
387
+ if (r.model && !seen.includes(r.model)) seen.push(r.model);
388
+ }
389
+ return seen;
390
+ }
391
+
392
+ export function appendUsageRun(
393
+ previous: unknown,
394
+ run: { at: string; records: readonly LlmUsage[]; provider?: string | null },
395
+ pricing: Record<string, ModelPrice> = {},
396
+ ): UsageLog {
397
+ const prev = (previous ?? {}) as Partial<UsageLog>;
398
+ const prevRuns = Array.isArray(prev.runs) ? prev.runs : [];
399
+ const prevRecords = Array.isArray(prev.records) ? prev.records : [];
400
+ const records = [...run.records];
401
+ const cached = records.length === 0;
402
+ const models: string[] = [];
403
+ for (const r of records) {
404
+ if (r.model && !models.includes(r.model)) models.push(r.model);
405
+ }
406
+ // A cached run inherits the models it is REUSING the output of, exactly as
407
+ // it inherits the provider — otherwise the stamp on a cached production
408
+ // names a provider with no models beside it.
409
+ const inheritedModels = cached
410
+ ? modelsOfLog({ runs: prevRuns, records: prevRecords })
411
+ : models;
412
+ // A cached run still names the provider it INHERITS, so the history reads
413
+ // as a continuous account rather than a gap.
414
+ const provider =
415
+ records[0]?.provider ??
416
+ run.provider ??
417
+ providerOfLog({ runs: prevRuns, records: prevRecords });
418
+ const totals = summarizeUsage(records, pricing);
419
+ const entry: UsageRun = { at: run.at, provider, models: inheritedModels, cached, records, totals };
420
+ return {
421
+ runs: [...prevRuns, entry],
422
+ // Never overwrite a real accounting with an empty one.
423
+ records: cached ? prevRecords : records,
424
+ totals: cached && prevRecords.length > 0 ? summarizeUsage(prevRecords, pricing) : totals,
425
+ };
426
+ }
package/src/report.ts ADDED
@@ -0,0 +1,36 @@
1
+ import type { Production } from "./schema";
2
+ import { TimeMap } from "./timemap";
3
+
4
+ function fmt(t: number): string {
5
+ const m = Math.floor(t / 60);
6
+ const s = (t % 60).toFixed(2).padStart(5, "0");
7
+ return `${String(m).padStart(2, "0")}:${s}`;
8
+ }
9
+
10
+ /** Human-readable account of every cut: what, where, why, and how much. */
11
+ export function formatCutReport(production: Production): string {
12
+ const cutlist = production.cutlist ?? [];
13
+ const removals = cutlist.filter((s) => s.kind === "remove");
14
+ const map = new TimeMap(cutlist);
15
+ const srcDur = production.source.probe.duration;
16
+ const lines: string[] = [];
17
+ lines.push(`ossclip cut report — cleanup level: ${production.cleanup}`);
18
+ lines.push(`source: ${production.source.path}`);
19
+ lines.push("");
20
+ if (removals.length === 0) {
21
+ lines.push("Nothing removed — the take plays exactly as recorded.");
22
+ } else {
23
+ for (const r of removals) {
24
+ const dur = (r.srcOut - r.srcIn).toFixed(2);
25
+ const conf = r.confidence !== undefined ? ` (conf ${r.confidence.toFixed(2)})` : "";
26
+ lines.push(` [${fmt(r.srcIn)} → ${fmt(r.srcOut)}] ${r.reason ?? "?"} −${dur}s${conf}`);
27
+ }
28
+ }
29
+ const removed = srcDur - map.outputDuration;
30
+ lines.push("");
31
+ lines.push(
32
+ `${removals.length} cut(s) · removed ${removed.toFixed(2)}s of ${srcDur.toFixed(2)}s ` +
33
+ `(${((removed / srcDur) * 100).toFixed(1)}%) · output ${map.outputDuration.toFixed(2)}s`,
34
+ );
35
+ return lines.join("\n");
36
+ }