@ossclip/core 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,73 @@
1
+ import Anthropic from "@anthropic-ai/sdk";
2
+ import { zodOutputFormat } from "@anthropic-ai/sdk/helpers/zod";
3
+ import type { z } from "zod/v4";
4
+ import type { LlmProvider } from "./provider";
5
+ import type { LlmUsage } from "./usage";
6
+
7
+ export const DEFAULT_CLAUDE_MODEL = "claude-opus-5";
8
+
9
+ /**
10
+ * Claude via the official SDK with schema-constrained structured output.
11
+ * Auth: ANTHROPIC_API_KEY env (SDK default resolution).
12
+ */
13
+ export class AnthropicProvider implements LlmProvider {
14
+ readonly name = "claude";
15
+ readonly usage: LlmUsage[] = [];
16
+ private client: Anthropic;
17
+
18
+ constructor(
19
+ private model: string = DEFAULT_CLAUDE_MODEL,
20
+ apiKey?: string,
21
+ ) {
22
+ this.client = new Anthropic(apiKey ? { apiKey } : {});
23
+ }
24
+
25
+ async complete<T>(req: {
26
+ system: string;
27
+ user: string;
28
+ schema: z.ZodType<T>;
29
+ schemaName: string;
30
+ maxTokens?: number;
31
+ }): Promise<T> {
32
+ const started = Date.now();
33
+ const response = await this.client.messages.parse({
34
+ model: this.model,
35
+ max_tokens: req.maxTokens ?? 16000,
36
+ system: req.system,
37
+ messages: [{ role: "user", content: req.user }],
38
+ output_config: { format: zodOutputFormat(req.schema) },
39
+ });
40
+ // Recorded before the failure checks: a refusal or a truncation still
41
+ // consumed tokens, and a run that fell over is exactly when the cost of
42
+ // getting there matters.
43
+ this.usage.push({
44
+ provider: this.name,
45
+ model: response.model ?? this.model,
46
+ schemaName: req.schemaName,
47
+ // The API reports cached tokens OUTSIDE `input_tokens`; ossclip's
48
+ // `inputTokens` means every input token however it was served, so they
49
+ // are folded in here and also reported separately for the discount note.
50
+ inputTokens:
51
+ response.usage.input_tokens +
52
+ (response.usage.cache_read_input_tokens ?? 0) +
53
+ (response.usage.cache_creation_input_tokens ?? 0),
54
+ outputTokens: response.usage.output_tokens,
55
+ cachedInputTokens:
56
+ (response.usage.cache_read_input_tokens ?? 0) +
57
+ (response.usage.cache_creation_input_tokens ?? 0),
58
+ exact: true,
59
+ billed: true,
60
+ ms: Date.now() - started,
61
+ });
62
+ if (response.stop_reason === "refusal") {
63
+ throw new Error(`claude declined the request (${response.stop_details?.category ?? "unspecified"})`);
64
+ }
65
+ if (response.stop_reason === "max_tokens") {
66
+ throw new Error("claude output truncated at max_tokens");
67
+ }
68
+ if (response.parsed_output == null) {
69
+ throw new Error("claude returned no parseable structured output");
70
+ }
71
+ return response.parsed_output as T;
72
+ }
73
+ }
@@ -0,0 +1,330 @@
1
+ import { z } from "zod/v4";
2
+ import type { Transcript } from "../schema";
3
+ import { LayoutSchema, SceneComponentIdSchema } from "../scene-schema";
4
+ import { SCENE_REGISTRY } from "../scene-registry";
5
+ import { MAX_SCENE_SEC } from "../assemble";
6
+ import { COVER_MAX_WORDS, coverHeadline } from "../cover";
7
+ import type { LlmProvider } from "./provider";
8
+
9
+ /** Call 1 — the editorial call (PHASE1 §4): moments, copy, component picks. */
10
+ export const MomentSchema = z.object({
11
+ startWord: z.number().int().nonnegative(),
12
+ endWord: z.number().int().nonnegative(),
13
+ purpose: z.string().max(100),
14
+ /** Short on-screen copy for this beat — the fallback TitleCard title. */
15
+ onScreenCopy: z.string().max(60),
16
+ /** "none" = plain talking head with captions; otherwise a library component. */
17
+ sceneKind: z.union([SceneComponentIdSchema, z.literal("none")]),
18
+ /**
19
+ * Stage layout for this scene (PLAN Task B). Optional so an older cached
20
+ * beat sheet still parses; omitted means the component's registry default.
21
+ * Feasibility is enforced by `repairMomentLayouts` either way — the schema
22
+ * carries the request, the repair pass is the constraint (§35's lesson).
23
+ */
24
+ layout: LayoutSchema.optional().describe(
25
+ "stage layout for this scene; omit for the component default. NEVER a layout the framing brief marks UNAVAILABLE for these words",
26
+ ),
27
+ rationale: z.string().max(120).optional(),
28
+ });
29
+ export type Moment = z.infer<typeof MomentSchema>;
30
+
31
+ export const BeatSheetSchema = z.object({
32
+ hook: z.string().max(120),
33
+ /**
34
+ * Banner text for the cover image (FINDINGS §31). Written here rather than
35
+ * by a second LLM call, because the producer is already choosing the hook —
36
+ * this is the same editorial judgement, shortened for a thumbnail.
37
+ */
38
+ coverText: z
39
+ .string()
40
+ .max(60)
41
+ .optional()
42
+ .describe(
43
+ `cover banner: at most ${COVER_MAX_WORDS} words, the hook compressed to a thumbnail headline`,
44
+ ),
45
+ moments: z.array(MomentSchema).min(1).max(12),
46
+ });
47
+ export type BeatSheet = z.infer<typeof BeatSheetSchema>;
48
+
49
+ /**
50
+ * The `--clip` highlight request (R19 §93d): asked for IN THE SAME editorial
51
+ * call as the beat sheet — the producer is already reading the whole
52
+ * transcript and ranking moments, so the window costs approximately nothing
53
+ * here, and a second call would let two editorial judgements disagree.
54
+ */
55
+ export const ClipHighlightSchema = z.object({
56
+ startWord: z.number().int().nonnegative(),
57
+ endWord: z.number().int().nonnegative(),
58
+ reason: z
59
+ .string()
60
+ .max(200)
61
+ .describe("one line: why THIS window is the strongest stretch of the take"),
62
+ });
63
+ export type ClipHighlight = z.infer<typeof ClipHighlightSchema>;
64
+
65
+ export const ClipBeatSheetSchema = BeatSheetSchema.extend({
66
+ highlight: ClipHighlightSchema.describe(
67
+ "the single contiguous window to produce — every moment above must lie inside it",
68
+ ),
69
+ });
70
+
71
+ export const PRODUCER_SYSTEM = `You are the producer for a short-form vertical video (Reels/Shorts/TikTok). You receive a word-indexed transcript of a talking-head take that has already been cut. Your job is EDITORIAL: segment the take into moments, pick which moments deserve a graphic scene, and write the on-screen copy.
72
+
73
+ Virality grammar — follow these as hard policies:
74
+ - The first moment is the hook: the strongest claim or number anywhere in the take, on screen within 2 seconds.
75
+ - A pattern interrupt every 3-6 seconds: alternate graphic moments with plain talking-head moments ("none").
76
+ - On-screen copy is SHORT: numbers over adjectives, verbs over descriptions, never full sentences.
77
+ - Use contrast/negation beats (StrikethroughReveal, RuleCard with struck alternatives) when the speaker rejects an idea.
78
+ - End with a payoff or takeaway moment.
79
+ - Moments must be contiguous-ish spans of the transcript, 5-10 seconds of speech each, in transcript order, non-overlapping.
80
+ - COVERAGE: graphics should be on screen for roughly 40-50% of the runtime. Each graphic holds at most ~5 seconds, then hands the frame back — so MOST moments can carry one. Spread them evenly: never leave a stretch longer than ~10 seconds with no graphic.
81
+ - VARIETY: never the same component twice in a row, and prefer a component you have NOT used yet in this video — reuse a treatment only when the beat genuinely calls for it. A repeat reads as a template.
82
+ - Keep the face LARGE: prefer StatCard/RuleCard/ScreenshotFrame (they sit under a big face) over TitleCard (face becomes a small bubble); use FlowDiagram/TerminalMock sparingly — they remove the face entirely and only earn that when the graphic IS the point.
83
+ - The transcript is ASR output and may contain mishearings: an unfamiliar proper noun is more likely a mistranscription of a common phrase than a real entity — write on-screen copy with the common-sense reading, never a suspected mishearing.
84
+ - FRAMING: when the prompt carries a "Camera framing" brief, it is measured from the footage and is a HARD constraint: on words marked CLOSE, never choose a layout listed as UNAVAILABLE there — pick a \`layout\` that keeps the whole head in frame (pip-bubble, graphic-only, full-bleed) or leave the moment as "none". You may set \`layout\` on any moment; omit it to accept the component's default.
85
+ - COVER: also write \`coverText\` — the hook compressed to a thumbnail headline, AT MOST ${COVER_MAX_WORDS} WORDS. It is read at a glance in a profile grid, so it must stand alone without the video: the claim or the number, no lead-in, no ellipsis.`;
86
+
87
+ /**
88
+ * The `--clip` request, appended to the USER prompt (R19 §93d). The tuned
89
+ * PRODUCER_SYSTEM stays untouched — this adds the window request without
90
+ * rewriting the editorial instructions the beat sheet already follows.
91
+ */
92
+ export function buildClipAddendum(targetSec: number): string {
93
+ return (
94
+ `\n\nCLIP SELECTION: the source is long-form, and only ONE window of roughly ` +
95
+ `${targetSec.toFixed(0)} seconds will be produced.\n` +
96
+ `1. First choose the strongest contiguous ~${targetSec.toFixed(0)}s of speech — the highlight: ` +
97
+ `self-contained, hook-worthy at its start, resolving by its end. Return it as \`highlight\` ` +
98
+ `(word range + a one-line reason).\n` +
99
+ `2. Prefer a window that starts at a sentence start and ends at a sentence end — a boundary ` +
100
+ `mid-sentence will be snapped to the nearest sentence afterwards.\n` +
101
+ `3. Then write the beat sheet AS IF the highlight were the whole take: the hook and EVERY ` +
102
+ `moment must lie inside the highlight's word range. Plan nothing outside it.`
103
+ );
104
+ }
105
+
106
+ export function buildBeatsUserPrompt(
107
+ transcript: Transcript,
108
+ duration: number,
109
+ intent: string | undefined,
110
+ framingBrief?: string,
111
+ clip?: { targetSec: number },
112
+ aspect?: "9:16" | "16:9",
113
+ ): string {
114
+ const words = transcript.words
115
+ .map((w, i) => `[${i}]${w.text}`)
116
+ .join(" ");
117
+ const menu = Object.entries(SCENE_REGISTRY)
118
+ .map(([id, meta]) => `- ${id}: ${meta.whenToUse}`)
119
+ .join("\n");
120
+ return (
121
+ `Intent: ${intent ?? "make this clear, punchy and viral-worthy"}\n` +
122
+ (clip
123
+ ? `Target clip length: ~${clip.targetSec.toFixed(0)}s (see CLIP SELECTION below)\n\n`
124
+ : `Output duration after the cut: ${duration.toFixed(1)}s\n\n`) +
125
+ // Landscape layout guidance (R21 §101): without it the first real 16:9
126
+ // run put nearly every graphic in a lower third. A deterministic variety
127
+ // pass downstream is the guarantee; this is the steer.
128
+ (aspect === "16:9"
129
+ ? `Output frame: LANDSCAPE 16:9. Vary the \`layout\` deliberately — lower-third, ` +
130
+ `split-left, split-right and blurred-behind are all available; never the same ` +
131
+ `layout twice in a row, and never a list/terminal/chat component in a lower-third ` +
132
+ `(the band is too shallow for a stack).\n\n`
133
+ : "") +
134
+ `Scene components available (sceneKind values; "none" = talking head only):\n${menu}\n\n` +
135
+ // The framing brief sits ABOVE the transcript so the constraint is read
136
+ // before the content it constrains (Task A).
137
+ (framingBrief ? `${framingBrief}\n\n` : "") +
138
+ `Word-indexed transcript (word indices refer to THIS list):\n${words}` +
139
+ (clip ? buildClipAddendum(clip.targetSec) : "")
140
+ );
141
+ }
142
+
143
+ export interface BeatsValidationIssue {
144
+ /** Index of the offending moment, or -1 for a sheet-wide issue. */
145
+ moment: number;
146
+ issue: string;
147
+ }
148
+
149
+ /** Fraction of the runtime that should show a graphic (FINDINGS §7). */
150
+ export const GRAPHICS_COVERAGE_TARGET = 0.45;
151
+ /**
152
+ * Below this runtime the percentage budget starves the video (FINDINGS §29):
153
+ * 45% of a 32s take is 14s, which at the 5s per-scene cap buys only three
154
+ * graphics — and short-form is exactly where density matters most. Under this
155
+ * threshold the scene COUNT floor wins over the percentage.
156
+ */
157
+ export const SHORT_TAKE_SEC = 45;
158
+ export const SHORT_TAKE_MIN_GRAPHICS = 4;
159
+
160
+ /** A moment's approximate seconds of speech, from the transcript word stamps. */
161
+ function momentDuration(m: Moment, transcript: Transcript): number {
162
+ const first = transcript.words[m.startWord];
163
+ const last = transcript.words[m.endWord];
164
+ return first && last ? Math.max(0, last.end - first.start) : 0;
165
+ }
166
+
167
+ function momentMidpoint(m: Moment, transcript: Transcript): number {
168
+ const first = transcript.words[m.startWord];
169
+ const last = transcript.words[m.endWord];
170
+ return first && last ? (first.start + last.end) / 2 : 0;
171
+ }
172
+
173
+ /** Semantic validation beyond the schema; repairs what it can, reports the rest. */
174
+ export function normalizeBeatSheet(
175
+ sheet: BeatSheet,
176
+ transcript: Transcript,
177
+ ): { sheet: BeatSheet; issues: BeatsValidationIssue[] } {
178
+ const wordCount = transcript.words.length;
179
+ const issues: BeatsValidationIssue[] = [];
180
+ const moments: Moment[] = [];
181
+ const maxIndex = Math.max(0, wordCount - 1);
182
+ for (let i = 0; i < sheet.moments.length; i++) {
183
+ const m = { ...sheet.moments[i]! };
184
+ if (m.startWord > maxIndex) {
185
+ issues.push({ moment: i, issue: `startWord ${m.startWord} beyond transcript (${maxIndex})` });
186
+ continue;
187
+ }
188
+ m.endWord = Math.min(m.endWord, maxIndex);
189
+ if (m.endWord < m.startWord) {
190
+ issues.push({ moment: i, issue: "endWord before startWord" });
191
+ continue;
192
+ }
193
+ const prev = moments[moments.length - 1];
194
+ if (prev && m.startWord <= prev.endWord) {
195
+ const shifted = prev.endWord + 1;
196
+ if (shifted > m.endWord) {
197
+ issues.push({ moment: i, issue: "fully overlaps previous moment" });
198
+ continue;
199
+ }
200
+ issues.push({ moment: i, issue: `overlapped previous; startWord ${m.startWord} → ${shifted}` });
201
+ m.startWord = shifted;
202
+ }
203
+ moments.push(m);
204
+ }
205
+
206
+ // ---- Graphics scheduling (FINDINGS §7/§8/§9) -----------------------------
207
+ // One coverage budget in SECONDS instead of a moment-count cap, so the
208
+ // per-scene time cap and the demotion can't stack multiplicatively (§7).
209
+ // Demotion order is by clustering + same-kind adjacency, so survivors stay
210
+ // spread across the timeline (§8) and varied (§9); hook and payoff are
211
+ // always spared.
212
+ const surviving = () => moments.flatMap((m, i) => (m.sceneKind !== "none" ? [i] : []));
213
+ const estShow = (i: number) =>
214
+ Math.min(momentDuration(moments[i]!, transcript), MAX_SCENE_SEC);
215
+ const runtime =
216
+ transcript.words.length > 0
217
+ ? transcript.words[transcript.words.length - 1]!.end - transcript.words[0]!.start
218
+ : 0;
219
+ const budget = GRAPHICS_COVERAGE_TARGET * runtime;
220
+
221
+ const demote = (i: number, why: string) => {
222
+ issues.push({ moment: i, issue: `demoted ${moments[i]!.sceneKind} to "none" (${why})` });
223
+ moments[i] = { ...moments[i]!, sceneKind: "none" };
224
+ };
225
+
226
+ // On a short take the count floor outranks the percentage — never demote
227
+ // below it, whatever the coverage budget says (§29).
228
+ const minGraphics = runtime < SHORT_TAKE_SEC ? SHORT_TAKE_MIN_GRAPHICS : 0;
229
+
230
+ for (;;) {
231
+ const graphics = surviving();
232
+ const shown = graphics.reduce((acc, i) => acc + estShow(i), 0);
233
+ if (shown <= budget + 1e-6) break;
234
+ if (graphics.length <= minGraphics) {
235
+ issues.push({
236
+ moment: 0,
237
+ issue:
238
+ `short take (${runtime.toFixed(0)}s): keeping ${graphics.length} graphics ` +
239
+ `over the ${(GRAPHICS_COVERAGE_TARGET * 100).toFixed(0)}% budget`,
240
+ });
241
+ break;
242
+ }
243
+ // Hook and payoff stay. Among the rest, demote whichever removal opens
244
+ // the SMALLEST gap between its surviving neighbours — the survivors stay
245
+ // spread instead of the tail (or middle) getting hollowed out (§8).
246
+ // Same-kind neighbours make a candidate maximally demotable (§9).
247
+ const candidates = graphics.slice(1, -1);
248
+ if (candidates.length === 0) break;
249
+ let pick = candidates[0]!;
250
+ let pickCost = Infinity;
251
+ for (const i of candidates) {
252
+ const pos = graphics.indexOf(i);
253
+ const prev = moments[graphics[pos - 1]!]!;
254
+ const next = moments[graphics[pos + 1]!]!;
255
+ const openedGap =
256
+ momentMidpoint(next, transcript) - momentMidpoint(prev, transcript);
257
+ const sameKind =
258
+ prev.sceneKind === moments[i]!.sceneKind || next.sceneKind === moments[i]!.sceneKind;
259
+ const cost = openedGap - (sameKind ? 1e6 : 0);
260
+ if (cost < pickCost) {
261
+ pickCost = cost;
262
+ pick = i;
263
+ }
264
+ }
265
+ demote(pick, `coverage ${(GRAPHICS_COVERAGE_TARGET * 100).toFixed(0)}%`);
266
+ }
267
+
268
+ // Variety pass independent of budget (§9): adjacent survivors must differ in
269
+ // kind — demote the later of a same-kind pair (the earlier if the later is
270
+ // the payoff; never the hook).
271
+ for (;;) {
272
+ const graphics = surviving();
273
+ if (graphics.length <= minGraphics) break; // §29 floor outranks variety too
274
+ const pair = graphics.findIndex(
275
+ (idx, p) => p > 0 && moments[idx]!.sceneKind === moments[graphics[p - 1]!]!.sceneKind,
276
+ );
277
+ if (pair === -1) break;
278
+ const later = graphics[pair]!;
279
+ const earlier = graphics[pair - 1]!;
280
+ const isPayoff = pair === graphics.length - 1;
281
+ const target = isPayoff ? (pair - 1 === 0 ? -1 : earlier) : later;
282
+ if (target === -1) break; // both hook and payoff — leave the repeat alone
283
+ demote(target, `duplicate adjacent ${moments[target]!.sceneKind}`);
284
+ }
285
+
286
+ // The cover banner (FINDINGS §35). Two things went wrong at once: this
287
+ // function used to drop `coverText` on the floor, so the CLI fell back to
288
+ // the full hook — and nothing enforced the nine-word cap the prompt asks
289
+ // for. Both are fixed here, where every path to a beat sheet passes.
290
+ const requested = sheet.coverText?.trim() || sheet.hook;
291
+ const coverText = coverHeadline(requested);
292
+ if (coverText !== requested) {
293
+ issues.push({ moment: -1, issue: `coverText shortened to "${coverText}"` });
294
+ }
295
+
296
+ return { sheet: { hook: sheet.hook, coverText, moments }, issues };
297
+ }
298
+
299
+ export async function generateBeatSheet(
300
+ provider: LlmProvider,
301
+ transcript: Transcript,
302
+ duration: number,
303
+ intent: string | undefined,
304
+ speaker?: string,
305
+ framingBrief?: string,
306
+ clip?: { targetSec: number },
307
+ aspect?: "9:16" | "16:9",
308
+ ): Promise<{ sheet: BeatSheet; issues: BeatsValidationIssue[]; highlight?: ClipHighlight }> {
309
+ const user =
310
+ (speaker ? `The speaker: ${speaker}\n\n` : "") +
311
+ buildBeatsUserPrompt(transcript, duration, intent, framingBrief, clip, aspect);
312
+ if (clip) {
313
+ // Same editorial call, extended schema (R19 §93d) — the highlight and the
314
+ // beat sheet come from ONE judgement, so they cannot disagree.
315
+ const raw = await provider.complete({
316
+ system: PRODUCER_SYSTEM,
317
+ user,
318
+ schema: ClipBeatSheetSchema,
319
+ schemaName: "clip_beat_sheet",
320
+ });
321
+ return { ...normalizeBeatSheet(raw, transcript), highlight: raw.highlight };
322
+ }
323
+ const raw = await provider.complete({
324
+ system: PRODUCER_SYSTEM,
325
+ user,
326
+ schema: BeatSheetSchema,
327
+ schemaName: "beat_sheet",
328
+ });
329
+ return normalizeBeatSheet(raw, transcript);
330
+ }
@@ -0,0 +1,150 @@
1
+ import { z } from "zod/v4";
2
+ import { run } from "../exec";
3
+ import type { LlmProvider } from "./provider";
4
+ import { estimateTokens, type LlmUsage } from "./usage";
5
+
6
+ /**
7
+ * Claude via the locally installed Claude Code CLI (`claude -p`) instead of
8
+ * the metered API. Uses whatever auth the CLI holds — for Pro/Max subscribers
9
+ * that's the subscription, so producing a video consumes plan usage rather
10
+ * than pay-per-token API credits.
11
+ *
12
+ * Unlike the API path there is no server-enforced structured output, so the
13
+ * schema is stated in the prompt and enforced here with zod, with one
14
+ * self-repair retry before throwing (the caller adds its own fallback on top).
15
+ *
16
+ * Requirements: `claude` on PATH (or OSSCLIP_CLAUDE_BIN) and a logged-in
17
+ * Claude Code (`claude` → /login). Each call pays ~1-2s of CLI startup.
18
+ */
19
+ export class ClaudeCliProvider implements LlmProvider {
20
+ readonly name = "claude-cli";
21
+ readonly usage: LlmUsage[] = [];
22
+
23
+ constructor(
24
+ private model?: string,
25
+ private bin: string = process.env.OSSCLIP_CLAUDE_BIN ?? "claude",
26
+ ) {}
27
+
28
+ async complete<T>(req: {
29
+ system: string;
30
+ user: string;
31
+ schema: z.ZodType<T>;
32
+ schemaName: string;
33
+ }): Promise<T> {
34
+ const schemaText = JSON.stringify(z.toJSONSchema(req.schema));
35
+ const base =
36
+ `${req.system}\n\n${req.user}\n\n` +
37
+ `Respond with ONLY a JSON object valid against this JSON Schema ("${req.schemaName}"). ` +
38
+ `No markdown fences, no commentary, no tool use — just the JSON:\n${schemaText}`;
39
+
40
+ let lastError = "";
41
+ for (let attempt = 0; attempt < 2; attempt++) {
42
+ const prompt =
43
+ attempt === 0
44
+ ? base
45
+ : `${base}\n\nYour previous reply failed validation:\n${lastError}\nReturn ONLY the corrected JSON.`;
46
+ const args = ["-p", "--output-format", "json", "--max-turns", "1"];
47
+ if (this.model) args.push("--model", this.model);
48
+ const started = Date.now();
49
+ let stdout = "";
50
+ try {
51
+ ({ stdout } = await run(this.bin, args, { stdin: prompt }));
52
+ } catch (err) {
53
+ lastError = err instanceof Error ? err.message : String(err);
54
+ continue;
55
+ }
56
+ // Recorded per ATTEMPT, before validation: a reply that failed the
57
+ // schema still spent the tokens, and a retry is exactly the cost a user
58
+ // would want to see rather than have quietly absorbed.
59
+ const envelope = parseCliEnvelope(stdout);
60
+ this.usage.push({
61
+ provider: this.name,
62
+ model: envelope.model ?? this.model,
63
+ schemaName: req.schemaName,
64
+ inputTokens: envelope.inputTokens ?? estimateTokens(prompt),
65
+ outputTokens: envelope.outputTokens ?? estimateTokens(envelope.result ?? stdout),
66
+ cachedInputTokens: envelope.cachedInputTokens,
67
+ reportedCostUsd: envelope.costUsd,
68
+ exact: envelope.inputTokens !== undefined,
69
+ // The whole point of this provider: Pro/Max auth means the plan pays,
70
+ // not a card. The cost is still reported, as what the same tokens
71
+ // would have cost on the API.
72
+ billed: false,
73
+ ms: Date.now() - started,
74
+ });
75
+ try {
76
+ return req.schema.parse(JSON.parse(extractJsonObject(unwrapCliEnvelope(stdout))));
77
+ } catch (err) {
78
+ lastError = err instanceof Error ? err.message : String(err);
79
+ }
80
+ }
81
+ throw new Error(
82
+ `claude CLI ('${this.bin}') did not produce valid ${req.schemaName} JSON: ${lastError.slice(0, 400)}\n` +
83
+ `Is Claude Code installed and logged in? (npm i -g @anthropic-ai/claude-code; run 'claude' once to /login)`,
84
+ );
85
+ }
86
+ }
87
+
88
+ /**
89
+ * The accounting half of the same envelope: tokens, cost and model, all
90
+ * optional because the CLI's envelope shape is not ours to depend on. Every
91
+ * field absent is a valid outcome — the caller falls back to estimates and
92
+ * says so — so this never throws on a shape it doesn't recognise.
93
+ */
94
+ export function parseCliEnvelope(stdout: string): {
95
+ result?: string;
96
+ model?: string;
97
+ inputTokens?: number;
98
+ outputTokens?: number;
99
+ cachedInputTokens?: number;
100
+ costUsd?: number;
101
+ } {
102
+ let env: Record<string, unknown>;
103
+ try {
104
+ env = JSON.parse(stdout.trim()) as Record<string, unknown>;
105
+ } catch {
106
+ return {};
107
+ }
108
+ if (!env || typeof env !== "object") return {};
109
+ const num = (v: unknown): number | undefined => (typeof v === "number" ? v : undefined);
110
+ const u = (env.usage ?? {}) as Record<string, unknown>;
111
+ const cacheRead = num(u.cache_read_input_tokens) ?? 0;
112
+ const cacheWrite = num(u.cache_creation_input_tokens) ?? 0;
113
+ const plainInput = num(u.input_tokens);
114
+ // `modelUsage` is keyed by the model id actually served — more reliable than
115
+ // the alias passed in with --model ("opus" resolves to a dated id).
116
+ const modelUsage = (env.modelUsage ?? {}) as Record<string, unknown>;
117
+ return {
118
+ result: typeof env.result === "string" ? env.result : undefined,
119
+ model: Object.keys(modelUsage)[0],
120
+ inputTokens: plainInput === undefined ? undefined : plainInput + cacheRead + cacheWrite,
121
+ outputTokens: num(u.output_tokens),
122
+ cachedInputTokens: cacheRead + cacheWrite || undefined,
123
+ costUsd: num(env.total_cost_usd),
124
+ };
125
+ }
126
+
127
+ /** `claude -p --output-format json` wraps the reply in a result envelope. */
128
+ export function unwrapCliEnvelope(stdout: string): string {
129
+ const trimmed = stdout.trim();
130
+ try {
131
+ const envelope = JSON.parse(trimmed) as { result?: unknown; is_error?: boolean };
132
+ if (envelope && typeof envelope.result === "string") {
133
+ if (envelope.is_error) throw new Error(`claude CLI errored: ${envelope.result.slice(0, 300)}`);
134
+ return envelope.result;
135
+ }
136
+ } catch (err) {
137
+ if (err instanceof Error && err.message.startsWith("claude CLI errored")) throw err;
138
+ // Not an envelope — fall through and treat stdout as the reply itself.
139
+ }
140
+ return trimmed;
141
+ }
142
+
143
+ /** Tolerate markdown fences / prose around the JSON object. */
144
+ export function extractJsonObject(text: string): string {
145
+ const unfenced = text.replace(/```(?:json)?/g, "");
146
+ const start = unfenced.indexOf("{");
147
+ const end = unfenced.lastIndexOf("}");
148
+ if (start === -1 || end <= start) throw new Error(`no JSON object in reply: ${text.slice(0, 200)}`);
149
+ return unfenced.slice(start, end + 1);
150
+ }