ossclip 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/produce.ts ADDED
@@ -0,0 +1,1407 @@
1
+ import { createHash } from "node:crypto";
2
+ import { createReadStream } from "node:fs";
3
+ import { mkdir, readFile, writeFile, rename } from "node:fs/promises";
4
+ import { existsSync } from "node:fs";
5
+ import { basename, dirname, isAbsolute, join, resolve } from "node:path";
6
+ import { z } from "zod/v4";
7
+ import {
8
+ LayoutSchema,
9
+ SceneSchema,
10
+ TimeMap,
11
+ TranscriptSchema,
12
+ analyze,
13
+ applyOverrides,
14
+ applyRepairs,
15
+ assembleScenes,
16
+ applyCaptionEdits,
17
+ buildCaptionLines,
18
+ buildCutlist,
19
+ buildZoomPlan,
20
+ checkGrounding,
21
+ rejectCtaKeyword,
22
+ coverHeadline,
23
+ cropFilter,
24
+ detectContentRect,
25
+ letterboxedSeconds,
26
+ type ContentRect,
27
+ createFaceDetector,
28
+ createProvider,
29
+ createTieredProvider,
30
+ defaultProviderName,
31
+ defaultTheme,
32
+ detectSilences,
33
+ dropHiddenCues,
34
+ emptyOverrideDoc,
35
+ extractAudio,
36
+ fillPlainCues,
37
+ splitCues,
38
+ landscapeLayout,
39
+ formatCutReport,
40
+ formatUsageLine,
41
+ formatUsageReport,
42
+ loadConfig,
43
+ loudnorm,
44
+ MAX_NORMALIZE_UPSCALE,
45
+ ZOOM_MAX_SCALE,
46
+ assessCueFraming,
47
+ bakeNormalizedSource,
48
+ planNormalization,
49
+ type NormalizePlan,
50
+ makeMezzanine,
51
+ measureFace,
52
+ measureFaceInWindows,
53
+ pickCoverFrame,
54
+ measureLevels,
55
+ probe,
56
+ produceScenes,
57
+ reclampPinnedTiming,
58
+ reconcileCopy,
59
+ repairTranscript,
60
+ resolveTheme,
61
+ run,
62
+ runWhisper,
63
+ scanSourceText,
64
+ appendUsageRun,
65
+ OverrideDocSchema,
66
+ CLIP_SNAP_TOLERANCE,
67
+ ClipWindowSchema,
68
+ boundCutlistToWindow,
69
+ formatClipTime,
70
+ parseClipWindowPin,
71
+ sliceRawTranscript,
72
+ sliceRepairs,
73
+ sliceTranscript,
74
+ type Analysis,
75
+ type AppliedRepair,
76
+ type CleanupLevel,
77
+ type ClipWindow,
78
+ type LlmProvider,
79
+ type Production,
80
+ type ProviderName,
81
+ type Scene,
82
+ type SceneComponentId,
83
+ type Segment,
84
+ type Transcript,
85
+ } from "@ossclip/core";
86
+ import { recordRecentProject } from "./edit";
87
+ import { renderCover, renderProduction } from "@ossclip/renderer";
88
+ import {
89
+ coverTextRect,
90
+ layoutSlots,
91
+ regionsDuring,
92
+ routeAroundSourceText,
93
+ } from "@ossclip/scenes/geometry";
94
+
95
+ export interface ProduceOptions {
96
+ out?: string;
97
+ cleanup: CleanupLevel;
98
+ transcript?: string;
99
+ render: boolean;
100
+ mezzanine: boolean;
101
+ workdir?: string;
102
+ inspect?: boolean;
103
+ /** Override the measured silence threshold (dBFS). */
104
+ noiseDb?: number;
105
+ /** Hand-authored scenes JSON (Scene[]) — skips the LLM entirely. */
106
+ scenes?: string;
107
+ /** Run the producer brain to plan scenes. */
108
+ produce?: boolean;
109
+ intent?: string;
110
+ provider?: ProviderName;
111
+ llmModel?: string;
112
+ /** Model for mechanical calls; "same" sends everything to the main model. */
113
+ llmFastModel?: string;
114
+ /** Who is on camera — steers repair and exempts their name from grounding. */
115
+ speaker?: string;
116
+ /** Repair ASR mishearings before captions/producer/grounding (default on). */
117
+ repair?: boolean;
118
+ /** Override the whisper model for this run (A/B base.en vs small.en). */
119
+ whisperModel?: string;
120
+ /** Debug: force every graphic moment to this component. */
121
+ forceComponent?: SceneComponentId;
122
+ /** Write a cover image beside the video (default on). */
123
+ cover?: boolean;
124
+ /** Explicit cover output path, overriding <out>.cover.jpg. */
125
+ coverPath?: string;
126
+ /** Treat the source as an already-edited reel with burned-in graphics. */
127
+ sourceIsEdited?: boolean;
128
+ /**
129
+ * How the source meets the vertical frame. `cover` (default) crops it to
130
+ * fill; `contain` shows the WHOLE frame inset against the backdrop, which is
131
+ * the answer for a landscape take whose content matters beyond the speaker's
132
+ * face.
133
+ */
134
+ sourceFit?: "cover" | "contain";
135
+ /**
136
+ * Output shape (R15). `9:16` is the vertical default every layout was tuned
137
+ * for; `16:9` exports 1920×1080 for YouTube/desktop, where there is no
138
+ * platform chrome to dodge and a landscape source needs no cropping at all.
139
+ */
140
+ aspect?: "9:16" | "16:9";
141
+ /**
142
+ * `--clip <seconds>` (R19 §93): produce only the strongest ~N-second window
143
+ * of a long take, chosen by the producer in the same editorial call as the
144
+ * beat sheet. Requires `--produce`; a source already at or under the target
145
+ * (+20% tolerance) is a no-op, not an error.
146
+ */
147
+ clip?: number;
148
+ /**
149
+ * `--clip-window <start:end>` (§93g): the RESOLVED window's word range,
150
+ * recorded into `command.json` so the editor's Render replays the SAME
151
+ * window with zero LLM calls. Written by clip runs; not for hand use.
152
+ */
153
+ clipWindow?: string;
154
+ }
155
+
156
+ function sha1File(path: string): Promise<string> {
157
+ return new Promise((res, rej) => {
158
+ const h = createHash("sha1");
159
+ createReadStream(path)
160
+ .on("data", (c) => h.update(c))
161
+ .on("end", () => res(h.digest("hex")))
162
+ .on("error", rej);
163
+ });
164
+ }
165
+
166
+ /**
167
+ * Frame-area share above which a layout's video slot is the SUBJECT rather
168
+ * than an inset. `video-top` is 42% and full-bleed 100%; the pip bubble is 5%.
169
+ */
170
+ const PRIMARY_VIDEO_SLOT_AREA = 0.2;
171
+
172
+ async function preflight(bin: string, hint: string): Promise<void> {
173
+ try {
174
+ await run(bin, ["-version"], { allowNonZero: true });
175
+ } catch {
176
+ try {
177
+ await run(bin, ["--help"], { allowNonZero: true });
178
+ } catch {
179
+ throw new Error(`'${bin}' not found. ${hint}`);
180
+ }
181
+ }
182
+ }
183
+
184
+ export async function produce(inputArg: string, opts: ProduceOptions): Promise<void> {
185
+ const cfg = loadConfig();
186
+ const input = resolve(inputArg);
187
+ if (!existsSync(input)) throw new Error(`input not found: ${input}`);
188
+
189
+ // §93b: the window is an editorial judgement, and there is deliberately no
190
+ // heuristic fallback — an automatically-guessed 60 seconds reads as a bug,
191
+ // not a limitation. Refused up front, before any work is spent.
192
+ if (opts.clip !== undefined) {
193
+ if (opts.scenes) {
194
+ throw new Error(
195
+ "--clip cannot be combined with hand-authored --scenes — the window is the producer's call.",
196
+ );
197
+ }
198
+ if (!opts.produce) {
199
+ throw new Error(
200
+ "--clip needs the producer's editorial judgement: add --produce. " +
201
+ "There is no heuristic fallback for picking the window.",
202
+ );
203
+ }
204
+ }
205
+ if (opts.clipWindow !== undefined && opts.clip === undefined) {
206
+ throw new Error("--clip-window is recorded by --clip runs for replay — pass --clip too.");
207
+ }
208
+
209
+ await preflight(cfg.ffmpegPath, "Install ffmpeg (brew install ffmpeg / apt install ffmpeg) or set OSSCLIP_FFMPEG.");
210
+ await preflight(cfg.ffprobePath, "Install ffmpeg (provides ffprobe) or set OSSCLIP_FFPROBE.");
211
+
212
+ // The output frame — every rect downstream is a fraction of THIS, and the
213
+ // stage geometry now takes it as an argument rather than assuming portrait.
214
+ const landscape = opts.aspect === "16:9";
215
+ const frame = landscape ? { width: 1920, height: 1080 } : { width: 1080, height: 1920 };
216
+
217
+ const hash = (await sha1File(input)).slice(0, 8);
218
+ const workRoot = opts.workdir ? resolve(opts.workdir) : join(dirname(input), ".ossclip");
219
+ const work = join(
220
+ workRoot,
221
+ `${basename(input).replace(/\.[^.]+$/, "")}-${hash}${landscape ? "-16x9" : ""}`,
222
+ );
223
+ await mkdir(work, { recursive: true });
224
+ const tools = { ffmpegPath: cfg.ffmpegPath, ffprobePath: cfg.ffprobePath };
225
+
226
+ console.log(`▸ workdir ${work}`);
227
+ const sourceProbe = await probe(tools, input);
228
+ console.log(
229
+ `▸ source ${sourceProbe.width}x${sourceProbe.height} @ ${sourceProbe.fps.toFixed(2)}fps · ${sourceProbe.duration.toFixed(2)}s`,
230
+ );
231
+ if (!sourceProbe.hasAudio) throw new Error("source has no audio stream — nothing to cut by");
232
+
233
+ // §93c: a source already at or under the target is a no-op, not an error —
234
+ // nobody should have to remember to drop the flag per input. The tolerance
235
+ // matches the sentence-snap band, so "just over" doesn't force a selection.
236
+ let clipTargetSec = opts.clip;
237
+ if (
238
+ clipTargetSec !== undefined &&
239
+ sourceProbe.duration <= clipTargetSec * (1 + CLIP_SNAP_TOLERANCE)
240
+ ) {
241
+ console.log(
242
+ `▸ source is ${sourceProbe.duration.toFixed(1)}s — already within the ` +
243
+ `${clipTargetSec}s clip target (+${(CLIP_SNAP_TOLERANCE * 100).toFixed(0)}% tolerance); ` +
244
+ "producing the whole take",
245
+ );
246
+ clipTargetSec = undefined;
247
+ }
248
+
249
+ const audioPath = join(work, "audio.wav");
250
+ if (!existsSync(audioPath)) {
251
+ console.log("▸ extracting audio…");
252
+ await extractAudio(tools, input, audioPath);
253
+ }
254
+
255
+ // Letterbox detection (PLAN Task 7): a file's frame is not always its
256
+ // picture — bars baked into the pixels wasted most of the video slot on one
257
+ // real clip. Measured once, before anything geometric; every downstream
258
+ // pass crops to the content rect so the bars stop existing.
259
+ const detection = await detectContentRect(tools, input, sourceProbe, { cacheDir: work });
260
+ const contentTimeline = detection.timeline;
261
+ const cropVf = cropFilter(detection.uniform);
262
+ if (detection.uniform && !detection.uniform.full) {
263
+ console.log(
264
+ `▸ source is letterboxed: content ${detection.uniform.w}×${detection.uniform.h} at ` +
265
+ `x ${detection.uniform.x}, y ${detection.uniform.y} (bars trimmed everywhere downstream)`,
266
+ );
267
+ }
268
+
269
+
270
+ let transcript: Transcript;
271
+ const transcriptCache = join(work, "transcript.json");
272
+ if (opts.transcript) {
273
+ transcript = TranscriptSchema.parse(JSON.parse(await readFile(resolve(opts.transcript), "utf8")));
274
+ console.log(`▸ transcript injected from ${opts.transcript} (${transcript.words.length} words)`);
275
+ } else if (existsSync(transcriptCache)) {
276
+ transcript = TranscriptSchema.parse(JSON.parse(await readFile(transcriptCache, "utf8")));
277
+ console.log(`▸ transcript cached (${transcript.words.length} words)`);
278
+ } else {
279
+ await preflight(
280
+ cfg.whisperPath,
281
+ "Install whisper.cpp (https://github.com/ggml-org/whisper.cpp) or set OSSCLIP_WHISPER.",
282
+ );
283
+ const model = opts.whisperModel ?? cfg.model;
284
+ const modelPath = isAbsolute(model) ? model : join(cfg.modelDir, `ggml-${model}.bin`);
285
+ if (!existsSync(modelPath)) {
286
+ throw new Error(
287
+ `whisper model not found at ${modelPath}.\n` +
288
+ `Download one, e.g.:\n curl -L -o ${modelPath} https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-${model}.bin`,
289
+ );
290
+ }
291
+ console.log(`▸ transcribing (${basename(modelPath)})…`);
292
+ transcript = await runWhisper(
293
+ { whisperPath: cfg.whisperPath, modelPath, outBase: join(work, "whisper") },
294
+ audioPath,
295
+ );
296
+ console.log(`▸ transcribed ${transcript.words.length} words`);
297
+ }
298
+ await writeFile(transcriptCache, JSON.stringify(transcript, null, 2));
299
+
300
+ const levels = await measureLevels({ ffmpegPath: cfg.ffmpegPath }, audioPath);
301
+ console.log(
302
+ `▸ levels: floor ${levels.floorDb.toFixed(1)} dB · speech ${levels.speechDb.toFixed(1)} dB ` +
303
+ `→ silence threshold ${levels.thresholdDb.toFixed(1)} dB`,
304
+ );
305
+ console.log("▸ analyzing silences…");
306
+ const silences = await detectSilences(
307
+ { ffmpegPath: cfg.ffmpegPath, noiseDb: opts.noiseDb ?? levels.thresholdDb },
308
+ audioPath,
309
+ );
310
+ // `let`, not `const` (R19 §93): a clip run re-derives all three from the
311
+ // transcript sliced to the chosen window, further down.
312
+ let analysis: Analysis = analyze(transcript, silences, sourceProbe.duration, levels);
313
+ let cutlist: Segment[] = buildCutlist({
314
+ transcript,
315
+ analysis,
316
+ duration: sourceProbe.duration,
317
+ level: opts.cleanup,
318
+ });
319
+ let map = new TimeMap(cutlist);
320
+
321
+ // ---- Transcript repair (FINDINGS §17/§21) --------------------------------
322
+ // Deliberately AFTER the cutlist: the cut is computed from raw ASR, so the
323
+ // same input and --cleanup always produce the same edit whether or not an
324
+ // LLM ran. Everything downstream that a viewer READS — captions, scene copy,
325
+ // the grounding check — uses the repaired transcript instead, so a
326
+ // mishearing can't reach the screen twice in two different spellings.
327
+ const providerName = opts.provider ?? defaultProviderName();
328
+ let provider: LlmProvider | null = null;
329
+ const needsLlm = opts.produce === true;
330
+ if (needsLlm) {
331
+ if (!opts.provider) {
332
+ console.log(
333
+ providerName === "claude-cli"
334
+ ? "▸ no ANTHROPIC_API_KEY — using the Claude Code CLI (subscription auth)"
335
+ : "▸ ANTHROPIC_API_KEY found — using the Claude API",
336
+ );
337
+ }
338
+ provider = createTieredProvider(providerName, {
339
+ model: opts.llmModel,
340
+ fastModel: opts.llmFastModel ?? cfg.fastModel,
341
+ });
342
+ }
343
+
344
+ let rawTranscript = transcript;
345
+ let repairs: AppliedRepair[] = [];
346
+ if (provider && opts.repair !== false) {
347
+ const rawKey = createHash("sha1")
348
+ .update(
349
+ JSON.stringify([
350
+ providerName,
351
+ opts.llmModel,
352
+ opts.llmFastModel ?? cfg.fastModel,
353
+ opts.speaker ?? cfg.speaker,
354
+ rawTranscript.words.map((w) => w.text),
355
+ ]),
356
+ )
357
+ .digest("hex")
358
+ .slice(0, 8);
359
+ const repairCache = join(work, `repairs-${rawKey}.json`);
360
+ if (existsSync(repairCache)) {
361
+ repairs = JSON.parse(await readFile(repairCache, "utf8")) as AppliedRepair[];
362
+ transcript = applyRepairs(
363
+ rawTranscript,
364
+ repairs.filter((r) => r.applied),
365
+ ).transcript;
366
+ console.log(`▸ repairs cached (${repairs.filter((r) => r.applied).length})`);
367
+ } else {
368
+ const result = await repairTranscript(provider, rawTranscript, {
369
+ speaker: opts.speaker ?? cfg.speaker,
370
+ // A repair may not merge words across a cut.
371
+ isCut: (startSec, endSec) =>
372
+ cutlist.some(
373
+ (s) => s.kind === "remove" && s.srcIn < endSec && s.srcOut > startSec,
374
+ ),
375
+ });
376
+ transcript = result.transcript;
377
+ repairs = result.applied;
378
+ if (result.error) {
379
+ // NEVER cache a FAILURE (§106). `repairTranscript` fails soft — a
380
+ // dead provider returns zero repairs, which on disk is
381
+ // indistinguishable from "this take needed none". Caching it made the
382
+ // failure permanent: every later run read `[]` and skipped the pass
383
+ // entirely, so the mishearings stayed in the captions and no amount
384
+ // of re-running could fix them. Same family as §78 — an artefact
385
+ // describing a state that isn't the one it was produced under.
386
+ console.log(
387
+ ` ⚠ transcript repair unavailable: ${result.error}\n` +
388
+ " (not cached — the next run retries the pass)",
389
+ );
390
+ } else {
391
+ await writeFile(repairCache, JSON.stringify(repairs, null, 2));
392
+ }
393
+ }
394
+ for (const r of repairs) {
395
+ console.log(
396
+ r.applied
397
+ ? ` ▸ repaired "${r.heard}" → "${r.correction}"`
398
+ : ` ⚠ repair refused: "${r.heard}" → "${r.correction}" (${r.rejected})`,
399
+ );
400
+ }
401
+ }
402
+
403
+ // ---- Framing measurement (PLAN Tasks A+B) --------------------------------
404
+ /**
405
+ * Mixed framing (option (a), decided with the author 2026-07-28): a source
406
+ * that alternates framings is NORMALIZED — every segment cropped to the
407
+ * tightest field of view the take ever shows, placed on that segment's own
408
+ * measured face, and baked into one uniform file. The PLAN is computed here,
409
+ * before the producer, because the producer needs the framing brief: which
410
+ * word ranges are close shots, and which layouts those rule out. The BAKE
411
+ * itself runs after the scenes exist.
412
+ */
413
+ let framingPlan: NormalizePlan | null = null;
414
+ if (!detection.uniform) {
415
+ const boxed = letterboxedSeconds(contentTimeline);
416
+ console.log(
417
+ `▸ source framing CHANGES mid-take: ${contentTimeline.length} segments, ` +
418
+ `${boxed.toFixed(1)}s of ${sourceProbe.duration.toFixed(1)}s letterboxed`,
419
+ );
420
+ for (const seg of contentTimeline) {
421
+ console.log(
422
+ ` · ${seg.startSec.toFixed(1)}–${seg.endSec.toFixed(1)}s ` +
423
+ (seg.rect.full
424
+ ? "full frame"
425
+ : `content ${seg.rect.w}×${seg.rect.h} at x ${seg.rect.x}, y ${seg.rect.y}`),
426
+ );
427
+ }
428
+ // Each segment's face, measured inside ITS OWN rect — a single median
429
+ // across mixed framings averages two coordinate systems and points the
430
+ // crop at neither (the old C5 gap; it put the eyes at the top of the
431
+ // frame on the motivating clip).
432
+ const segmentFaces = await measureFaceInWindows(
433
+ tools,
434
+ input,
435
+ contentTimeline.map((seg) => ({
436
+ startSec: seg.startSec,
437
+ endSec: seg.endSec,
438
+ cropVf: cropFilter(seg.rect),
439
+ })),
440
+ { workDir: work },
441
+ );
442
+ framingPlan = planNormalization(contentTimeline, segmentFaces, frame);
443
+ }
444
+
445
+ // ---- Scenes: hand-authored file, or the producer brain (PHASE1 §4) ----
446
+ let scenes: Scene[] = [];
447
+ /** Editorial output kept for the cover (§31): hook + its thumbnail form. */
448
+ let beatSheet: { hook: string; coverText?: string } | undefined;
449
+ /** Who planned this run (R16 §78) — stamped into production.json below. */
450
+ let producerStamp: Production["producer"];
451
+ /** The resolved `--clip` window (R19 §93) — set only on a clip run; feeds
452
+ * `production.json`, the report, and the command.json pin below. */
453
+ let clipWindow: ClipWindow | null = null;
454
+ if (opts.scenes) {
455
+ scenes = z.array(SceneSchema).parse(JSON.parse(await readFile(resolve(opts.scenes), "utf8")));
456
+ console.log(`▸ scenes injected from ${opts.scenes} (${scenes.length})`);
457
+ } else if (provider) {
458
+ // Keyed on the repaired transcript's TEXT, not its word count: a repair
459
+ // that swaps "coach and" for "code churn" leaves the count identical, and
460
+ // a count-keyed cache would silently replan from the stale wording.
461
+ // Camera-framing constraints for the producer (PLAN Tasks A+B): windows
462
+ // and canvas from the normalization plan, slot shapes from the stage —
463
+ // injected here because core must stay scenes-free. Only primary slots
464
+ // (video is the subject) are subject to the head-fits rule; a pip bubble
465
+ // is MEANT to be a tight head shot.
466
+ const framingCtx = framingPlan
467
+ ? {
468
+ windows: framingPlan.segments.map((s, i) => ({
469
+ startSec: s.startSec,
470
+ endSec: s.endSec,
471
+ faceFracOfCanvas: framingPlan.faceFracOfCanvas[i] ?? 0,
472
+ })),
473
+ canvasAspect: framingPlan.canvas.width / framingPlan.canvas.height,
474
+ layouts: LayoutSchema.options.map((layout) => {
475
+ const v = layoutSlots(layout).video;
476
+ return {
477
+ layout,
478
+ slotAspect: (v.rect.w * frame.width) / (v.rect.h * frame.height),
479
+ primary: v.opacity > 0 && v.rect.w * v.rect.h >= PRIMARY_VIDEO_SLOT_AREA,
480
+ };
481
+ }),
482
+ zoom: ZOOM_MAX_SCALE,
483
+ }
484
+ : undefined;
485
+ // ---- Clip window resolution (R19 §93) ---------------------------------
486
+ // Authority order: the command.json pin (§93g — the editor's Render must
487
+ // replay the SAME window) > the workdir's window cache > ONE extended
488
+ // beat-sheet call (§93d). Pin and cache both yield a window with zero LLM
489
+ // calls; only a first run selects.
490
+ let clipFresh: Awaited<ReturnType<typeof produceScenes>> | null = null;
491
+ if (clipTargetSec !== undefined) {
492
+ const windowKey = createHash("sha1")
493
+ .update(
494
+ JSON.stringify([
495
+ providerName,
496
+ opts.llmModel,
497
+ opts.intent,
498
+ clipTargetSec,
499
+ framingCtx ?? null,
500
+ transcript.words.map((w) => w.text),
501
+ ]),
502
+ )
503
+ .digest("hex")
504
+ .slice(0, 8);
505
+ const clipWindowCache = join(work, `clipwindow-${windowKey}.json`);
506
+ if (opts.clipWindow) {
507
+ clipWindow = parseClipWindowPin(transcript, opts.clipWindow);
508
+ console.log(
509
+ `▸ clip window pinned by the recorded command: words ` +
510
+ `${clipWindow.startWord}–${clipWindow.endWord}`,
511
+ );
512
+ } else if (existsSync(clipWindowCache)) {
513
+ clipWindow = ClipWindowSchema.parse(JSON.parse(await readFile(clipWindowCache, "utf8")));
514
+ console.log("▸ clip window cached");
515
+ } else {
516
+ console.log(`▸ selecting the strongest ~${clipTargetSec}s window (${providerName})…`);
517
+ clipFresh = await produceScenes(provider, {
518
+ transcript,
519
+ outputDuration: clipTargetSec,
520
+ intent: opts.intent,
521
+ speaker: opts.speaker ?? cfg.speaker,
522
+ forceComponent: opts.forceComponent,
523
+ framing: framingCtx,
524
+ clip: { targetSec: clipTargetSec },
525
+ aspect: landscape ? "16:9" : "9:16",
526
+ });
527
+ clipWindow = clipFresh.clip!.window;
528
+ for (const note of clipFresh.clip!.notes) console.log(` ▸ ${note}`);
529
+ await writeFile(clipWindowCache, JSON.stringify(clipWindow, null, 2));
530
+ }
531
+
532
+ // Slice the pipeline state to the window (§93.1), then let everything
533
+ // downstream — captions, scenes, zoom, the editor — run unchanged on
534
+ // the slice. The raw transcript slices by TIME (repairs may change word
535
+ // counts, so raw and repaired index spaces need not line up); repairs
536
+ // shift with it so production.json stays a reproducible pair.
537
+ console.log(
538
+ `▸ clip: ${formatClipTime(clipWindow.startSec)}–${formatClipTime(clipWindow.endSec)} ` +
539
+ `of ${formatClipTime(sourceProbe.duration)} — ${clipWindow.reason}`,
540
+ );
541
+ const rawSlice = sliceRawTranscript(rawTranscript, clipWindow);
542
+ repairs = sliceRepairs(repairs, rawSlice.offset, rawSlice.transcript.words.length);
543
+ rawTranscript = rawSlice.transcript;
544
+ transcript = sliceTranscript(transcript, clipWindow);
545
+ analysis = analyze(rawTranscript, silences, sourceProbe.duration, levels);
546
+ cutlist = boundCutlistToWindow(
547
+ buildCutlist({
548
+ transcript: rawTranscript,
549
+ analysis,
550
+ duration: sourceProbe.duration,
551
+ level: opts.cleanup,
552
+ }),
553
+ clipWindow,
554
+ sourceProbe.duration,
555
+ );
556
+ map = new TimeMap(cutlist);
557
+ }
558
+
559
+ const cacheKey = createHash("sha1")
560
+ .update(
561
+ JSON.stringify([
562
+ providerName,
563
+ opts.llmModel,
564
+ opts.intent,
565
+ opts.cleanup,
566
+ opts.forceComponent ?? null,
567
+ // The framing constraints steer layout choice, so a change in the
568
+ // measured framing must invalidate the cached plan.
569
+ framingCtx ?? null,
570
+ // §93f: the clip target and the RESOLVED window key the plan too —
571
+ // without them a clip run and a full run of the same source would
572
+ // collide and answer from each other's cache (the §78 failure
573
+ // mode). Keyed POST-resolution so a replay that derives the same
574
+ // window hits the same entries.
575
+ clipTargetSec ?? null,
576
+ clipWindow ? `${clipWindow.startWord}:${clipWindow.endWord}` : null,
577
+ transcript.words.map((w) => w.text),
578
+ ]),
579
+ )
580
+ .digest("hex")
581
+ .slice(0, 8);
582
+ const sceneCache = join(work, `scenes-${cacheKey}.json`);
583
+ // The cover needs the editorial copy, which is not in the scene list — a
584
+ // cached run must still be able to write one.
585
+ const beatCache = join(work, `beatsheet-${cacheKey}.json`);
586
+ if (clipFresh) {
587
+ // The selection call already planned the scenes (§93d: ONE editorial
588
+ // call chooses the window and the beats inside it) — adopt them and
589
+ // cache under the post-resolution key so re-runs and replays hit it.
590
+ scenes = clipFresh.scenes;
591
+ beatSheet = { hook: clipFresh.beatSheet.hook, coverText: clipFresh.beatSheet.coverText };
592
+ console.log(`▸ hook: ${clipFresh.beatSheet.hook}`);
593
+ console.log(
594
+ `▸ planned ${clipFresh.beatSheet.moments.length} moments, ${scenes.length} scenes` +
595
+ (clipFresh.failures.length > 0
596
+ ? ` (${clipFresh.failures.length} fell back to TitleCard)`
597
+ : ""),
598
+ );
599
+ for (const issue of clipFresh.beatIssues) {
600
+ console.log(` ⚠ moment ${issue.moment}: ${issue.issue}`);
601
+ }
602
+ await writeFile(sceneCache, JSON.stringify(scenes, null, 2));
603
+ await writeFile(beatCache, JSON.stringify(beatSheet, null, 2));
604
+ } else if (existsSync(sceneCache)) {
605
+ scenes = z.array(SceneSchema).parse(JSON.parse(await readFile(sceneCache, "utf8")));
606
+ console.log(`▸ scenes cached (${scenes.length})`);
607
+ if (existsSync(beatCache)) {
608
+ beatSheet = JSON.parse(await readFile(beatCache, "utf8")) as typeof beatSheet;
609
+ }
610
+ } else {
611
+ console.log(`▸ producing scenes (${providerName})…`);
612
+ if (opts.forceComponent) console.log(`▸ forcing every graphic to ${opts.forceComponent}`);
613
+ const result = await produceScenes(provider, {
614
+ transcript,
615
+ outputDuration: map.outputDuration,
616
+ intent: opts.intent,
617
+ speaker: opts.speaker ?? cfg.speaker,
618
+ forceComponent: opts.forceComponent,
619
+ framing: framingCtx,
620
+ aspect: landscape ? "16:9" : "9:16",
621
+ });
622
+ scenes = result.scenes;
623
+ beatSheet = { hook: result.beatSheet.hook, coverText: result.beatSheet.coverText };
624
+ console.log(`▸ hook: ${result.beatSheet.hook}`);
625
+ console.log(
626
+ `▸ planned ${result.beatSheet.moments.length} moments, ${scenes.length} scenes` +
627
+ (result.failures.length > 0 ? ` (${result.failures.length} fell back to TitleCard)` : ""),
628
+ );
629
+ for (const issue of result.beatIssues) {
630
+ console.log(` ⚠ moment ${issue.moment}: ${issue.issue}`);
631
+ }
632
+ // Cache props only — overrides are user-owned and live in overrides.json,
633
+ // never in production.json (that file is derived and every `produce`
634
+ // run overwrites it, per the merge rule in `overrides.ts`).
635
+ await writeFile(sceneCache, JSON.stringify(scenes, null, 2));
636
+ await writeFile(beatCache, JSON.stringify(beatSheet, null, 2));
637
+ }
638
+ }
639
+
640
+ // Every LLM call is behind us — repair, beat sheet, one per scene — so this
641
+ // is where a run can finally answer "what did that cost" (FINDINGS §36).
642
+ // A cached run legitimately spends nothing, and says so rather than
643
+ // reporting a zero that looks like a bug.
644
+ if (provider) {
645
+ console.log(
646
+ provider.usage.length === 0
647
+ ? "▸ llm: no calls — repairs and scenes came from the workdir cache"
648
+ : formatUsageLine(provider.usage, cfg.pricing),
649
+ );
650
+ // APPEND, never replace (R16 §78): a fully-cached re-run makes no calls,
651
+ // and overwriting the file with `records: []` erased which provider had
652
+ // planned the video. A malformed or pre-§78 file is a valid input — it
653
+ // becomes the history's first entry rather than an error.
654
+ const logPath = join(work, "usage.json");
655
+ let existing: unknown = {};
656
+ try {
657
+ existing = JSON.parse(await readFile(logPath, "utf8"));
658
+ } catch {
659
+ existing = {};
660
+ }
661
+ const log = appendUsageRun(
662
+ existing,
663
+ { at: new Date().toISOString(), records: provider.usage, provider: providerName },
664
+ cfg.pricing,
665
+ );
666
+ await writeFile(logPath, JSON.stringify(log, null, 2));
667
+ // …and stamp it onto the artefact it explains, so `production.json` says
668
+ // who planned it without a second file to cross-reference.
669
+ const last = log.runs[log.runs.length - 1]!;
670
+ producerStamp = {
671
+ provider: last.provider ?? providerName,
672
+ models: last.models,
673
+ cached: last.cached,
674
+ at: last.at,
675
+ };
676
+ }
677
+
678
+ // The overlay and the caption under it must spell the same word (§21).
679
+ // The repair pass runs before the producer, so they normally already agree;
680
+ // this catches the residue where the producer read through a mishearing the
681
+ // repair pass missed ("Orchestration Tax" over a caption reading "text").
682
+ // Word count and timings are untouched, so scene anchors stay valid.
683
+ if (scenes.length > 0) {
684
+ const reconciled = reconcileCopy(transcript, scenes);
685
+ transcript = reconciled.transcript;
686
+ repairs = [...repairs, ...reconciled.applied];
687
+ for (const r of reconciled.applied) {
688
+ console.log(` ▸ caption "${r.heard}" → "${r.correction}" (matches the on-screen copy)`);
689
+ }
690
+ }
691
+
692
+ // Landscape keeps the frame whole (R15): the split-screen layouts are
693
+ // vertical-format answers, and applying them to 16:9 crops the picture into
694
+ // a letterbox for no gain. Remapped here — before assembly — so cues,
695
+ // captions, the framing report and the editor all see one set of layouts.
696
+ if (landscape) {
697
+ const remapped = scenes.filter((sc) => landscapeLayout(sc.layout) !== sc.layout);
698
+ for (const sc of remapped) {
699
+ console.log(` ▸ ${sc.id}: ${sc.layout} → ${landscapeLayout(sc.layout)} (landscape)`);
700
+ sc.layout = landscapeLayout(sc.layout);
701
+ }
702
+ if (remapped.length > 0) {
703
+ console.log(`▸ ${remapped.length} scene(s) re-laid out for the 16:9 frame`);
704
+ }
705
+
706
+ // Layout VARIETY (R21 §101): the first real landscape run put nearly
707
+ // every graphic in a lower third — legal, monotonous, and for stack
708
+ // components actively broken (a BulletList's legibility floor cannot fit
709
+ // 0.18 of frame height; it rendered cropped). Two deterministic rules in
710
+ // time order: stack components never take the shallow band, and no two
711
+ // consecutive graphics share a layout. The prompt asks the producer for
712
+ // the same variety (see the aspect hint); this pass is the guarantee.
713
+ // The editor's per-scene layout override still wins over all of it.
714
+ const STACK_COMPONENTS = new Set<string>(["BulletList", "ChatMock", "TerminalMock"]);
715
+ const VARIETY_CYCLE: Array<Scene["layout"]> = [
716
+ "lower-third",
717
+ "split-right",
718
+ "blurred-behind",
719
+ "split-left",
720
+ ];
721
+ let prevLayout: Scene["layout"] | null = null;
722
+ let varied = 0;
723
+ for (const sc of scenes) {
724
+ let want = sc.layout;
725
+ if (STACK_COMPONENTS.has(sc.component) && want === "lower-third") want = "split-right";
726
+ if (want === prevLayout) {
727
+ const alternatives = VARIETY_CYCLE.filter(
728
+ (l) => l !== prevLayout && !(STACK_COMPONENTS.has(sc.component) && l === "lower-third"),
729
+ );
730
+ want = alternatives[Math.max(0, VARIETY_CYCLE.indexOf(want)) % alternatives.length]!;
731
+ }
732
+ if (want !== sc.layout) {
733
+ console.log(` ▸ ${sc.id}: ${sc.layout} → ${want} (landscape variety)`);
734
+ sc.layout = want;
735
+ varied++;
736
+ }
737
+ prevLayout = want;
738
+ }
739
+ if (varied > 0) console.log(`▸ ${varied} scene(s) re-laid out for variety`);
740
+ }
741
+
742
+ const { cues: assembled, dropped } = assembleScenes(scenes, transcript, map);
743
+ for (const d of dropped) console.log(` ⚠ scene ${d.id} dropped: ${d.reason}`);
744
+
745
+ // ---- Framing bake (plan step C / option (a)) ----------------------------
746
+ // The MEASUREMENT ran before the producer (Tasks A+B need it in the beat-
747
+ // sheet prompt); the BAKE stays here, after the scenes exist, so a future
748
+ // scene-aware bake has the cues in scope. Moving measurement up changes no
749
+ // edit decision — the cut is still computed on raw ASR above.
750
+ let analysisInput = input;
751
+ let analysisProbe = sourceProbe;
752
+ let analysisCropVf = cropVf;
753
+ let cacheTag = "";
754
+ let fitFallback = false;
755
+ if (framingPlan) {
756
+ const plan = framingPlan;
757
+ if (plan.ok) {
758
+ const planHash = createHash("sha1").update(JSON.stringify(plan)).digest("hex").slice(0, 8);
759
+ const baked = join(work, `content-${planHash}.mp4`);
760
+ if (!existsSync(baked)) {
761
+ console.log(
762
+ `▸ normalizing framing: one ${plan.canvas.width}×${plan.canvas.height} field of view ` +
763
+ `across ${plan.segments.length} segments (upscale ×${plan.coverUpscale.toFixed(2)})…`,
764
+ );
765
+ await bakeNormalizedSource(tools, input, plan, baked);
766
+ } else {
767
+ console.log(`▸ normalized framing cached (${basename(baked)})`);
768
+ }
769
+ analysisInput = baked;
770
+ analysisProbe = await probe(tools, baked);
771
+ analysisCropVf = "";
772
+ cacheTag = basename(baked);
773
+ } else {
774
+ fitFallback = true;
775
+ console.log(
776
+ ` ⚠ strip too small to unify (would upscale ×${plan.coverUpscale.toFixed(2)} > ` +
777
+ `${MAX_NORMALIZE_UPSCALE}) — letterboxed stretches render FITTED at natural size; ` +
778
+ `framing will visibly change at ${contentTimeline.length - 1} boundaries`,
779
+ );
780
+ }
781
+ }
782
+ const contentRect: ContentRect = detection.uniform ?? {
783
+ x: 0, y: 0, w: analysisProbe.width, h: analysisProbe.height, full: true,
784
+ };
785
+ /** The picture's dimensions — what every geometric consumer reasons about. */
786
+ const content = { width: contentRect.w, height: contentRect.h };
787
+
788
+ // Face measurement (FINDINGS §13): one static crop offset per source,
789
+ // measured rather than guessed; cached in the workdir like the transcript.
790
+ const faceSamples = 9;
791
+ const faceBox = await measureFace(tools, analysisInput, analysisProbe.duration, {
792
+ cacheDir: work,
793
+ cropVf: analysisCropVf,
794
+ cacheTag,
795
+ samples: faceSamples,
796
+ });
797
+ console.log(
798
+ faceBox
799
+ ? `▸ face at ${(faceBox.centerYFrac * 100).toFixed(0)}% down the frame, ` +
800
+ `${(faceBox.sizeFrac * 100).toFixed(0)}% tall ` +
801
+ `(${faceBox.framesDetected}/${faceBox.framesSampled} frames` +
802
+ `${faceBox.framesRotated ? `, ${faceBox.framesRotated} recovered by tilt sweep` : ""})`
803
+ : // A miss must be LOUD (PLAN Task 8): the silent fallback to the
804
+ // assumed selfie framing is how a wrong crop shipped unnoticed.
805
+ `▸ no face detected in ${faceSamples} sampled frames — using the ASSUMED framing; ` +
806
+ "the crop may be wrong, check the output",
807
+ );
808
+
809
+ // A landscape source loses most of its width to the vertical frame, and how
810
+ // much is arithmetic, not opinion: cover-cropping displays the picture at
811
+ // `height × aspect` and keeps only the frame's width of it. Said out loud
812
+ // because the result LOOKS deliberate — a tight talking head — and nothing
813
+ // else in the run would mention that the desk, the screen and the second
814
+ // person are simply gone.
815
+ if (!landscape && opts.sourceFit !== "contain" && content.height > 0) {
816
+ const displayedW = frame.height * (content.width / content.height);
817
+ if (displayedW > frame.width * 1.05) {
818
+ console.log(
819
+ `▸ source is ${(content.width / content.height).toFixed(2)}:1 — a full-frame crop keeps ` +
820
+ `${((frame.width / displayedW) * 100).toFixed(0)}% of its width. ` +
821
+ "Use --source-fit contain to show the whole frame instead.",
822
+ );
823
+ }
824
+ }
825
+
826
+ // ---- Route around the source's own burned-in text (FINDINGS §26) --------
827
+ // Fed a finished reel, ossclip would otherwise stack its layer on an
828
+ // existing one — cropping through the source's title and then restating it
829
+ // underneath. Graphics move to a clear slot or are skipped; captions never
830
+ // are, they just relocate.
831
+ const sourceText = await scanSourceText(tools, analysisInput, analysisProbe.duration, {
832
+ cacheDir: work,
833
+ assumeEdited: opts.sourceIsEdited,
834
+ cropVf: analysisCropVf,
835
+ cacheTag,
836
+ });
837
+ if (sourceText.regions.length > 0) {
838
+ console.log(
839
+ sourceText.assumed
840
+ ? "▸ --source-is-edited: assuming burned-in text in the title and caption bands"
841
+ : `▸ source already has on-screen text in ${sourceText.regions.length} band(s) ` +
842
+ `(${sourceText.framesSampled} frames sampled)`,
843
+ );
844
+ }
845
+ // Detection reports SOURCE time; everything downstream — cues, captions, the
846
+ // crop — is output time. Identical only while nothing is cut, so convert
847
+ // rather than let a cut silently slide every region out of place. Regions
848
+ // whose window is entirely removed drop out with it.
849
+ const textRegions = sourceText.regions.flatMap((r) => {
850
+ if (!Number.isFinite(r.endSec)) return [{ ...r, startSec: 0, endSec: map.outputDuration }];
851
+ const startSec = map.toOutputClamped(r.startSec);
852
+ const endSec = map.toOutputClamped(r.endSec);
853
+ return endSec > startSec ? [{ ...r, startSec, endSec }] : [];
854
+ });
855
+ const routed = routeAroundSourceText(assembled, textRegions);
856
+ for (const r of routed.relayouts) {
857
+ console.log(` ▸ scene ${r.id}: ${r.from} → ${r.to} (source text in the way)`);
858
+ }
859
+ for (const m of routed.moved) {
860
+ console.log(
861
+ ` ▸ scene ${m.id}: graphic moved into the free band at ` +
862
+ `${(m.y * 100).toFixed(0)}-${((m.y + m.h) * 100).toFixed(0)}%`,
863
+ );
864
+ }
865
+ for (const s of routed.skipped) console.log(` ⚠ scene ${s.id} skipped: ${s.reason}`);
866
+
867
+ // Source-text routing picks from a component's `altLayouts`, which include
868
+ // the vertical split layouts — so in landscape it can hand back exactly what
869
+ // the remap above removed. Re-assert the constraint on its OUTPUT, and say
870
+ // when that costs a text dodge: the graphic keeps whatever free-band rect
871
+ // routing gave it, but the frame stays whole (R15).
872
+ if (landscape) {
873
+ for (const c of routed.cues) {
874
+ const want = landscapeLayout(c.layout);
875
+ if (want === c.layout) continue;
876
+ console.log(
877
+ ` ▸ ${c.id}: ${c.layout} → ${want} (landscape; source-text routing had moved it)`,
878
+ );
879
+ c.layout = want;
880
+ }
881
+ }
882
+
883
+ // ---- The user's edit layer (SPEC: direct manipulation) -------------------
884
+ // Read AFTER assembly so hand edits sit on top of whatever the producer just
885
+ // planned, and never in production.json — that file is ours to overwrite.
886
+ const overridesPath = join(work, "overrides.json");
887
+ let overrideDoc = emptyOverrideDoc();
888
+ if (existsSync(overridesPath)) {
889
+ const parsed = OverrideDocSchema.safeParse(
890
+ JSON.parse(await readFile(overridesPath, "utf8")),
891
+ );
892
+ if (!parsed.success) {
893
+ // Hand-editable user data: refuse rather than silently resetting it.
894
+ throw new Error(`${overridesPath} is not valid: ${parsed.error.message}`);
895
+ }
896
+ overrideDoc = parsed.data;
897
+ }
898
+ const { cues: editedCues } = applyOverrides(routed.cues, overrideDoc);
899
+ // Scenes the user deleted in the editor drop here — their windows become
900
+ // plain takes in the fill below, which is Task C's payoff for Task A.
901
+ const { cues: visibleCues, hidden: hiddenIds } = dropHiddenCues(editedCues, overrideDoc);
902
+ if (hiddenIds.length > 0) {
903
+ console.log(`▸ ${hiddenIds.length} scene(s) hidden by the edit layer: ${hiddenIds.join(", ")}`);
904
+ }
905
+ const theme = resolveTheme(defaultTheme, overrideDoc);
906
+
907
+ // A pin freezes a scene's ABSOLUTE time against whatever its neighbours'
908
+ // timing was when it was set. This same plan may since have re-anchored
909
+ // those neighbours (a `--cleanup` level change, new source material), so
910
+ // the pin can now overlap one of them or leave the array out of time
911
+ // order — re-clamp it here, the same way the editor clamps a pinned nudge
912
+ // at drag time, rather than letting an overlap reach `SceneLayer`/
913
+ // `buildCaptionLines`.
914
+ const { cues: reclamped, adjusted } = reclampPinnedTiming(visibleCues);
915
+ for (const id of adjusted) {
916
+ console.log(` ⚠ pinned timing for ${id} overlapped a re-planned neighbour — clamped back in bounds`);
917
+ }
918
+
919
+ // Fill the timeline (PLAN 2026-07-30 Task A): every gap between graphic
920
+ // cues becomes a plain take, split at the cuts so a block never straddles
921
+ // one. Then a SECOND override pass, because the user's framing edits on
922
+ // `take-*` ids target cues that only exist after the fill. That pass is a
923
+ // no-op on the graphic cues it already touched — same component ⇒ no swap
924
+ // ⇒ the prop merge is idempotent — so do not "simplify" it away. Orphans
925
+ // are reported from THIS pass: only now does the id universe include the
926
+ // takes, so a `take-2-1` edit whose take merged away reports here instead
927
+ // of every take id reporting on the first pass.
928
+ const filled = fillPlainCues(reclamped, {
929
+ outputDurationSec: map.outputDuration,
930
+ clipStarts: map.spans.map((s) => s.outIn),
931
+ });
932
+ // User splits (R16 §61) — after the fill so takes split like scenes, and
933
+ // before the final override pass so edits on the `id@ms` halves land.
934
+ const split = splitCues(filled, overrideDoc.splits);
935
+ if (overrideDoc.splits.length > 0) {
936
+ console.log(`▸ ${overrideDoc.splits.length} scene split(s) from the edit layer`);
937
+ }
938
+ const { cues: mergedCues, orphans: rawOrphans } = applyOverrides(split, overrideDoc);
939
+ // Halves the user deleted AFTER splitting: their hidden override targets an
940
+ // `id@ms` id that only exists post-split, so the first drop above never saw
941
+ // it. Same order as the editor's live memo.
942
+ const { cues: sceneCues, hidden: hiddenHalves } = dropHiddenCues(mergedCues, overrideDoc);
943
+ if (hiddenHalves.length > 0) {
944
+ console.log(`▸ ${hiddenHalves.length} split half(s) hidden by the edit layer`);
945
+ }
946
+ // A hidden scene's id is absent from the filled list by construction —
947
+ // that's a deletion doing its job, not a lost edit.
948
+ const orphans = rawOrphans.filter((id) => !hiddenIds.includes(id));
949
+ const editedCount = Object.keys(overrideDoc.scenes).length;
950
+ if (editedCount > 0) {
951
+ console.log(`▸ applied your edits to ${editedCount - orphans.length - hiddenIds.length} scene(s)`);
952
+ }
953
+ for (const id of orphans) {
954
+ console.log(` ⚠ edit for ${id} dropped — the plan no longer has that scene`);
955
+ }
956
+
957
+ const graphicCues = sceneCues.filter((c) => c.kind !== "plain");
958
+ if (graphicCues.length > 0) {
959
+ console.log(
960
+ `▸ ${graphicCues.length} scene(s) on stage, ` +
961
+ `${sceneCues.length - graphicCues.length} plain take(s) filling the gaps: ` +
962
+ graphicCues.map((c) => `${c.component ?? c.id}@${c.startSec.toFixed(1)}s`).join(", "),
963
+ );
964
+ }
965
+
966
+ // Grounding post-check (FINDINGS §14a): flags label tokens the take never
967
+ // says — a hallucinated hook label is visible here without watching the video.
968
+ const groundingIssues = checkGrounding(scenes, transcript, opts.speaker ?? cfg.speaker);
969
+ for (const g of groundingIssues) {
970
+ console.log(` ⚠ grounding: ${g.component} ${g.sceneId} ${g.field} "${g.token}" — not in the take`);
971
+ }
972
+
973
+ // CTA-keyword post-check: the keyword mechanic renders ONE shape of ask, and
974
+ // a "reply with a number" prompt is not it. Dropping the prop here — before
975
+ // the cue, the scene file and the caption track ever read it — is what makes
976
+ // ChatMock fall back to rendering the exchange the producer actually planned
977
+ // (`chatBubbles` collapses to a single bubble only while a keyword is set).
978
+ const ctaRejections: Array<{ sceneId: string; keyword: string; reason: string }> = [];
979
+ for (const holder of [...scenes, ...graphicCues]) {
980
+ const kw = holder.props?.keyword;
981
+ if (typeof kw !== "string" || kw.length === 0) continue;
982
+ const reason = rejectCtaKeyword(kw);
983
+ if (!reason) continue;
984
+ delete (holder.props as Record<string, unknown>).keyword;
985
+ ctaRejections.push({ sceneId: holder.id, keyword: kw, reason });
986
+ }
987
+ // Scenes and cues carry the same resolved props, so each rejection is seen
988
+ // twice; report the word once.
989
+ for (const r of [...new Map(ctaRejections.map((r) => [r.keyword, r])).values()]) {
990
+ console.log(` ⚠ CTA keyword dropped — ${r.reason}`);
991
+ }
992
+
993
+ const production: Production = {
994
+ version: 1,
995
+ source: { path: input, probe: sourceProbe, audioPath, face: faceBox },
996
+ cleanup: opts.cleanup,
997
+ intent: opts.intent,
998
+ // The RAW transcript, because `analysis` and `cutlist` index into it —
999
+ // storing the repaired one here would leave those pointing at words that
1000
+ // no longer exist. Repairs are kept alongside so the repaired transcript
1001
+ // stays derivable (`applyRepairs`) rather than being a second truth.
1002
+ transcript: rawTranscript,
1003
+ repairs: repairs.length > 0 ? repairs : undefined,
1004
+ analysis,
1005
+ cutlist,
1006
+ ...(clipWindow && clipTargetSec !== undefined
1007
+ ? { clip: { targetSec: clipTargetSec, ...clipWindow } }
1008
+ : {}),
1009
+ scenes: scenes.length > 0 ? scenes : undefined,
1010
+ producer: producerStamp,
1011
+ theme,
1012
+ render: { ...frame, fps: 30 },
1013
+ };
1014
+ await writeFile(join(work, "production.json"), JSON.stringify(production, null, 2));
1015
+
1016
+ let report = formatCutReport(production);
1017
+ // §93h: a tool that discards 19 of 20 minutes owes the user an account of
1018
+ // why those 19 — the window, its share of the take, and the model's reason.
1019
+ if (clipWindow && clipTargetSec !== undefined) {
1020
+ const dur = clipWindow.endSec - clipWindow.startSec;
1021
+ report +=
1022
+ `\nclip window (--clip ${clipTargetSec}):\n` +
1023
+ ` ${formatClipTime(clipWindow.startSec)}–${formatClipTime(clipWindow.endSec)} of ` +
1024
+ `${formatClipTime(sourceProbe.duration)} (${dur.toFixed(1)}s selected, ` +
1025
+ `${((dur / sourceProbe.duration) * 100).toFixed(0)}% of the take)\n` +
1026
+ ` reason: ${clipWindow.reason}\n`;
1027
+ }
1028
+ const landed = repairs.filter((r) => r.applied);
1029
+ if (landed.length > 0) {
1030
+ report +=
1031
+ "\ntranscript repairs (mishearings corrected before captions — FINDINGS §17/§21):\n" +
1032
+ landed.map((r) => ` "${r.heard}" → "${r.correction}"`).join("\n") +
1033
+ "\n";
1034
+ }
1035
+ const refused = repairs.filter((r) => !r.applied);
1036
+ if (refused.length > 0) {
1037
+ report +=
1038
+ "\nrepairs refused (proposed but not a mishearing):\n" +
1039
+ refused.map((r) => ` "${r.heard}" → "${r.correction}": ${r.rejected}`).join("\n") +
1040
+ "\n";
1041
+ }
1042
+ if (groundingIssues.length > 0) {
1043
+ report +=
1044
+ "\ngrounding warnings (labels the take never says — FINDINGS §14):\n" +
1045
+ groundingIssues
1046
+ .map((g) => ` ${g.component} ${g.sceneId} ${g.field}: "${g.token}"`)
1047
+ .join("\n") +
1048
+ "\n";
1049
+ }
1050
+ if (provider) {
1051
+ report += formatUsageReport(provider.usage, cfg.pricing);
1052
+ // A cached run has no usage block to print, and used to leave the report
1053
+ // silent about who planned the video (R16 §78) — the same erasure the
1054
+ // usage log had. Name the provider it is reusing.
1055
+ if (provider.usage.length === 0 && producerStamp) {
1056
+ report +=
1057
+ `\nllm: no calls this run — planned by ${producerStamp.provider}` +
1058
+ (producerStamp.models.length > 0 ? ` (${producerStamp.models.join(", ")})` : "") +
1059
+ `, reused from the workdir cache\n`;
1060
+ }
1061
+ }
1062
+ // R21 §105 — the standard honesty line, in the artefact people forward.
1063
+ report +=
1064
+ "\nnote: the cut, captions and graphics are AI-generated — review the output before publishing.\n";
1065
+ await writeFile(join(work, "report.txt"), report);
1066
+ console.log("");
1067
+ console.log(report);
1068
+ console.log("");
1069
+
1070
+ const baseCaptionLines = buildCaptionLines(transcript, map, {
1071
+ // GRAPHIC cues only: a plain take is presentationally a gap, and letting
1072
+ // the fill's derived boundaries re-split caption lines would change
1073
+ // caption output for zero visual reason (PLAN Task A4.4).
1074
+ breakpoints: graphicCues.flatMap((c) => [c.startSec, c.endSec]),
1075
+ });
1076
+ // The user's retyped caption words (editor, PLAN 2026-07-29 Task 7 scope
1077
+ // (a)). Guarded per word: a stale edit — the pipeline re-derived a
1078
+ // different word at that position — is dropped LOUDLY, never applied to
1079
+ // the wrong word and never silently forgotten.
1080
+ const { lines: captionLines, dropped: staleCaptionEdits } = applyCaptionEdits(
1081
+ baseCaptionLines,
1082
+ overrideDoc.captions,
1083
+ );
1084
+ const liveCaptionEdits = Object.keys(overrideDoc.captions).length - staleCaptionEdits.length;
1085
+ if (liveCaptionEdits > 0) console.log(`▸ ${liveCaptionEdits} caption word(s) retyped by the editor`);
1086
+ for (const d of staleCaptionEdits) {
1087
+ console.log(
1088
+ ` ⚠ caption edit at word ${d.index} dropped: expected "${d.expected}" there, ` +
1089
+ `the transcript now has "${d.found}"`,
1090
+ );
1091
+ }
1092
+
1093
+ // Micro zoom punches (FINDINGS §15) reversing at real phrase breaks (§18).
1094
+ // Breaths are source-time; TimeMap has no span mapper, so both ends go
1095
+ // through toOutputClamped — a pause that was cut collapses to one instant,
1096
+ // which is still a boundary (a jump cut is a phrase break too).
1097
+ // One move per cut-free clip: ramp in, then hold. The clip starts ARE the
1098
+ // cuts — every point the source jumps — so a take that removed nothing is
1099
+ // one clip and gets exactly one slow push.
1100
+ const zoom = buildZoomPlan(map.outputDuration, {
1101
+ clipStarts: map.spans.map((s) => s.outIn),
1102
+ });
1103
+ console.log(
1104
+ `▸ zoom: ${zoom.clips} clip(s), ${zoom.rampSec}s push then hold ` +
1105
+ `(${zoom.segments.length} segments)`,
1106
+ );
1107
+
1108
+ // ---- Per-scene framing (plan step D) ------------------------------------
1109
+ // A slot wider than the source canvas is cover-cropped VERTICALLY, so it
1110
+ // shows only a fraction of the canvas height and the face grows by the
1111
+ // inverse. `video-top` is a wide band against a portrait canvas, which is
1112
+ // why a close-up moment placed there loses its crown. Not fixable by
1113
+ // cropping — the pixels a wide band wants do not exist in a portrait
1114
+ // close-up — so it is REPORTED here, and the producer is what has to stop
1115
+ // choosing that layout for those moments (steps A and B).
1116
+ if (framingPlan) {
1117
+ const toSource = (outSec: number): number => {
1118
+ for (const sp of map.spans) {
1119
+ if (outSec >= sp.outIn && outSec < sp.outOut) return sp.srcIn + (outSec - sp.outIn);
1120
+ }
1121
+ return map.spans[map.spans.length - 1]?.srcOut ?? outSec;
1122
+ };
1123
+ // Only layouts where the video IS the subject. A `pip-bubble` is a small
1124
+ // circular inset and a `graphic-only` slot is not even drawn (opacity 0):
1125
+ // a tight head-shot is what a bubble is FOR, so judging it against the
1126
+ // same head-fits rule would report a defect for working as designed.
1127
+ const issues = assessCueFraming(
1128
+ graphicCues.flatMap((c) => {
1129
+ const v = layoutSlots(c.layout).video;
1130
+ if (v.opacity <= 0 || v.rect.w * v.rect.h < PRIMARY_VIDEO_SLOT_AREA) return [];
1131
+ return [{
1132
+ id: c.id,
1133
+ layout: c.layout,
1134
+ startSec: toSource(c.startSec),
1135
+ endSec: toSource(c.endSec),
1136
+ slot: { width: v.rect.w * frame.width, height: v.rect.h * frame.height },
1137
+ }];
1138
+ }),
1139
+ framingPlan.segments,
1140
+ framingPlan.faceFracOfCanvas,
1141
+ framingPlan.canvas,
1142
+ ZOOM_MAX_SCALE,
1143
+ );
1144
+ const tight = issues.filter((f) => f.headFracOfSlot > 1);
1145
+ for (const f of tight) {
1146
+ console.log(
1147
+ ` ⚠ ${f.cueId} (${f.layout}): head is ${(f.headFracOfSlot * 100).toFixed(0)}% of its ` +
1148
+ `video slot — the crop will trim it. This layout is too wide for how close ` +
1149
+ `the speaker is here.`,
1150
+ );
1151
+ }
1152
+ if (issues.length > 0 && tight.length === 0) {
1153
+ const worst = issues.reduce((a, b) => (b.headFracOfSlot > a.headFracOfSlot ? b : a));
1154
+ console.log(
1155
+ `▸ framing: every scene fits its slot (tightest ${worst.cueId} at ` +
1156
+ `${(worst.headFracOfSlot * 100).toFixed(0)}% of its band)`,
1157
+ );
1158
+ }
1159
+ }
1160
+
1161
+ let renderVideo = analysisInput;
1162
+ // A letterboxed source MUST go through the re-encode even under
1163
+ // --no-mezzanine: the bars are pixels in the file, and cropping them here is
1164
+ // what lets every layout and zoom downstream treat the picture as the frame.
1165
+ // The cropped file gets its own name so a pre-crop cache is never reused.
1166
+ // A NORMALIZED source skips this outright: the bake already carries the
1167
+ // mezzanine's encode settings, and re-encoding it would be a second
1168
+ // generation of loss for nothing.
1169
+ if (analysisInput === input && (opts.mezzanine || !contentRect.full)) {
1170
+ const mezz = join(work, contentRect.full ? "mezzanine.mp4" : "mezzanine-content.mp4");
1171
+ if (!existsSync(mezz)) {
1172
+ console.log(
1173
+ contentRect.full
1174
+ ? "▸ building mezzanine (dense keyframes)…"
1175
+ : "▸ building mezzanine (dense keyframes, letterbox bars trimmed)…",
1176
+ );
1177
+ await makeMezzanine(tools, input, mezz, { cropVf: cropVf || undefined });
1178
+ }
1179
+ renderVideo = mezz;
1180
+ }
1181
+
1182
+ // Comment-CTA keyword (FINDINGS §16), scoped to the ask (FINDINGS §22).
1183
+ // Read off the timed CUE, not the untimed scene: the cue carries the same
1184
+ // resolved props AND the window, so the keyword can never come from a scene
1185
+ // that assembleScenes dropped, and the caption track knows exactly when the
1186
+ // ask is on screen. Quoting marks the word you type in the comments — every
1187
+ // other time the speaker merely says it, it must render plainly.
1188
+ const ctaCue = [...graphicCues]
1189
+ .reverse()
1190
+ .find((c) => typeof c.props?.keyword === "string" && (c.props.keyword as string).length > 0);
1191
+ const ctaKeyword = ctaCue ? (ctaCue.props!.keyword as string) : undefined;
1192
+ const ctaWindow = ctaCue
1193
+ ? { startSec: ctaCue.startSec, endSec: ctaCue.endSec }
1194
+ : undefined;
1195
+ if (ctaKeyword) {
1196
+ console.log(
1197
+ `▸ CTA keyword "${ctaKeyword}" styled only at ` +
1198
+ `${ctaWindow!.startSec.toFixed(1)}–${ctaWindow!.endSec.toFixed(1)}s`,
1199
+ );
1200
+ }
1201
+
1202
+ const props = {
1203
+ videoFileName: basename(renderVideo),
1204
+ spans: [...map.spans],
1205
+ captionLines,
1206
+ sceneCues,
1207
+ theme,
1208
+ // The PRISTINE, pre-override cues/theme — everything above this line
1209
+ // already has the CURRENT `overrides.json` baked in (so `sceneCues`/
1210
+ // `theme` are exactly what got rendered). The editor needs an unmerged
1211
+ // base to re-apply overrides onto instead: merging the live doc onto an
1212
+ // already-merged base is add-only, so a reset/un-pin/undo in a second
1213
+ // editing session would have nothing to fall back to and render as if
1214
+ // it never happened, even though `overrides.json` on disk is correct.
1215
+ baseSceneCues: routed.cues,
1216
+ baseTheme: defaultTheme,
1217
+ baseCaptionLines,
1218
+ settings: production.render,
1219
+ outputDurationSec: map.outputDuration,
1220
+ // The aspect travels with the measurement because the crop math needs it:
1221
+ // `object-fit: cover` spills vertically for a portrait source and
1222
+ // HORIZONTALLY for a landscape one, and the stage cannot tell which
1223
+ // without being told what shape the source is.
1224
+ face: faceBox
1225
+ ? {
1226
+ centerYFrac: faceBox.centerYFrac,
1227
+ centerXFrac: faceBox.centerXFrac,
1228
+ sizeFrac: faceBox.sizeFrac,
1229
+ // The CONTENT's shape, not the container's — with bars trimmed the
1230
+ // rendered video IS the content rect (PLAN Task 7).
1231
+ sourceAspect: content.height > 0 ? content.width / content.height : undefined,
1232
+ }
1233
+ : null,
1234
+ zoomPlan: zoom.segments,
1235
+ ctaKeyword,
1236
+ ctaWindow,
1237
+ sourceTextRegions: textRegions,
1238
+ // Sent ONLY on the fit fallback (option (b)): a normalized mixed source is
1239
+ // already one uniform file, and a uniform source had its bars cropped into
1240
+ // the mezzanine — cropping either again at render time would eat the
1241
+ // picture twice.
1242
+ ...(fitFallback
1243
+ ? {
1244
+ contentTimeline,
1245
+ sourceSize: { width: sourceProbe.width, height: sourceProbe.height },
1246
+ contentCropMode: "fit" as const,
1247
+ }
1248
+ : {}),
1249
+ // `--source-fit contain`: show the whole frame instead of cropping it.
1250
+ // The size sent is the PICTURE's, not the container's — with bars trimmed
1251
+ // into the mezzanine the rendered video IS the content rect, and fitting
1252
+ // against the container's shape would inset a frame that no longer exists.
1253
+ // Listed after the fit fallback so it wins on a source that is both mixed
1254
+ // and asked to be shown whole.
1255
+ ...(opts.sourceFit === "contain"
1256
+ ? { sourceFit: "contain" as const, sourceSize: content }
1257
+ : {}),
1258
+ };
1259
+ await writeFile(join(work, "render-props.json"), JSON.stringify(props, null, 2));
1260
+
1261
+ if (!opts.render) {
1262
+ console.log(`▸ skipping render (--no-render). Props at ${join(work, "render-props.json")}`);
1263
+ return;
1264
+ }
1265
+
1266
+ const outPath = resolve(opts.out ?? input.replace(/(\.[^.]+)?$/, ".ossclip.mp4"));
1267
+ const rawPath = join(work, "render-raw.mp4");
1268
+ console.log("▸ rendering…");
1269
+ let lastPct = -10;
1270
+ await renderProduction(props, {
1271
+ publicDir: dirname(renderVideo),
1272
+ outPath: rawPath,
1273
+ browserExecutable: cfg.browserExecutable,
1274
+ onProgress: (p) => {
1275
+ const pct = Math.floor(p * 100);
1276
+ if (pct >= lastPct + 10) {
1277
+ lastPct = pct;
1278
+ process.stdout.write(` ${pct}%\n`);
1279
+ }
1280
+ },
1281
+ });
1282
+ console.log("▸ normalizing loudness…");
1283
+ const normPath = join(work, "render-norm.mp4");
1284
+ await loudnorm(tools, rawPath, normPath);
1285
+ await rename(normPath, outPath);
1286
+
1287
+ // ---- Cover image (FINDINGS §31) -----------------------------------------
1288
+ // A separate file, not a burned-in intro: both platforms accept a custom
1289
+ // cover, so nothing has to be pickable from the video — and spending the
1290
+ // opening seconds on a title card fights the hook-in-2s policy directly.
1291
+ if (opts.cover !== false) {
1292
+ // §35's cap applies here too: a cached beat sheet from before the fix, or
1293
+ // the hook fallback, must not slip a 13-word paragraph onto a thumbnail.
1294
+ const coverText = coverHeadline(beatSheet?.coverText ?? beatSheet?.hook ?? "");
1295
+ if (!coverText) {
1296
+ console.log("▸ no cover text (run --produce for one) — skipping cover");
1297
+ } else {
1298
+ const detector = await createFaceDetector();
1299
+ const pick = await pickCoverFrame(tools, analysisInput, analysisProbe.duration, {
1300
+ cacheDir: work,
1301
+ cropVf: analysisCropVf,
1302
+ detectFace: (pixels, w, h) => {
1303
+ const d = detector(pixels, w, h);
1304
+ // pico returns [row, col, size, score] in detection-frame pixels,
1305
+ // and that frame is cropped exactly like the cover — so these
1306
+ // fractions are the cover's own geometry, not the source's.
1307
+ return d ? { centerXFrac: d[1] / w, centerYFrac: d[0] / h, sizeFrac: d[2] / h } : null;
1308
+ },
1309
+ });
1310
+ if (!pick) {
1311
+ console.log("▸ no usable cover frame found — skipping cover");
1312
+ } else {
1313
+ const frameName = "cover-frame.png";
1314
+ await run(cfg.ffmpegPath, [
1315
+ "-v", "error",
1316
+ "-ss", pick.timeSec.toFixed(3),
1317
+ "-i", analysisInput,
1318
+ "-frames:v", "1",
1319
+ "-vf", `${analysisCropVf ? `${analysisCropVf},` : ""}scale=${frame.width}:${frame.height}:force_original_aspect_ratio=increase,crop=${frame.width}:${frame.height}`,
1320
+ "-y", join(work, frameName),
1321
+ ]);
1322
+ const coverPath = resolve(
1323
+ opts.coverPath ?? outPath.replace(/(\.[^.]+)?$/, ".cover.jpg"),
1324
+ );
1325
+ // §34: if the source's own title is up at this instant, the frame
1326
+ // already has a headline. Adding ours states the same claim twice in
1327
+ // one image — a cover with one title beats a cover with two.
1328
+ const sourceTitled = regionsDuring(
1329
+ sourceText.regions,
1330
+ pick.timeSec - 0.5,
1331
+ pick.timeSec + 0.5,
1332
+ ).length > 0;
1333
+ console.log(
1334
+ `▸ cover from ${pick.timeSec.toFixed(1)}s ` +
1335
+ `(${pick.hasFace ? "face" : "no face"}, sharpness ${pick.sharpness.toFixed(0)})…`,
1336
+ );
1337
+ if (sourceTitled) {
1338
+ console.log(" ▸ source already has a title in this frame — shipping it without a banner");
1339
+ } else if (pick.face) {
1340
+ const band = coverTextRect(pick.face, frame);
1341
+ console.log(
1342
+ ` ▸ banner in the ${band.y + band.h / 2 < pick.face.centerYFrac ? "band above" : "band below"} ` +
1343
+ `the face (${(band.y * 100).toFixed(0)}-${((band.y + band.h) * 100).toFixed(0)}%)`,
1344
+ );
1345
+ }
1346
+ await renderCover(
1347
+ {
1348
+ frameFileName: frameName,
1349
+ text: sourceTitled ? "" : coverText,
1350
+ theme,
1351
+ face: pick.face,
1352
+ // The cover is the OUTPUT's thumbnail — a landscape render gets a
1353
+ // landscape cover (R16 §76). The still was already extracted at
1354
+ // this size; only the composition disagreed.
1355
+ frame: { width: frame.width, height: frame.height },
1356
+ },
1357
+ { publicDir: work, outPath: coverPath, browserExecutable: cfg.browserExecutable },
1358
+ );
1359
+ console.log(`✓ cover → ${coverPath}`);
1360
+ }
1361
+ }
1362
+ }
1363
+ // Record THIS invocation so the editor's Render button can replay it (R11
1364
+ // Task 4). Nothing else can reconstruct it — production.json has the
1365
+ // source path, cleanup and intent, but not --produce, --out or the LLM
1366
+ // flags — and guessing would silently render a different video than the
1367
+ // one on screen. execArgv carries the module loader (tsx in dev), so the
1368
+ // replay works from source and from a compiled build alike.
1369
+ // The provider may have been AUTO-DETECTED from this shell's environment
1370
+ // (a GEMINI_/ANTHROPIC_ key exported here). The editor's Render replays
1371
+ // this argv from the EDIT SERVER's environment, which may not have that
1372
+ // key — and the auto-detection would then silently pick a DIFFERENT
1373
+ // provider (R16 §75). Pin the RESOLVED choice into the recorded args —
1374
+ // never the key itself; secrets stay out of the workdir — so a replay
1375
+ // uses the same configuration or fails loudly asking for it.
1376
+ const recordedArgs = process.argv.slice(2);
1377
+ if (provider && !recordedArgs.includes("--llm")) {
1378
+ recordedArgs.push("--llm", providerName);
1379
+ }
1380
+ // §93g: pin the RESOLVED window, exactly as §75 pinned the provider. The
1381
+ // editor's Render replays this argv; if replay re-asked the model and got a
1382
+ // slightly different window, every saved override — anchored to scene ids
1383
+ // and word indices — would land on the wrong words. The word range, not
1384
+ // just `--clip 60`, is what makes replay deterministic with zero LLM calls.
1385
+ if (clipWindow && !recordedArgs.includes("--clip-window")) {
1386
+ recordedArgs.push("--clip-window", `${clipWindow.startWord}:${clipWindow.endWord}`);
1387
+ }
1388
+ await writeFile(
1389
+ join(work, "command.json"),
1390
+ JSON.stringify(
1391
+ {
1392
+ execPath: process.execPath,
1393
+ execArgv: process.execArgv,
1394
+ script: process.argv[1],
1395
+ args: recordedArgs,
1396
+ cwd: process.cwd(),
1397
+ out: outPath,
1398
+ },
1399
+ null,
1400
+ 2,
1401
+ ),
1402
+ );
1403
+ // Every produce run is a project the picker should offer (R17 §83) —
1404
+ // best-effort, so a read-only home dir never fails the render.
1405
+ await recordRecentProject(work);
1406
+ console.log(`✓ done → ${outPath}`);
1407
+ }