@ossclip/core 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,348 @@
1
+ import { readFile, unlink, writeFile } from "node:fs/promises";
2
+ import { existsSync } from "node:fs";
3
+ import { join } from "node:path";
4
+ import { run } from "./exec";
5
+
6
+ /**
7
+ * Burned-in text detection (FINDINGS §26, rebuilt for §32).
8
+ *
9
+ * ossclip assumes raw footage. Fed a finished reel it has no idea anything is
10
+ * already on screen, so it crops through the source's own title and then says
11
+ * much the same thing in different words directly beneath — two competing
12
+ * titles, one of them clipped.
13
+ *
14
+ * The product rule is asymmetric, and deliberately so:
15
+ * - CAPTIONS ALWAYS GO IN. They are the accessibility layer, so they move
16
+ * rather than disappear.
17
+ * - ossclip's own graphics must not overlap existing elements at all. If no
18
+ * free region can hold a scene, that scene is skipped.
19
+ *
20
+ * The first version reported zero regions on the exact footage it was built
21
+ * for. Measuring a reproduction of that clip showed why, and neither cause was
22
+ * the discriminator everyone suspected:
23
+ *
24
+ * 1. A burned-in title is TRANSIENT — it ran 6s of a 12s clip. The detector
25
+ * demanded a band be busy in half of ALL sampled frames, so a title that
26
+ * occupies a third of the runtime was voted out by the frames it was
27
+ * never in. Regions are now time-scoped, which is both the fix and the
28
+ * more honest model: a title only conflicts with scenes that share its
29
+ * window.
30
+ * 2. The edge threshold sat INSIDE the background noise. Measured on the
31
+ * reproduction: the title band scores 0.345 while every other band scores
32
+ * 0.021-0.069. The old 0.055 cut through that noise band, which is what
33
+ * made the golden fixture false-positive — and then the bimodality gate
34
+ * added to suppress it was blamed for suppressing real text too.
35
+ *
36
+ * Three signals now have to agree, each rejecting a different impostor:
37
+ * density (is anything drawn), bimodality (glyphs sit at the luminance
38
+ * extremes; scenery spreads across the midtones), and stroke structure (text
39
+ * is many SHORT runs per row; colour bars are a handful of very wide ones).
40
+ * Per-band scores are written to the cache so thresholds stay settable from
41
+ * measurements rather than guesses.
42
+ */
43
+
44
+ /** Occupancy rect in frame fractions, scoped to when it is on screen. */
45
+ export interface TextRegion {
46
+ x: number;
47
+ y: number;
48
+ w: number;
49
+ h: number;
50
+ /** SOURCE time this region is visible. */
51
+ startSec: number;
52
+ endSec: number;
53
+ /** Share of the samples inside its own window that saw it. */
54
+ confidence: number;
55
+ }
56
+
57
+ export interface BandScore {
58
+ edge: number;
59
+ bimodal: number;
60
+ stroke: number;
61
+ }
62
+
63
+ export interface SourceTextScan {
64
+ regions: TextRegion[];
65
+ framesSampled: number;
66
+ /** True when the user asserted the source is edited, skipping detection. */
67
+ assumed: boolean;
68
+ /** Per-sample, per-band measurements — kept so thresholds stay evidence-based. */
69
+ debug?: Array<{ timeSec: number; bands: BandScore[] }>;
70
+ }
71
+
72
+ /**
73
+ * Analysis width. The HEIGHT is whatever the source's aspect gives, rather
74
+ * than a fixed 9:16 — all three signals below are geometric, and squeezing a
75
+ * 16:9 frame into a portrait box turns every glyph stroke into a sliver and
76
+ * every horizontal run into a short one. That is the shape of text, so a
77
+ * stretched landscape source would score as text everywhere.
78
+ */
79
+ const DET_W = 240;
80
+ /** Rows of the analysis grid — bands are the unit, since text runs across. */
81
+ export const BANDS = 24;
82
+ /** Luminance step that counts as a glyph edge. */
83
+ const EDGE_THRESHOLD = 42;
84
+
85
+ /**
86
+ * Thresholds, set from measurements on a reproduction of the §32 clip
87
+ * (white-on-black title over a colour-bar background):
88
+ *
89
+ * band edge bimodal stroke
90
+ * title 0.345 0.76 high
91
+ * colour bars 0.042 0.33 low (few, very wide runs)
92
+ * checkerboard 0.069 0.30 high (dense, but low contrast)
93
+ *
94
+ * Each threshold sits in the gap, not at the edge of the noise.
95
+ */
96
+ const BAND_EDGE_RATIO = 0.12;
97
+ const BAND_BIMODALITY = 0.5;
98
+ const BAND_STROKE = 0.25;
99
+
100
+ /** A row needs at least this many transitions to look like a line of glyphs. */
101
+ const MIN_ROW_TRANSITIONS = 6;
102
+ /** …and its runs must be short relative to the frame: glyphs, not bars. */
103
+ const MAX_STROKE_FRACTION = 1 / 12;
104
+
105
+ /**
106
+ * Three scores per band.
107
+ *
108
+ * - `edge`: share of pixels sitting on a horizontal luminance step.
109
+ * - `bimodal`: share of pixels at the luminance extremes.
110
+ * - `stroke`: share of ROWS whose transitions are many and closely spaced.
111
+ *
112
+ * The third is what separates text from a colour-bar test pattern, which is
113
+ * every bit as bimodal as white-on-black type but is a handful of enormous
114
+ * runs rather than dozens of narrow ones.
115
+ */
116
+ export function bandScores(pixels: Uint8Array, w: number, h: number): BandScore[] {
117
+ const bandHeight = Math.max(1, Math.floor(h / BANDS));
118
+ const maxRun = w * MAX_STROKE_FRACTION;
119
+ const out: BandScore[] = [];
120
+ for (let b = 0; b < BANDS; b++) {
121
+ const y0 = b * bandHeight;
122
+ const y1 = Math.min(h, y0 + bandHeight);
123
+ let edges = 0;
124
+ let extreme = 0;
125
+ let count = 0;
126
+ let strokeRows = 0;
127
+ let rows = 0;
128
+ for (let y = y0; y < y1; y++) {
129
+ let transitions = 0;
130
+ let lastTransition = 0;
131
+ let shortRuns = 0;
132
+ for (let x = 1; x < w - 1; x++) {
133
+ const i = y * w + x;
134
+ const v = pixels[i]!;
135
+ if (v <= 48 || v >= 207) extreme++;
136
+ count++;
137
+ if (Math.abs(pixels[i + 1]! - pixels[i - 1]!) >= EDGE_THRESHOLD) {
138
+ edges++;
139
+ if (x - lastTransition > 1) {
140
+ transitions++;
141
+ if (x - lastTransition <= maxRun) shortRuns++;
142
+ lastTransition = x;
143
+ }
144
+ }
145
+ }
146
+ rows++;
147
+ if (transitions >= MIN_ROW_TRANSITIONS && shortRuns >= transitions * 0.6) strokeRows++;
148
+ }
149
+ out.push({
150
+ edge: count > 0 ? edges / count : 0,
151
+ bimodal: count > 0 ? extreme / count : 0,
152
+ stroke: rows > 0 ? strokeRows / rows : 0,
153
+ });
154
+ }
155
+ return out;
156
+ }
157
+
158
+ /** Does this band look like burned-in text? All three signals must agree. */
159
+ export function bandIsText(s: BandScore): boolean {
160
+ return s.edge >= BAND_EDGE_RATIO && s.bimodal >= BAND_BIMODALITY && s.stroke >= BAND_STROKE;
161
+ }
162
+
163
+ /**
164
+ * Turn per-sample band occupancy into time-scoped regions.
165
+ *
166
+ * Consecutive busy samples in a band become one region spanning their window;
167
+ * vertically adjacent bands sharing a window merge into a block. No global
168
+ * persistence vote — a title that runs a third of the clip is still a title,
169
+ * it just conflicts with a third of the scenes.
170
+ */
171
+ export function regionsFromSamples(
172
+ samples: Array<{ timeSec: number; busy: boolean[] }>,
173
+ halfStepSec: number,
174
+ ): TextRegion[] {
175
+ const perBand: TextRegion[] = [];
176
+ for (let b = 0; b < BANDS; b++) {
177
+ let runStart: number | null = null;
178
+ let seen = 0;
179
+ const flush = (endIdx: number) => {
180
+ if (runStart === null) return;
181
+ perBand.push({
182
+ x: 0,
183
+ y: b / BANDS,
184
+ w: 1,
185
+ h: 1 / BANDS,
186
+ startSec: Math.max(0, samples[runStart]!.timeSec - halfStepSec),
187
+ endSec: samples[endIdx]!.timeSec + halfStepSec,
188
+ confidence: seen / (endIdx - runStart + 1),
189
+ });
190
+ runStart = null;
191
+ seen = 0;
192
+ };
193
+ for (let i = 0; i < samples.length; i++) {
194
+ if (samples[i]!.busy[b]) {
195
+ if (runStart === null) runStart = i;
196
+ seen++;
197
+ } else if (runStart !== null) {
198
+ flush(i - 1);
199
+ }
200
+ }
201
+ if (runStart !== null) flush(samples.length - 1);
202
+ }
203
+
204
+ // Merge vertically adjacent bands whose windows overlap.
205
+ perBand.sort((a, b) => a.y - b.y || a.startSec - b.startSec);
206
+ const merged: TextRegion[] = [];
207
+ for (const r of perBand) {
208
+ const prev = merged[merged.length - 1];
209
+ const adjacent = prev && Math.abs(prev.y + prev.h - r.y) < 1e-9;
210
+ const overlaps = prev && r.startSec < prev.endSec && prev.startSec < r.endSec;
211
+ if (prev && adjacent && overlaps) {
212
+ prev.h += r.h;
213
+ prev.startSec = Math.min(prev.startSec, r.startSec);
214
+ prev.endSec = Math.max(prev.endSec, r.endSec);
215
+ prev.confidence = Math.min(prev.confidence, r.confidence);
216
+ } else {
217
+ merged.push({ ...r });
218
+ }
219
+ }
220
+
221
+ // Pad each merged region by one band. Detection localizes GLYPHS, but the
222
+ // graphic behind them — the rounded plate a title sits on — reaches past the
223
+ // last row of type, and it is the PLATE a crop visibly slices. Measured on
224
+ // the real reel: glyphs at 17-25% of the source, the black box from ~12.5%,
225
+ // and the rendered crop cut the box while clearing the text. One band is the
226
+ // detector's own resolution, so this claims no more precision than the
227
+ // measurement has. Padding happens after merging so it cannot fuse regions
228
+ // that the evidence kept apart.
229
+ const pad = 1 / BANDS;
230
+ for (const r of merged) {
231
+ const top = Math.max(0, r.y - pad);
232
+ r.h = Math.min(1, r.y + r.h + pad) - top;
233
+ r.y = top;
234
+ }
235
+ return merged;
236
+ }
237
+
238
+ /**
239
+ * Bands a conservative run assumes are occupied when detection is skipped
240
+ * (`--source-is-edited`). Burned-in titles sit in the upper third and burned-in
241
+ * captions in the lower-middle — the two places an editor puts them, and the
242
+ * two places ossclip most wants to draw. Assumed regions span the whole clip,
243
+ * because without detection there is no way to know when they are up.
244
+ */
245
+ export const ASSUMED_EDITED_REGIONS: TextRegion[] = [
246
+ { x: 0, y: 0.12, w: 1, h: 0.2, startSec: 0, endSec: Number.POSITIVE_INFINITY, confidence: 1 },
247
+ { x: 0, y: 0.66, w: 1, h: 0.12, startSec: 0, endSec: Number.POSITIVE_INFINITY, confidence: 1 },
248
+ ];
249
+
250
+ export interface ScanSourceTextOptions {
251
+ samples?: number;
252
+ cacheDir?: string;
253
+ /** Skip detection and assume the conservative regions above. */
254
+ assumeEdited?: boolean;
255
+ /**
256
+ * ffmpeg filter trimming the source to its content rect (PLAN Task 7).
257
+ * Letterbox bars are hard black edges — exactly what the density signal
258
+ * fires on — and regions must be fractions of the frame that RENDERS, which
259
+ * is the cropped one.
260
+ */
261
+ cropVf?: string;
262
+ /** Extra cache key for when the scanned file is the normalized bake. */
263
+ cacheTag?: string;
264
+ }
265
+
266
+ /** Cache format version — bump to invalidate stale scans after a rebuild. */
267
+ const SCAN_VERSION = 3;
268
+
269
+ /**
270
+ * Sample the take and report where — and WHEN — it already has text burned in.
271
+ * Cached in the workdir beside `face.json`: like the face box, this is a
272
+ * property of the source, not of a render.
273
+ */
274
+ export async function scanSourceText(
275
+ tools: { ffmpegPath: string },
276
+ videoPath: string,
277
+ durationSec: number,
278
+ opts: ScanSourceTextOptions = {},
279
+ ): Promise<SourceTextScan> {
280
+ if (opts.assumeEdited) {
281
+ return { regions: ASSUMED_EDITED_REGIONS, framesSampled: 0, assumed: true };
282
+ }
283
+ const cachePath = opts.cacheDir ? join(opts.cacheDir, "source-text.json") : null;
284
+ if (cachePath && existsSync(cachePath)) {
285
+ const cached = JSON.parse(await readFile(cachePath, "utf8")) as SourceTextScan & {
286
+ version?: number;
287
+ cropVf?: string;
288
+ cacheTag?: string;
289
+ };
290
+ // Regions are fractions of the analyzed frame, so a scan made against a
291
+ // different crop (or a different baked file) describes geometry that no
292
+ // longer renders.
293
+ if (
294
+ cached.version === SCAN_VERSION &&
295
+ (cached.cropVf ?? "") === (opts.cropVf ?? "") &&
296
+ (cached.cacheTag ?? "") === (opts.cacheTag ?? "")
297
+ ) {
298
+ return cached;
299
+ }
300
+ }
301
+
302
+ // A title can be short; sample densely enough to catch a ~2s one.
303
+ const samples = opts.samples ?? Math.min(40, Math.max(12, Math.round(durationSec / 1.5)));
304
+ const step = durationSec / samples;
305
+ const collected: Array<{ timeSec: number; busy: boolean[] }> = [];
306
+ const debug: SourceTextScan["debug"] = [];
307
+
308
+ for (let i = 0; i < samples; i++) {
309
+ const t = step * (i + 0.5);
310
+ const framePath = join(opts.cacheDir ?? ".", `text-frame-${i}.gray`);
311
+ await run(tools.ffmpegPath, [
312
+ "-v", "error",
313
+ "-ss", t.toFixed(3),
314
+ "-i", videoPath,
315
+ "-frames:v", "1",
316
+ // -2: height follows the source aspect, rounded to an even number.
317
+ "-vf", `${opts.cropVf ? `${opts.cropVf},` : ""}scale=${DET_W}:-2`,
318
+ "-pix_fmt", "gray",
319
+ "-f", "rawvideo",
320
+ "-y", framePath,
321
+ ]);
322
+ const pixels = new Uint8Array(await readFile(framePath));
323
+ await unlink(framePath).catch(() => {});
324
+ const detH = Math.floor(pixels.length / DET_W);
325
+ if (detH < BANDS) continue;
326
+ const scores = bandScores(pixels, DET_W, detH);
327
+ collected.push({ timeSec: t, busy: scores.map(bandIsText) });
328
+ debug.push({ timeSec: t, bands: scores });
329
+ }
330
+
331
+ const scan: SourceTextScan = {
332
+ regions: regionsFromSamples(collected, step / 2),
333
+ framesSampled: collected.length,
334
+ assumed: false,
335
+ debug,
336
+ };
337
+ if (cachePath) {
338
+ await writeFile(
339
+ cachePath,
340
+ JSON.stringify(
341
+ { version: SCAN_VERSION, cropVf: opts.cropVf ?? "", cacheTag: opts.cacheTag ?? "", ...scan },
342
+ null,
343
+ 2,
344
+ ),
345
+ );
346
+ }
347
+ return scan;
348
+ }
package/src/timemap.ts ADDED
@@ -0,0 +1,115 @@
1
+ import type { Segment, Word } from "./schema";
2
+
3
+ export interface KeptSpan {
4
+ srcIn: number;
5
+ srcOut: number;
6
+ outIn: number;
7
+ outOut: number;
8
+ }
9
+
10
+ /**
11
+ * Source-time ↔ output-time mapping derived from a cutlist.
12
+ *
13
+ * All overlay timings (captions, scenes, SFX) live in OUTPUT time; this is the
14
+ * only place source time is translated. Invariants (property-tested):
15
+ * - spans are sorted, non-overlapping, and contiguous in output time
16
+ * - outputDuration === Σ (srcOut - srcIn) over kept spans
17
+ * - toSource(toOutput(t)) === t for t strictly inside a kept span
18
+ * - toOutput(toSource(o)) === o for any kept output instant (projection identity)
19
+ * - toOutput is monotonically non-decreasing over kept source time
20
+ *
21
+ * Note: an output instant exactly at a cut boundary has TWO source preimages
22
+ * (the end of the span before the cut and the start of the span after);
23
+ * toSource deterministically returns the earlier one.
24
+ */
25
+ export class TimeMap {
26
+ readonly spans: readonly KeptSpan[];
27
+ readonly outputDuration: number;
28
+
29
+ constructor(cutlist: readonly Segment[]) {
30
+ let prevOut = -Infinity;
31
+ for (const s of cutlist) {
32
+ if (s.srcOut < s.srcIn) throw new Error(`segment ends before it starts: ${s.srcIn}..${s.srcOut}`);
33
+ if (s.srcIn < prevOut) throw new Error(`cutlist segments overlap at ${s.srcIn}`);
34
+ prevOut = s.srcOut;
35
+ }
36
+ const spans: KeptSpan[] = [];
37
+ let out = 0;
38
+ for (const s of cutlist) {
39
+ if (s.kind !== "keep" || s.srcOut <= s.srcIn) continue;
40
+ const dur = s.srcOut - s.srcIn;
41
+ spans.push({ srcIn: s.srcIn, srcOut: s.srcOut, outIn: out, outOut: out + dur });
42
+ out += dur;
43
+ }
44
+ this.spans = spans;
45
+ this.outputDuration = out;
46
+ }
47
+
48
+ /** Output time for a source instant, or null when the instant was cut. */
49
+ toOutput(tSrc: number): number | null {
50
+ // Exact containment first — a tolerance must never steal an instant that
51
+ // exactly belongs to another span (removed segments can be arbitrarily short).
52
+ for (const sp of this.spans) {
53
+ if (tSrc >= sp.srcIn && tSrc <= sp.srcOut) return sp.outIn + (tSrc - sp.srcIn);
54
+ }
55
+ // Then tolerate float-ulp overshoot at edges: clamp into the nearest span
56
+ // only when the instant is within EPS of it.
57
+ const EPS = 1e-9;
58
+ let best: KeptSpan | null = null;
59
+ let bestDist = Infinity;
60
+ for (const sp of this.spans) {
61
+ const dist = tSrc < sp.srcIn ? sp.srcIn - tSrc : tSrc - sp.srcOut;
62
+ if (dist < bestDist) {
63
+ bestDist = dist;
64
+ best = sp;
65
+ }
66
+ }
67
+ if (best && bestDist <= EPS) {
68
+ const clamped = Math.min(Math.max(tSrc, best.srcIn), best.srcOut);
69
+ return best.outIn + (clamped - best.srcIn);
70
+ }
71
+ return null;
72
+ }
73
+
74
+ /**
75
+ * Output time for a source instant, clamping instants that fall in removed
76
+ * regions to the nearest kept edge. Used for caption/overlay boundaries.
77
+ */
78
+ toOutputClamped(tSrc: number): number {
79
+ const exact = this.toOutput(tSrc);
80
+ if (exact !== null) return exact;
81
+ let best = 0;
82
+ for (const sp of this.spans) {
83
+ if (sp.srcOut <= tSrc) best = sp.outOut;
84
+ else if (sp.srcIn >= tSrc) return sp.outIn;
85
+ }
86
+ return best;
87
+ }
88
+
89
+ /** Source time for an output instant. Output time is contiguous, so this is total. */
90
+ toSource(tOut: number): number {
91
+ const spans = this.spans;
92
+ if (spans.length === 0) return 0;
93
+ const first = spans[0]!;
94
+ if (tOut <= first.outIn) return first.srcIn;
95
+ for (const sp of spans) {
96
+ if (tOut >= sp.outIn && tOut <= sp.outOut) return sp.srcIn + (tOut - sp.outIn);
97
+ }
98
+ return spans[spans.length - 1]!.srcOut;
99
+ }
100
+
101
+ /**
102
+ * Map a word into output time. Returns null when the word was entirely cut
103
+ * (e.g. a removed filler); ends are clamped when a cut clips the word edge.
104
+ */
105
+ mapWord(w: Word): { start: number; end: number } | null {
106
+ const mid = (w.start + w.end) / 2;
107
+ if (this.toOutput(mid) === null && this.toOutput(w.start) === null && this.toOutput(w.end) === null) {
108
+ return null;
109
+ }
110
+ const start = this.toOutputClamped(w.start);
111
+ const end = this.toOutputClamped(w.end);
112
+ if (end <= start) return null;
113
+ return { start, end };
114
+ }
115
+ }
@@ -0,0 +1,67 @@
1
+ import { readFile } from "node:fs/promises";
2
+ import { run } from "./exec";
3
+ import type { Transcript, Word } from "./schema";
4
+
5
+ /** Shape of whisper.cpp's `-oj` JSON output (the fields we consume). */
6
+ export interface WhisperJson {
7
+ result?: { language?: string };
8
+ transcription: Array<{
9
+ offsets: { from: number; to: number }; // milliseconds
10
+ text: string;
11
+ }>;
12
+ }
13
+
14
+ const NOISE_TOKEN = /^[[(].*[\])]$/; // [BLANK_AUDIO], (buzzing), [MUSIC] …
15
+
16
+ /**
17
+ * Convert whisper.cpp `-ml 1` segments (≈ one token each) into words.
18
+ * Tokens beginning with whitespace start a new word; bare continuations
19
+ * ("'s", "ing") merge into the previous word. Bracketed noise markers drop.
20
+ */
21
+ export function parseWhisperJson(json: WhisperJson): Transcript {
22
+ const words: Word[] = [];
23
+ for (const seg of json.transcription ?? []) {
24
+ const raw = seg.text;
25
+ if (!raw || !raw.trim()) continue;
26
+ const text = raw.trim();
27
+ if (NOISE_TOKEN.test(text)) continue;
28
+ const startsWord = /^\s/.test(raw) || words.length === 0;
29
+ const start = seg.offsets.from / 1000;
30
+ const end = seg.offsets.to / 1000;
31
+ const last = words[words.length - 1];
32
+ if (!startsWord && last) {
33
+ last.text += text;
34
+ last.end = Math.max(last.end, end);
35
+ } else {
36
+ words.push({ text, start, end });
37
+ }
38
+ }
39
+ // Whisper occasionally emits zero-length or inverted stamps; repair minimally.
40
+ for (let i = 0; i < words.length; i++) {
41
+ const w = words[i]!;
42
+ if (w.end <= w.start) w.end = w.start + 0.05;
43
+ const next = words[i + 1];
44
+ if (next && next.start < w.end) next.start = w.end;
45
+ }
46
+ return { language: json.result?.language ?? "en", words };
47
+ }
48
+
49
+ export interface WhisperOptions {
50
+ whisperPath: string;
51
+ modelPath: string;
52
+ /** Output base path; whisper writes `${outBase}.json`. */
53
+ outBase: string;
54
+ }
55
+
56
+ export async function runWhisper(opts: WhisperOptions, wavPath: string): Promise<Transcript> {
57
+ await run(opts.whisperPath, [
58
+ "-m", opts.modelPath,
59
+ "-f", wavPath,
60
+ "-oj",
61
+ "-of", opts.outBase,
62
+ "-ml", "1",
63
+ "--no-prints",
64
+ ]);
65
+ const json = JSON.parse(await readFile(`${opts.outBase}.json`, "utf8")) as WhisperJson;
66
+ return parseWhisperJson(json);
67
+ }
package/src/zoom.ts ADDED
@@ -0,0 +1,154 @@
1
+ /**
2
+ * Idle camera movement (FINDINGS §15): the cut-driven punch-in only fires at
3
+ * cuts, so a clean take sits visually static for 8–12 s at a time. This is the
4
+ * independent driver.
5
+ *
6
+ * ## Why this was rewritten (2026-07-28)
7
+ *
8
+ * The first version reversed direction at every speech-phrase boundary, on the
9
+ * theory that a phrase break is a natural place for the camera to turn around.
10
+ * On the author's own 64s take that found 24 boundaries and duly produced 24
11
+ * reversals, and the verdict was immediate: "the weird constant zooming in and
12
+ * zooming out". Reversing at a boundary is defensible ONCE; doing it every two
13
+ * seconds for a minute reads as a wobble, not as camera work. The bug was the
14
+ * contract, not the boundary detection — so the boundary machinery is gone
15
+ * rather than tuned.
16
+ *
17
+ * ## What it does now
18
+ *
19
+ * Within one cut-free CLIP the zoom moves in exactly one direction: a cosine
20
+ * ease from 1 to `maxScale` over `rampSec`, then a HOLD at `maxScale` for the
21
+ * rest of the clip. A cut resets it to 1, which is the one place a step is
22
+ * already justified — `EdlVideo`'s punch-in steps there too, and the frame
23
+ * changes anyway.
24
+ *
25
+ * Holding after the ramp is what keeps the move readable. Stretching 1 → 1.08
26
+ * across a 64s take is ~0.12%/s, which is indistinguishable from no motion; a
27
+ * bounded ramp followed by a hold is a slow push that arrives somewhere and
28
+ * stays — the author's "zoomed-out to zoomed-in, then keep that perspective
29
+ * consistent", composed from their own two options.
30
+ *
31
+ * A clip shorter than `rampSec` gets a PARTIAL push at the same rate rather
32
+ * than a compressed full one, so a 2s clip and a 20s clip move at the same
33
+ * speed. Making short clips complete the push would make them zoom visibly
34
+ * faster, which is the oscillation problem in a new costume.
35
+ *
36
+ * `maxScale: 1` disables the driver outright (the author's "no zoom" option)
37
+ * without needing a separate flag.
38
+ */
39
+
40
+ export interface ZoomSegment {
41
+ startSec: number;
42
+ endSec: number;
43
+ from: number;
44
+ to: number;
45
+ }
46
+
47
+ export interface ZoomPlan {
48
+ segments: ZoomSegment[];
49
+ /** How many cut-free clips the plan covers — logged, never inferred. */
50
+ clips: number;
51
+ /** The ramp actually used, so the log can't drift from the behaviour. */
52
+ rampSec: number;
53
+ }
54
+
55
+ export interface ZoomPlanOptions {
56
+ /** Zoomed-in extreme; the other extreme is 1. `1` disables the zoom. */
57
+ maxScale?: number;
58
+ /**
59
+ * Clip starts in OUTPUT time — the kept spans' `outIn`, i.e. every point the
60
+ * source jumps. A missing or empty list means the take is one clip, which is
61
+ * the correct reading of a cutlist that removed nothing.
62
+ */
63
+ clipStarts?: readonly number[];
64
+ /** Seconds the push takes to arrive before it holds. */
65
+ rampSec?: number;
66
+ }
67
+
68
+ /**
69
+ * Zoom amplitude, exported so the stage can budget crop margins against it.
70
+ *
71
+ * 5%, not 8%: at 8% the push was eating enough of an already-tight frame to
72
+ * clip a forehead on a close-up, and the amount a viewer should register is
73
+ * "the camera is alive", not "the camera moved". Every crop budget that cites
74
+ * this constant tightens with it rather than needing its own edit.
75
+ */
76
+ export const ZOOM_MAX_SCALE = 1.05;
77
+
78
+ /**
79
+ * How long the push takes. Long enough that the movement is never noticed as
80
+ * movement, short enough that it has arrived while the viewer is still on the
81
+ * hook — and, deliberately, far shorter than a typical clip so most of a clip
82
+ * is the settled perspective rather than a drift.
83
+ */
84
+ export const ZOOM_RAMP_SEC = 8;
85
+
86
+ /** Clip starts, cleaned: in range, unique, sorted, and always including 0. */
87
+ function clipBoundaries(starts: readonly number[] | undefined, duration: number): number[] {
88
+ const seen = new Set<number>([0]);
89
+ for (const t of starts ?? []) {
90
+ if (Number.isFinite(t) && t > 0 && t < duration) seen.add(t);
91
+ }
92
+ return [...seen].sort((a, b) => a - b);
93
+ }
94
+
95
+ export function buildZoomPlan(
96
+ outputDurationSec: number,
97
+ opts: ZoomPlanOptions = {},
98
+ ): ZoomPlan {
99
+ const maxScale = opts.maxScale ?? ZOOM_MAX_SCALE;
100
+ const rampSec = opts.rampSec ?? ZOOM_RAMP_SEC;
101
+ if (outputDurationSec <= 0) return { segments: [], clips: 0, rampSec };
102
+
103
+ const starts = clipBoundaries(opts.clipStarts, outputDurationSec);
104
+ const segments: ZoomSegment[] = [];
105
+
106
+ for (let i = 0; i < starts.length; i++) {
107
+ const start = starts[i]!;
108
+ const end = i + 1 < starts.length ? starts[i + 1]! : outputDurationSec;
109
+ const length = end - start;
110
+ if (length <= 1e-9) continue;
111
+
112
+ const rampEnd = Math.min(start + rampSec, end);
113
+ // Same rate for every clip: a short clip stops partway up rather than
114
+ // racing to the top.
115
+ const reached = 1 + (maxScale - 1) * Math.min(1, (rampEnd - start) / Math.max(rampSec, 1e-9));
116
+
117
+ segments.push({ startSec: start, endSec: rampEnd, from: 1, to: reached });
118
+ if (end - rampEnd > 1e-9) {
119
+ segments.push({ startSec: rampEnd, endSec: end, from: reached, to: reached });
120
+ }
121
+ }
122
+
123
+ return { segments, clips: starts.length, rampSec };
124
+ }
125
+
126
+ /**
127
+ * Scale at output time t — cosine-eased across each segment (never linear), so
128
+ * the push starts and settles gently. 1 outside the plan.
129
+ *
130
+ * A hold segment has `from === to`, which the same easing renders as a
131
+ * constant; no special case is needed and none should be added, because the
132
+ * ramp/hold boundary must not be a place where two code paths could disagree.
133
+ *
134
+ * Segments are matched HALF-OPEN, `[start, end)`. Under the old oscillating
135
+ * contract this was academic — neighbouring segments shared a value at every
136
+ * boundary — but a cut is a boundary where they deliberately disagree: the
137
+ * previous clip holds at `maxScale` and the next starts at 1. Closed matching
138
+ * let the earlier segment win, so the first frame of a new clip rendered the
139
+ * PREVIOUS clip's zoom and the reset appeared one frame late. The final instant
140
+ * of the plan has no following segment and so is served by the last one.
141
+ */
142
+ export function zoomScaleAt(plan: readonly ZoomSegment[], tSec: number): number {
143
+ for (const seg of plan) {
144
+ if (tSec >= seg.startSec && tSec < seg.endSec) {
145
+ const span = seg.endSec - seg.startSec;
146
+ const p = span > 0 ? (tSec - seg.startSec) / span : 1;
147
+ const eased = 0.5 - 0.5 * Math.cos(Math.PI * p);
148
+ return seg.from + (seg.to - seg.from) * eased;
149
+ }
150
+ }
151
+ const last = plan[plan.length - 1];
152
+ if (last && tSec === last.endSec) return last.to;
153
+ return 1;
154
+ }