@ossclip/core 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +27 -0
- package/README.md +20 -0
- package/package.json +29 -0
- package/src/analyze.ts +299 -0
- package/src/assemble.ts +124 -0
- package/src/browser.ts +24 -0
- package/src/captions.ts +92 -0
- package/src/clip.ts +306 -0
- package/src/config.ts +66 -0
- package/src/content-rect-detect.ts +162 -0
- package/src/content-rect.ts +324 -0
- package/src/cover.ts +216 -0
- package/src/cta.ts +68 -0
- package/src/cutlist.ts +170 -0
- package/src/exec.ts +36 -0
- package/src/face.ts +519 -0
- package/src/fill.ts +110 -0
- package/src/framing.ts +277 -0
- package/src/grounding.ts +130 -0
- package/src/index.ts +27 -0
- package/src/ingest.ts +83 -0
- package/src/normalize.ts +397 -0
- package/src/overrides.ts +509 -0
- package/src/phonetics.ts +129 -0
- package/src/producer/anthropic.ts +73 -0
- package/src/producer/beats.ts +330 -0
- package/src/producer/claude-cli.ts +150 -0
- package/src/producer/gemini.ts +197 -0
- package/src/producer/index.ts +217 -0
- package/src/producer/mock.ts +101 -0
- package/src/producer/provider.ts +42 -0
- package/src/producer/repair.ts +474 -0
- package/src/producer/scene-props.ts +212 -0
- package/src/producer/tiered.ts +56 -0
- package/src/producer/usage.ts +426 -0
- package/src/report.ts +36 -0
- package/src/scene-registry.ts +246 -0
- package/src/scene-schema.ts +203 -0
- package/src/schema.ts +177 -0
- package/src/source-text.ts +348 -0
- package/src/timemap.ts +115 -0
- package/src/transcribe.ts +67 -0
- package/src/zoom.ts +154 -0
|
@@ -0,0 +1,348 @@
|
|
|
1
|
+
import { readFile, unlink, writeFile } from "node:fs/promises";
|
|
2
|
+
import { existsSync } from "node:fs";
|
|
3
|
+
import { join } from "node:path";
|
|
4
|
+
import { run } from "./exec";
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Burned-in text detection (FINDINGS §26, rebuilt for §32).
|
|
8
|
+
*
|
|
9
|
+
* ossclip assumes raw footage. Fed a finished reel it has no idea anything is
|
|
10
|
+
* already on screen, so it crops through the source's own title and then says
|
|
11
|
+
* much the same thing in different words directly beneath — two competing
|
|
12
|
+
* titles, one of them clipped.
|
|
13
|
+
*
|
|
14
|
+
* The product rule is asymmetric, and deliberately so:
|
|
15
|
+
* - CAPTIONS ALWAYS GO IN. They are the accessibility layer, so they move
|
|
16
|
+
* rather than disappear.
|
|
17
|
+
* - ossclip's own graphics must not overlap existing elements at all. If no
|
|
18
|
+
* free region can hold a scene, that scene is skipped.
|
|
19
|
+
*
|
|
20
|
+
* The first version reported zero regions on the exact footage it was built
|
|
21
|
+
* for. Measuring a reproduction of that clip showed why, and neither cause was
|
|
22
|
+
* the discriminator everyone suspected:
|
|
23
|
+
*
|
|
24
|
+
* 1. A burned-in title is TRANSIENT — it ran 6s of a 12s clip. The detector
|
|
25
|
+
* demanded a band be busy in half of ALL sampled frames, so a title that
|
|
26
|
+
* occupies a third of the runtime was voted out by the frames it was
|
|
27
|
+
* never in. Regions are now time-scoped, which is both the fix and the
|
|
28
|
+
* more honest model: a title only conflicts with scenes that share its
|
|
29
|
+
* window.
|
|
30
|
+
* 2. The edge threshold sat INSIDE the background noise. Measured on the
|
|
31
|
+
* reproduction: the title band scores 0.345 while every other band scores
|
|
32
|
+
* 0.021-0.069. The old 0.055 cut through that noise band, which is what
|
|
33
|
+
* made the golden fixture false-positive — and then the bimodality gate
|
|
34
|
+
* added to suppress it was blamed for suppressing real text too.
|
|
35
|
+
*
|
|
36
|
+
* Three signals now have to agree, each rejecting a different impostor:
|
|
37
|
+
* density (is anything drawn), bimodality (glyphs sit at the luminance
|
|
38
|
+
* extremes; scenery spreads across the midtones), and stroke structure (text
|
|
39
|
+
* is many SHORT runs per row; colour bars are a handful of very wide ones).
|
|
40
|
+
* Per-band scores are written to the cache so thresholds stay settable from
|
|
41
|
+
* measurements rather than guesses.
|
|
42
|
+
*/
|
|
43
|
+
|
|
44
|
+
/** Occupancy rect in frame fractions, scoped to when it is on screen. */
|
|
45
|
+
export interface TextRegion {
|
|
46
|
+
x: number;
|
|
47
|
+
y: number;
|
|
48
|
+
w: number;
|
|
49
|
+
h: number;
|
|
50
|
+
/** SOURCE time this region is visible. */
|
|
51
|
+
startSec: number;
|
|
52
|
+
endSec: number;
|
|
53
|
+
/** Share of the samples inside its own window that saw it. */
|
|
54
|
+
confidence: number;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export interface BandScore {
|
|
58
|
+
edge: number;
|
|
59
|
+
bimodal: number;
|
|
60
|
+
stroke: number;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
export interface SourceTextScan {
|
|
64
|
+
regions: TextRegion[];
|
|
65
|
+
framesSampled: number;
|
|
66
|
+
/** True when the user asserted the source is edited, skipping detection. */
|
|
67
|
+
assumed: boolean;
|
|
68
|
+
/** Per-sample, per-band measurements — kept so thresholds stay evidence-based. */
|
|
69
|
+
debug?: Array<{ timeSec: number; bands: BandScore[] }>;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Analysis width. The HEIGHT is whatever the source's aspect gives, rather
|
|
74
|
+
* than a fixed 9:16 — all three signals below are geometric, and squeezing a
|
|
75
|
+
* 16:9 frame into a portrait box turns every glyph stroke into a sliver and
|
|
76
|
+
* every horizontal run into a short one. That is the shape of text, so a
|
|
77
|
+
* stretched landscape source would score as text everywhere.
|
|
78
|
+
*/
|
|
79
|
+
const DET_W = 240;
|
|
80
|
+
/** Rows of the analysis grid — bands are the unit, since text runs across. */
|
|
81
|
+
export const BANDS = 24;
|
|
82
|
+
/** Luminance step that counts as a glyph edge. */
|
|
83
|
+
const EDGE_THRESHOLD = 42;
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* Thresholds, set from measurements on a reproduction of the §32 clip
|
|
87
|
+
* (white-on-black title over a colour-bar background):
|
|
88
|
+
*
|
|
89
|
+
* band edge bimodal stroke
|
|
90
|
+
* title 0.345 0.76 high
|
|
91
|
+
* colour bars 0.042 0.33 low (few, very wide runs)
|
|
92
|
+
* checkerboard 0.069 0.30 high (dense, but low contrast)
|
|
93
|
+
*
|
|
94
|
+
* Each threshold sits in the gap, not at the edge of the noise.
|
|
95
|
+
*/
|
|
96
|
+
const BAND_EDGE_RATIO = 0.12;
|
|
97
|
+
const BAND_BIMODALITY = 0.5;
|
|
98
|
+
const BAND_STROKE = 0.25;
|
|
99
|
+
|
|
100
|
+
/** A row needs at least this many transitions to look like a line of glyphs. */
|
|
101
|
+
const MIN_ROW_TRANSITIONS = 6;
|
|
102
|
+
/** …and its runs must be short relative to the frame: glyphs, not bars. */
|
|
103
|
+
const MAX_STROKE_FRACTION = 1 / 12;
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Three scores per band.
|
|
107
|
+
*
|
|
108
|
+
* - `edge`: share of pixels sitting on a horizontal luminance step.
|
|
109
|
+
* - `bimodal`: share of pixels at the luminance extremes.
|
|
110
|
+
* - `stroke`: share of ROWS whose transitions are many and closely spaced.
|
|
111
|
+
*
|
|
112
|
+
* The third is what separates text from a colour-bar test pattern, which is
|
|
113
|
+
* every bit as bimodal as white-on-black type but is a handful of enormous
|
|
114
|
+
* runs rather than dozens of narrow ones.
|
|
115
|
+
*/
|
|
116
|
+
export function bandScores(pixels: Uint8Array, w: number, h: number): BandScore[] {
|
|
117
|
+
const bandHeight = Math.max(1, Math.floor(h / BANDS));
|
|
118
|
+
const maxRun = w * MAX_STROKE_FRACTION;
|
|
119
|
+
const out: BandScore[] = [];
|
|
120
|
+
for (let b = 0; b < BANDS; b++) {
|
|
121
|
+
const y0 = b * bandHeight;
|
|
122
|
+
const y1 = Math.min(h, y0 + bandHeight);
|
|
123
|
+
let edges = 0;
|
|
124
|
+
let extreme = 0;
|
|
125
|
+
let count = 0;
|
|
126
|
+
let strokeRows = 0;
|
|
127
|
+
let rows = 0;
|
|
128
|
+
for (let y = y0; y < y1; y++) {
|
|
129
|
+
let transitions = 0;
|
|
130
|
+
let lastTransition = 0;
|
|
131
|
+
let shortRuns = 0;
|
|
132
|
+
for (let x = 1; x < w - 1; x++) {
|
|
133
|
+
const i = y * w + x;
|
|
134
|
+
const v = pixels[i]!;
|
|
135
|
+
if (v <= 48 || v >= 207) extreme++;
|
|
136
|
+
count++;
|
|
137
|
+
if (Math.abs(pixels[i + 1]! - pixels[i - 1]!) >= EDGE_THRESHOLD) {
|
|
138
|
+
edges++;
|
|
139
|
+
if (x - lastTransition > 1) {
|
|
140
|
+
transitions++;
|
|
141
|
+
if (x - lastTransition <= maxRun) shortRuns++;
|
|
142
|
+
lastTransition = x;
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
rows++;
|
|
147
|
+
if (transitions >= MIN_ROW_TRANSITIONS && shortRuns >= transitions * 0.6) strokeRows++;
|
|
148
|
+
}
|
|
149
|
+
out.push({
|
|
150
|
+
edge: count > 0 ? edges / count : 0,
|
|
151
|
+
bimodal: count > 0 ? extreme / count : 0,
|
|
152
|
+
stroke: rows > 0 ? strokeRows / rows : 0,
|
|
153
|
+
});
|
|
154
|
+
}
|
|
155
|
+
return out;
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
/** Does this band look like burned-in text? All three signals must agree. */
|
|
159
|
+
export function bandIsText(s: BandScore): boolean {
|
|
160
|
+
return s.edge >= BAND_EDGE_RATIO && s.bimodal >= BAND_BIMODALITY && s.stroke >= BAND_STROKE;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* Turn per-sample band occupancy into time-scoped regions.
|
|
165
|
+
*
|
|
166
|
+
* Consecutive busy samples in a band become one region spanning their window;
|
|
167
|
+
* vertically adjacent bands sharing a window merge into a block. No global
|
|
168
|
+
* persistence vote — a title that runs a third of the clip is still a title,
|
|
169
|
+
* it just conflicts with a third of the scenes.
|
|
170
|
+
*/
|
|
171
|
+
export function regionsFromSamples(
|
|
172
|
+
samples: Array<{ timeSec: number; busy: boolean[] }>,
|
|
173
|
+
halfStepSec: number,
|
|
174
|
+
): TextRegion[] {
|
|
175
|
+
const perBand: TextRegion[] = [];
|
|
176
|
+
for (let b = 0; b < BANDS; b++) {
|
|
177
|
+
let runStart: number | null = null;
|
|
178
|
+
let seen = 0;
|
|
179
|
+
const flush = (endIdx: number) => {
|
|
180
|
+
if (runStart === null) return;
|
|
181
|
+
perBand.push({
|
|
182
|
+
x: 0,
|
|
183
|
+
y: b / BANDS,
|
|
184
|
+
w: 1,
|
|
185
|
+
h: 1 / BANDS,
|
|
186
|
+
startSec: Math.max(0, samples[runStart]!.timeSec - halfStepSec),
|
|
187
|
+
endSec: samples[endIdx]!.timeSec + halfStepSec,
|
|
188
|
+
confidence: seen / (endIdx - runStart + 1),
|
|
189
|
+
});
|
|
190
|
+
runStart = null;
|
|
191
|
+
seen = 0;
|
|
192
|
+
};
|
|
193
|
+
for (let i = 0; i < samples.length; i++) {
|
|
194
|
+
if (samples[i]!.busy[b]) {
|
|
195
|
+
if (runStart === null) runStart = i;
|
|
196
|
+
seen++;
|
|
197
|
+
} else if (runStart !== null) {
|
|
198
|
+
flush(i - 1);
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
if (runStart !== null) flush(samples.length - 1);
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
// Merge vertically adjacent bands whose windows overlap.
|
|
205
|
+
perBand.sort((a, b) => a.y - b.y || a.startSec - b.startSec);
|
|
206
|
+
const merged: TextRegion[] = [];
|
|
207
|
+
for (const r of perBand) {
|
|
208
|
+
const prev = merged[merged.length - 1];
|
|
209
|
+
const adjacent = prev && Math.abs(prev.y + prev.h - r.y) < 1e-9;
|
|
210
|
+
const overlaps = prev && r.startSec < prev.endSec && prev.startSec < r.endSec;
|
|
211
|
+
if (prev && adjacent && overlaps) {
|
|
212
|
+
prev.h += r.h;
|
|
213
|
+
prev.startSec = Math.min(prev.startSec, r.startSec);
|
|
214
|
+
prev.endSec = Math.max(prev.endSec, r.endSec);
|
|
215
|
+
prev.confidence = Math.min(prev.confidence, r.confidence);
|
|
216
|
+
} else {
|
|
217
|
+
merged.push({ ...r });
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
// Pad each merged region by one band. Detection localizes GLYPHS, but the
|
|
222
|
+
// graphic behind them — the rounded plate a title sits on — reaches past the
|
|
223
|
+
// last row of type, and it is the PLATE a crop visibly slices. Measured on
|
|
224
|
+
// the real reel: glyphs at 17-25% of the source, the black box from ~12.5%,
|
|
225
|
+
// and the rendered crop cut the box while clearing the text. One band is the
|
|
226
|
+
// detector's own resolution, so this claims no more precision than the
|
|
227
|
+
// measurement has. Padding happens after merging so it cannot fuse regions
|
|
228
|
+
// that the evidence kept apart.
|
|
229
|
+
const pad = 1 / BANDS;
|
|
230
|
+
for (const r of merged) {
|
|
231
|
+
const top = Math.max(0, r.y - pad);
|
|
232
|
+
r.h = Math.min(1, r.y + r.h + pad) - top;
|
|
233
|
+
r.y = top;
|
|
234
|
+
}
|
|
235
|
+
return merged;
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
/**
|
|
239
|
+
* Bands a conservative run assumes are occupied when detection is skipped
|
|
240
|
+
* (`--source-is-edited`). Burned-in titles sit in the upper third and burned-in
|
|
241
|
+
* captions in the lower-middle — the two places an editor puts them, and the
|
|
242
|
+
* two places ossclip most wants to draw. Assumed regions span the whole clip,
|
|
243
|
+
* because without detection there is no way to know when they are up.
|
|
244
|
+
*/
|
|
245
|
+
export const ASSUMED_EDITED_REGIONS: TextRegion[] = [
|
|
246
|
+
{ x: 0, y: 0.12, w: 1, h: 0.2, startSec: 0, endSec: Number.POSITIVE_INFINITY, confidence: 1 },
|
|
247
|
+
{ x: 0, y: 0.66, w: 1, h: 0.12, startSec: 0, endSec: Number.POSITIVE_INFINITY, confidence: 1 },
|
|
248
|
+
];
|
|
249
|
+
|
|
250
|
+
export interface ScanSourceTextOptions {
|
|
251
|
+
samples?: number;
|
|
252
|
+
cacheDir?: string;
|
|
253
|
+
/** Skip detection and assume the conservative regions above. */
|
|
254
|
+
assumeEdited?: boolean;
|
|
255
|
+
/**
|
|
256
|
+
* ffmpeg filter trimming the source to its content rect (PLAN Task 7).
|
|
257
|
+
* Letterbox bars are hard black edges — exactly what the density signal
|
|
258
|
+
* fires on — and regions must be fractions of the frame that RENDERS, which
|
|
259
|
+
* is the cropped one.
|
|
260
|
+
*/
|
|
261
|
+
cropVf?: string;
|
|
262
|
+
/** Extra cache key for when the scanned file is the normalized bake. */
|
|
263
|
+
cacheTag?: string;
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
/** Cache format version — bump to invalidate stale scans after a rebuild. */
|
|
267
|
+
const SCAN_VERSION = 3;
|
|
268
|
+
|
|
269
|
+
/**
|
|
270
|
+
* Sample the take and report where — and WHEN — it already has text burned in.
|
|
271
|
+
* Cached in the workdir beside `face.json`: like the face box, this is a
|
|
272
|
+
* property of the source, not of a render.
|
|
273
|
+
*/
|
|
274
|
+
export async function scanSourceText(
|
|
275
|
+
tools: { ffmpegPath: string },
|
|
276
|
+
videoPath: string,
|
|
277
|
+
durationSec: number,
|
|
278
|
+
opts: ScanSourceTextOptions = {},
|
|
279
|
+
): Promise<SourceTextScan> {
|
|
280
|
+
if (opts.assumeEdited) {
|
|
281
|
+
return { regions: ASSUMED_EDITED_REGIONS, framesSampled: 0, assumed: true };
|
|
282
|
+
}
|
|
283
|
+
const cachePath = opts.cacheDir ? join(opts.cacheDir, "source-text.json") : null;
|
|
284
|
+
if (cachePath && existsSync(cachePath)) {
|
|
285
|
+
const cached = JSON.parse(await readFile(cachePath, "utf8")) as SourceTextScan & {
|
|
286
|
+
version?: number;
|
|
287
|
+
cropVf?: string;
|
|
288
|
+
cacheTag?: string;
|
|
289
|
+
};
|
|
290
|
+
// Regions are fractions of the analyzed frame, so a scan made against a
|
|
291
|
+
// different crop (or a different baked file) describes geometry that no
|
|
292
|
+
// longer renders.
|
|
293
|
+
if (
|
|
294
|
+
cached.version === SCAN_VERSION &&
|
|
295
|
+
(cached.cropVf ?? "") === (opts.cropVf ?? "") &&
|
|
296
|
+
(cached.cacheTag ?? "") === (opts.cacheTag ?? "")
|
|
297
|
+
) {
|
|
298
|
+
return cached;
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
// A title can be short; sample densely enough to catch a ~2s one.
|
|
303
|
+
const samples = opts.samples ?? Math.min(40, Math.max(12, Math.round(durationSec / 1.5)));
|
|
304
|
+
const step = durationSec / samples;
|
|
305
|
+
const collected: Array<{ timeSec: number; busy: boolean[] }> = [];
|
|
306
|
+
const debug: SourceTextScan["debug"] = [];
|
|
307
|
+
|
|
308
|
+
for (let i = 0; i < samples; i++) {
|
|
309
|
+
const t = step * (i + 0.5);
|
|
310
|
+
const framePath = join(opts.cacheDir ?? ".", `text-frame-${i}.gray`);
|
|
311
|
+
await run(tools.ffmpegPath, [
|
|
312
|
+
"-v", "error",
|
|
313
|
+
"-ss", t.toFixed(3),
|
|
314
|
+
"-i", videoPath,
|
|
315
|
+
"-frames:v", "1",
|
|
316
|
+
// -2: height follows the source aspect, rounded to an even number.
|
|
317
|
+
"-vf", `${opts.cropVf ? `${opts.cropVf},` : ""}scale=${DET_W}:-2`,
|
|
318
|
+
"-pix_fmt", "gray",
|
|
319
|
+
"-f", "rawvideo",
|
|
320
|
+
"-y", framePath,
|
|
321
|
+
]);
|
|
322
|
+
const pixels = new Uint8Array(await readFile(framePath));
|
|
323
|
+
await unlink(framePath).catch(() => {});
|
|
324
|
+
const detH = Math.floor(pixels.length / DET_W);
|
|
325
|
+
if (detH < BANDS) continue;
|
|
326
|
+
const scores = bandScores(pixels, DET_W, detH);
|
|
327
|
+
collected.push({ timeSec: t, busy: scores.map(bandIsText) });
|
|
328
|
+
debug.push({ timeSec: t, bands: scores });
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
const scan: SourceTextScan = {
|
|
332
|
+
regions: regionsFromSamples(collected, step / 2),
|
|
333
|
+
framesSampled: collected.length,
|
|
334
|
+
assumed: false,
|
|
335
|
+
debug,
|
|
336
|
+
};
|
|
337
|
+
if (cachePath) {
|
|
338
|
+
await writeFile(
|
|
339
|
+
cachePath,
|
|
340
|
+
JSON.stringify(
|
|
341
|
+
{ version: SCAN_VERSION, cropVf: opts.cropVf ?? "", cacheTag: opts.cacheTag ?? "", ...scan },
|
|
342
|
+
null,
|
|
343
|
+
2,
|
|
344
|
+
),
|
|
345
|
+
);
|
|
346
|
+
}
|
|
347
|
+
return scan;
|
|
348
|
+
}
|
package/src/timemap.ts
ADDED
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
import type { Segment, Word } from "./schema";
|
|
2
|
+
|
|
3
|
+
export interface KeptSpan {
|
|
4
|
+
srcIn: number;
|
|
5
|
+
srcOut: number;
|
|
6
|
+
outIn: number;
|
|
7
|
+
outOut: number;
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Source-time ↔ output-time mapping derived from a cutlist.
|
|
12
|
+
*
|
|
13
|
+
* All overlay timings (captions, scenes, SFX) live in OUTPUT time; this is the
|
|
14
|
+
* only place source time is translated. Invariants (property-tested):
|
|
15
|
+
* - spans are sorted, non-overlapping, and contiguous in output time
|
|
16
|
+
* - outputDuration === Σ (srcOut - srcIn) over kept spans
|
|
17
|
+
* - toSource(toOutput(t)) === t for t strictly inside a kept span
|
|
18
|
+
* - toOutput(toSource(o)) === o for any kept output instant (projection identity)
|
|
19
|
+
* - toOutput is monotonically non-decreasing over kept source time
|
|
20
|
+
*
|
|
21
|
+
* Note: an output instant exactly at a cut boundary has TWO source preimages
|
|
22
|
+
* (the end of the span before the cut and the start of the span after);
|
|
23
|
+
* toSource deterministically returns the earlier one.
|
|
24
|
+
*/
|
|
25
|
+
export class TimeMap {
|
|
26
|
+
readonly spans: readonly KeptSpan[];
|
|
27
|
+
readonly outputDuration: number;
|
|
28
|
+
|
|
29
|
+
constructor(cutlist: readonly Segment[]) {
|
|
30
|
+
let prevOut = -Infinity;
|
|
31
|
+
for (const s of cutlist) {
|
|
32
|
+
if (s.srcOut < s.srcIn) throw new Error(`segment ends before it starts: ${s.srcIn}..${s.srcOut}`);
|
|
33
|
+
if (s.srcIn < prevOut) throw new Error(`cutlist segments overlap at ${s.srcIn}`);
|
|
34
|
+
prevOut = s.srcOut;
|
|
35
|
+
}
|
|
36
|
+
const spans: KeptSpan[] = [];
|
|
37
|
+
let out = 0;
|
|
38
|
+
for (const s of cutlist) {
|
|
39
|
+
if (s.kind !== "keep" || s.srcOut <= s.srcIn) continue;
|
|
40
|
+
const dur = s.srcOut - s.srcIn;
|
|
41
|
+
spans.push({ srcIn: s.srcIn, srcOut: s.srcOut, outIn: out, outOut: out + dur });
|
|
42
|
+
out += dur;
|
|
43
|
+
}
|
|
44
|
+
this.spans = spans;
|
|
45
|
+
this.outputDuration = out;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/** Output time for a source instant, or null when the instant was cut. */
|
|
49
|
+
toOutput(tSrc: number): number | null {
|
|
50
|
+
// Exact containment first — a tolerance must never steal an instant that
|
|
51
|
+
// exactly belongs to another span (removed segments can be arbitrarily short).
|
|
52
|
+
for (const sp of this.spans) {
|
|
53
|
+
if (tSrc >= sp.srcIn && tSrc <= sp.srcOut) return sp.outIn + (tSrc - sp.srcIn);
|
|
54
|
+
}
|
|
55
|
+
// Then tolerate float-ulp overshoot at edges: clamp into the nearest span
|
|
56
|
+
// only when the instant is within EPS of it.
|
|
57
|
+
const EPS = 1e-9;
|
|
58
|
+
let best: KeptSpan | null = null;
|
|
59
|
+
let bestDist = Infinity;
|
|
60
|
+
for (const sp of this.spans) {
|
|
61
|
+
const dist = tSrc < sp.srcIn ? sp.srcIn - tSrc : tSrc - sp.srcOut;
|
|
62
|
+
if (dist < bestDist) {
|
|
63
|
+
bestDist = dist;
|
|
64
|
+
best = sp;
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
if (best && bestDist <= EPS) {
|
|
68
|
+
const clamped = Math.min(Math.max(tSrc, best.srcIn), best.srcOut);
|
|
69
|
+
return best.outIn + (clamped - best.srcIn);
|
|
70
|
+
}
|
|
71
|
+
return null;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* Output time for a source instant, clamping instants that fall in removed
|
|
76
|
+
* regions to the nearest kept edge. Used for caption/overlay boundaries.
|
|
77
|
+
*/
|
|
78
|
+
toOutputClamped(tSrc: number): number {
|
|
79
|
+
const exact = this.toOutput(tSrc);
|
|
80
|
+
if (exact !== null) return exact;
|
|
81
|
+
let best = 0;
|
|
82
|
+
for (const sp of this.spans) {
|
|
83
|
+
if (sp.srcOut <= tSrc) best = sp.outOut;
|
|
84
|
+
else if (sp.srcIn >= tSrc) return sp.outIn;
|
|
85
|
+
}
|
|
86
|
+
return best;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/** Source time for an output instant. Output time is contiguous, so this is total. */
|
|
90
|
+
toSource(tOut: number): number {
|
|
91
|
+
const spans = this.spans;
|
|
92
|
+
if (spans.length === 0) return 0;
|
|
93
|
+
const first = spans[0]!;
|
|
94
|
+
if (tOut <= first.outIn) return first.srcIn;
|
|
95
|
+
for (const sp of spans) {
|
|
96
|
+
if (tOut >= sp.outIn && tOut <= sp.outOut) return sp.srcIn + (tOut - sp.outIn);
|
|
97
|
+
}
|
|
98
|
+
return spans[spans.length - 1]!.srcOut;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* Map a word into output time. Returns null when the word was entirely cut
|
|
103
|
+
* (e.g. a removed filler); ends are clamped when a cut clips the word edge.
|
|
104
|
+
*/
|
|
105
|
+
mapWord(w: Word): { start: number; end: number } | null {
|
|
106
|
+
const mid = (w.start + w.end) / 2;
|
|
107
|
+
if (this.toOutput(mid) === null && this.toOutput(w.start) === null && this.toOutput(w.end) === null) {
|
|
108
|
+
return null;
|
|
109
|
+
}
|
|
110
|
+
const start = this.toOutputClamped(w.start);
|
|
111
|
+
const end = this.toOutputClamped(w.end);
|
|
112
|
+
if (end <= start) return null;
|
|
113
|
+
return { start, end };
|
|
114
|
+
}
|
|
115
|
+
}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import { readFile } from "node:fs/promises";
|
|
2
|
+
import { run } from "./exec";
|
|
3
|
+
import type { Transcript, Word } from "./schema";
|
|
4
|
+
|
|
5
|
+
/** Shape of whisper.cpp's `-oj` JSON output (the fields we consume). */
|
|
6
|
+
export interface WhisperJson {
|
|
7
|
+
result?: { language?: string };
|
|
8
|
+
transcription: Array<{
|
|
9
|
+
offsets: { from: number; to: number }; // milliseconds
|
|
10
|
+
text: string;
|
|
11
|
+
}>;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
const NOISE_TOKEN = /^[[(].*[\])]$/; // [BLANK_AUDIO], (buzzing), [MUSIC] …
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* Convert whisper.cpp `-ml 1` segments (≈ one token each) into words.
|
|
18
|
+
* Tokens beginning with whitespace start a new word; bare continuations
|
|
19
|
+
* ("'s", "ing") merge into the previous word. Bracketed noise markers drop.
|
|
20
|
+
*/
|
|
21
|
+
export function parseWhisperJson(json: WhisperJson): Transcript {
|
|
22
|
+
const words: Word[] = [];
|
|
23
|
+
for (const seg of json.transcription ?? []) {
|
|
24
|
+
const raw = seg.text;
|
|
25
|
+
if (!raw || !raw.trim()) continue;
|
|
26
|
+
const text = raw.trim();
|
|
27
|
+
if (NOISE_TOKEN.test(text)) continue;
|
|
28
|
+
const startsWord = /^\s/.test(raw) || words.length === 0;
|
|
29
|
+
const start = seg.offsets.from / 1000;
|
|
30
|
+
const end = seg.offsets.to / 1000;
|
|
31
|
+
const last = words[words.length - 1];
|
|
32
|
+
if (!startsWord && last) {
|
|
33
|
+
last.text += text;
|
|
34
|
+
last.end = Math.max(last.end, end);
|
|
35
|
+
} else {
|
|
36
|
+
words.push({ text, start, end });
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
// Whisper occasionally emits zero-length or inverted stamps; repair minimally.
|
|
40
|
+
for (let i = 0; i < words.length; i++) {
|
|
41
|
+
const w = words[i]!;
|
|
42
|
+
if (w.end <= w.start) w.end = w.start + 0.05;
|
|
43
|
+
const next = words[i + 1];
|
|
44
|
+
if (next && next.start < w.end) next.start = w.end;
|
|
45
|
+
}
|
|
46
|
+
return { language: json.result?.language ?? "en", words };
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
export interface WhisperOptions {
|
|
50
|
+
whisperPath: string;
|
|
51
|
+
modelPath: string;
|
|
52
|
+
/** Output base path; whisper writes `${outBase}.json`. */
|
|
53
|
+
outBase: string;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
export async function runWhisper(opts: WhisperOptions, wavPath: string): Promise<Transcript> {
|
|
57
|
+
await run(opts.whisperPath, [
|
|
58
|
+
"-m", opts.modelPath,
|
|
59
|
+
"-f", wavPath,
|
|
60
|
+
"-oj",
|
|
61
|
+
"-of", opts.outBase,
|
|
62
|
+
"-ml", "1",
|
|
63
|
+
"--no-prints",
|
|
64
|
+
]);
|
|
65
|
+
const json = JSON.parse(await readFile(`${opts.outBase}.json`, "utf8")) as WhisperJson;
|
|
66
|
+
return parseWhisperJson(json);
|
|
67
|
+
}
|
package/src/zoom.ts
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Idle camera movement (FINDINGS §15): the cut-driven punch-in only fires at
|
|
3
|
+
* cuts, so a clean take sits visually static for 8–12 s at a time. This is the
|
|
4
|
+
* independent driver.
|
|
5
|
+
*
|
|
6
|
+
* ## Why this was rewritten (2026-07-28)
|
|
7
|
+
*
|
|
8
|
+
* The first version reversed direction at every speech-phrase boundary, on the
|
|
9
|
+
* theory that a phrase break is a natural place for the camera to turn around.
|
|
10
|
+
* On the author's own 64s take that found 24 boundaries and duly produced 24
|
|
11
|
+
* reversals, and the verdict was immediate: "the weird constant zooming in and
|
|
12
|
+
* zooming out". Reversing at a boundary is defensible ONCE; doing it every two
|
|
13
|
+
* seconds for a minute reads as a wobble, not as camera work. The bug was the
|
|
14
|
+
* contract, not the boundary detection — so the boundary machinery is gone
|
|
15
|
+
* rather than tuned.
|
|
16
|
+
*
|
|
17
|
+
* ## What it does now
|
|
18
|
+
*
|
|
19
|
+
* Within one cut-free CLIP the zoom moves in exactly one direction: a cosine
|
|
20
|
+
* ease from 1 to `maxScale` over `rampSec`, then a HOLD at `maxScale` for the
|
|
21
|
+
* rest of the clip. A cut resets it to 1, which is the one place a step is
|
|
22
|
+
* already justified — `EdlVideo`'s punch-in steps there too, and the frame
|
|
23
|
+
* changes anyway.
|
|
24
|
+
*
|
|
25
|
+
* Holding after the ramp is what keeps the move readable. Stretching 1 → 1.08
|
|
26
|
+
* across a 64s take is ~0.12%/s, which is indistinguishable from no motion; a
|
|
27
|
+
* bounded ramp followed by a hold is a slow push that arrives somewhere and
|
|
28
|
+
* stays — the author's "zoomed-out to zoomed-in, then keep that perspective
|
|
29
|
+
* consistent", composed from their own two options.
|
|
30
|
+
*
|
|
31
|
+
* A clip shorter than `rampSec` gets a PARTIAL push at the same rate rather
|
|
32
|
+
* than a compressed full one, so a 2s clip and a 20s clip move at the same
|
|
33
|
+
* speed. Making short clips complete the push would make them zoom visibly
|
|
34
|
+
* faster, which is the oscillation problem in a new costume.
|
|
35
|
+
*
|
|
36
|
+
* `maxScale: 1` disables the driver outright (the author's "no zoom" option)
|
|
37
|
+
* without needing a separate flag.
|
|
38
|
+
*/
|
|
39
|
+
|
|
40
|
+
export interface ZoomSegment {
|
|
41
|
+
startSec: number;
|
|
42
|
+
endSec: number;
|
|
43
|
+
from: number;
|
|
44
|
+
to: number;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
export interface ZoomPlan {
|
|
48
|
+
segments: ZoomSegment[];
|
|
49
|
+
/** How many cut-free clips the plan covers — logged, never inferred. */
|
|
50
|
+
clips: number;
|
|
51
|
+
/** The ramp actually used, so the log can't drift from the behaviour. */
|
|
52
|
+
rampSec: number;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export interface ZoomPlanOptions {
|
|
56
|
+
/** Zoomed-in extreme; the other extreme is 1. `1` disables the zoom. */
|
|
57
|
+
maxScale?: number;
|
|
58
|
+
/**
|
|
59
|
+
* Clip starts in OUTPUT time — the kept spans' `outIn`, i.e. every point the
|
|
60
|
+
* source jumps. A missing or empty list means the take is one clip, which is
|
|
61
|
+
* the correct reading of a cutlist that removed nothing.
|
|
62
|
+
*/
|
|
63
|
+
clipStarts?: readonly number[];
|
|
64
|
+
/** Seconds the push takes to arrive before it holds. */
|
|
65
|
+
rampSec?: number;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* Zoom amplitude, exported so the stage can budget crop margins against it.
|
|
70
|
+
*
|
|
71
|
+
* 5%, not 8%: at 8% the push was eating enough of an already-tight frame to
|
|
72
|
+
* clip a forehead on a close-up, and the amount a viewer should register is
|
|
73
|
+
* "the camera is alive", not "the camera moved". Every crop budget that cites
|
|
74
|
+
* this constant tightens with it rather than needing its own edit.
|
|
75
|
+
*/
|
|
76
|
+
export const ZOOM_MAX_SCALE = 1.05;
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* How long the push takes. Long enough that the movement is never noticed as
|
|
80
|
+
* movement, short enough that it has arrived while the viewer is still on the
|
|
81
|
+
* hook — and, deliberately, far shorter than a typical clip so most of a clip
|
|
82
|
+
* is the settled perspective rather than a drift.
|
|
83
|
+
*/
|
|
84
|
+
export const ZOOM_RAMP_SEC = 8;
|
|
85
|
+
|
|
86
|
+
/** Clip starts, cleaned: in range, unique, sorted, and always including 0. */
|
|
87
|
+
function clipBoundaries(starts: readonly number[] | undefined, duration: number): number[] {
|
|
88
|
+
const seen = new Set<number>([0]);
|
|
89
|
+
for (const t of starts ?? []) {
|
|
90
|
+
if (Number.isFinite(t) && t > 0 && t < duration) seen.add(t);
|
|
91
|
+
}
|
|
92
|
+
return [...seen].sort((a, b) => a - b);
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
export function buildZoomPlan(
|
|
96
|
+
outputDurationSec: number,
|
|
97
|
+
opts: ZoomPlanOptions = {},
|
|
98
|
+
): ZoomPlan {
|
|
99
|
+
const maxScale = opts.maxScale ?? ZOOM_MAX_SCALE;
|
|
100
|
+
const rampSec = opts.rampSec ?? ZOOM_RAMP_SEC;
|
|
101
|
+
if (outputDurationSec <= 0) return { segments: [], clips: 0, rampSec };
|
|
102
|
+
|
|
103
|
+
const starts = clipBoundaries(opts.clipStarts, outputDurationSec);
|
|
104
|
+
const segments: ZoomSegment[] = [];
|
|
105
|
+
|
|
106
|
+
for (let i = 0; i < starts.length; i++) {
|
|
107
|
+
const start = starts[i]!;
|
|
108
|
+
const end = i + 1 < starts.length ? starts[i + 1]! : outputDurationSec;
|
|
109
|
+
const length = end - start;
|
|
110
|
+
if (length <= 1e-9) continue;
|
|
111
|
+
|
|
112
|
+
const rampEnd = Math.min(start + rampSec, end);
|
|
113
|
+
// Same rate for every clip: a short clip stops partway up rather than
|
|
114
|
+
// racing to the top.
|
|
115
|
+
const reached = 1 + (maxScale - 1) * Math.min(1, (rampEnd - start) / Math.max(rampSec, 1e-9));
|
|
116
|
+
|
|
117
|
+
segments.push({ startSec: start, endSec: rampEnd, from: 1, to: reached });
|
|
118
|
+
if (end - rampEnd > 1e-9) {
|
|
119
|
+
segments.push({ startSec: rampEnd, endSec: end, from: reached, to: reached });
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
return { segments, clips: starts.length, rampSec };
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* Scale at output time t — cosine-eased across each segment (never linear), so
|
|
128
|
+
* the push starts and settles gently. 1 outside the plan.
|
|
129
|
+
*
|
|
130
|
+
* A hold segment has `from === to`, which the same easing renders as a
|
|
131
|
+
* constant; no special case is needed and none should be added, because the
|
|
132
|
+
* ramp/hold boundary must not be a place where two code paths could disagree.
|
|
133
|
+
*
|
|
134
|
+
* Segments are matched HALF-OPEN, `[start, end)`. Under the old oscillating
|
|
135
|
+
* contract this was academic — neighbouring segments shared a value at every
|
|
136
|
+
* boundary — but a cut is a boundary where they deliberately disagree: the
|
|
137
|
+
* previous clip holds at `maxScale` and the next starts at 1. Closed matching
|
|
138
|
+
* let the earlier segment win, so the first frame of a new clip rendered the
|
|
139
|
+
* PREVIOUS clip's zoom and the reset appeared one frame late. The final instant
|
|
140
|
+
* of the plan has no following segment and so is served by the last one.
|
|
141
|
+
*/
|
|
142
|
+
export function zoomScaleAt(plan: readonly ZoomSegment[], tSec: number): number {
|
|
143
|
+
for (const seg of plan) {
|
|
144
|
+
if (tSec >= seg.startSec && tSec < seg.endSec) {
|
|
145
|
+
const span = seg.endSec - seg.startSec;
|
|
146
|
+
const p = span > 0 ? (tSec - seg.startSec) / span : 1;
|
|
147
|
+
const eased = 0.5 - 0.5 * Math.cos(Math.PI * p);
|
|
148
|
+
return seg.from + (seg.to - seg.from) * eased;
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
const last = plan[plan.length - 1];
|
|
152
|
+
if (last && tSec === last.endSec) return last.to;
|
|
153
|
+
return 1;
|
|
154
|
+
}
|