@ossclip/core 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,397 @@
1
+ import { run } from "./exec";
2
+ import type { ContentRectSegment } from "./content-rect";
3
+ import type { WindowFace } from "./face";
4
+
5
+ /**
6
+ * Framing normalization for a mixed-framing source (option (a), chosen with
7
+ * the author 2026-07-28).
8
+ *
9
+ * A source that alternates a letterboxed landscape strip with full-bleed
10
+ * portrait has no single framing — and rendering each segment's own framing
11
+ * cover-filled produced a ~3× apparent zoom jump at every boundary, nine
12
+ * times a minute on the motivating clip ("weird zoom-outs and zoom-ins").
13
+ * Smoothing the boundaries cannot fix that; the output would still alternate
14
+ * between two shots.
15
+ *
16
+ * The fix is editorial, applied at bake time: pick ONE field of view — the
17
+ * tightest the source ever shows, i.e. the strip, since the strip's pixels
18
+ * are all those stretches have — and crop every other segment down to a
19
+ * window of that same shape, placed on the measured face. The result is a
20
+ * single, uniform landscape source with constant apparent framing, and the
21
+ * ENTIRE downstream pipeline (face bias, source-text, cover, layouts, zoom)
22
+ * runs its ordinary uniform-source path on it. Tight but stable, by choice.
23
+ *
24
+ * When even the strip cannot cover the output frame without excessive
25
+ * upscaling, normalization refuses (`ok: false`) and the caller falls back to
26
+ * render-time FIT — the strip shown at its natural size rather than
27
+ * fake-zoomed (option (b)).
28
+ */
29
+
30
+ /** One baked stretch: this window of the source, scaled to the canvas. */
31
+ export interface NormalizeSegment {
32
+ startSec: number;
33
+ endSec: number;
34
+ /** Crop window in SOURCE pixels — always the canvas's aspect. */
35
+ window: { x: number; y: number; w: number; h: number };
36
+ }
37
+
38
+ export interface NormalizePlan {
39
+ /** The common frame every segment is cropped+scaled to. */
40
+ canvas: { width: number; height: number };
41
+ segments: NormalizeSegment[];
42
+ /**
43
+ * Per segment, the face's height as a fraction of the CANVAS after baking —
44
+ * what the framing actually achieved, as opposed to what it aimed for. This
45
+ * is the input to `assessCueFraming`, and eventually to telling the producer
46
+ * which windows can host which layouts.
47
+ */
48
+ faceFracOfCanvas: number[];
49
+ /**
50
+ * The upscale a full-bleed cover of the OUTPUT implies. The quality gate:
51
+ * past `MAX_NORMALIZE_UPSCALE` the picture would be visibly soft, and a
52
+ * soft fake is worse than an honest fit.
53
+ */
54
+ coverUpscale: number;
55
+ ok: boolean;
56
+ }
57
+
58
+ /**
59
+ * Ceiling on how far the canvas may be upscaled when a full-bleed layout
60
+ * covers the output with it. The motivating clip sits at 1920/808 ≈ 2.38 —
61
+ * soft but within reel norms; a strip much shorter than that is not worth
62
+ * faking a full-frame shot from.
63
+ */
64
+ export const MAX_NORMALIZE_UPSCALE = 2.6;
65
+
66
+ /**
67
+ * A head is about 1.55x the detector's face box tall — the box bounds eyes,
68
+ * nose and mouth, and `stage.ts` models the crown at 0.35x above it and the
69
+ * chin at 0.2x below (FINDINGS §19).
70
+ */
71
+ const HEAD_PER_FACE = 1.55;
72
+ const HEAD_ABOVE = 0.85;
73
+ const HEAD_BELOW = 0.7;
74
+
75
+ /**
76
+ * The tightest the framing may ever get, as face-box height over frame height.
77
+ *
78
+ * Derived, not tuned: the head must survive the idle zoom with margin left, so
79
+ * `HEAD_PER_FACE x F x ZOOM_MAX_SCALE <= 1 - 2 x margin`. At a 6% margin top
80
+ * and bottom that is `1.55 x F x 1.05 <= 0.88`, i.e. F <= 0.54.
81
+ *
82
+ * This is a CEILING on cropping in, never a target to reach. A segment whose
83
+ * face is already larger than this is left at its own framing rather than
84
+ * cropped further — there is no version of "consistent framing" worth cutting
85
+ * someone's forehead off for, which is exactly what the previous round did.
86
+ */
87
+ export const MAX_FACE_FRACTION = 0.54;
88
+
89
+ const even = (v: number): number => 2 * Math.floor(v / 2);
90
+ const clamp = (v: number, lo: number, hi: number): number => Math.min(Math.max(v, lo), hi);
91
+
92
+ const median = (xs: number[]): number => {
93
+ const s = [...xs].sort((a, b) => a - b);
94
+ const m = Math.floor(s.length / 2);
95
+ return s.length % 2 ? s[m]! : (s[m - 1]! + s[m]!) / 2;
96
+ };
97
+
98
+ /**
99
+ * Decide the canvas and each segment's crop window.
100
+ *
101
+ * `faces` is parallel to `timeline` — each face in ITS segment rect's own
102
+ * fractions (`measureFaceInWindows`).
103
+ *
104
+ * The window is sized by the FACE, not by a constant rect. Equalizing the
105
+ * canvas alone is not enough, and the author's clip proved it: its two
106
+ * framings are the same camera shot presented differently — the letterboxed
107
+ * strip is the whole landscape frame, the full-bleed stretches are a zoomed
108
+ * crop of it. So the face measures 0.28-0.44 of the frame in one state and
109
+ * 0.48-0.57 in the other, and cropping both to a constant-height window put
110
+ * the face at 108% of output height in the full-bleed stretches (head taller
111
+ * than the frame) against 57% in the strips. Same subject, wildly different
112
+ * size, at every boundary.
113
+ *
114
+ * Sizing each window as `faceHeight / targetFraction` makes the SUBJECT the
115
+ * constant instead, which is what "one consistent framing" has to mean. The
116
+ * target is the MEDIAN measured fraction: a segment whose face is smaller
117
+ * than the target crops in to match, and one whose face is larger can only
118
+ * zoom out as far as its own rect — clamping there rather than inventing
119
+ * pixels. The median (not the max) is what keeps that clamping rare and the
120
+ * upscale inside the quality gate.
121
+ */
122
+ export function planNormalization(
123
+ timeline: readonly ContentRectSegment[],
124
+ faces: ReadonlyArray<WindowFace | null>,
125
+ output: { width: number; height: number },
126
+ ): NormalizePlan {
127
+ const boxed = timeline.filter((s) => !s.rect.full);
128
+ // Callers only reach here for a mixed source, but refuse rather than crash.
129
+ if (boxed.length === 0 || timeline.length < 2) {
130
+ return {
131
+ canvas: { width: 0, height: 0 },
132
+ segments: [],
133
+ faceFracOfCanvas: [],
134
+ coverUpscale: Infinity,
135
+ ok: false,
136
+ };
137
+ }
138
+
139
+ /**
140
+ * The LARGEST face fraction in each segment, not the median. A window sized
141
+ * on the median is correct only at the median moment: the author's clip
142
+ * moves 29%-48% inside one 12s stretch, and sizing on 34% put the head past
143
+ * the frame edge whenever they leaned in — which is precisely the frame they
144
+ * flagged. Sizing on the maximum makes the tightest moment the safe one and
145
+ * every other moment merely roomier.
146
+ */
147
+ const measured = timeline.map((_, i) => faces[i]?.sizeFracMax ?? faces[i]?.sizeFrac ?? null);
148
+ const known = measured.filter((v): v is number => v !== null);
149
+
150
+ // ---- Window heights ------------------------------------------------------
151
+ // Without a single measurement there is no subject to hold constant, so the
152
+ // rect-shaped fallback stands: the tightest field of view, uniformly.
153
+ const target = known.length > 0 ? Math.min(median(known), MAX_FACE_FRACTION) : null;
154
+ const rectShapedHeights = (): number[] => {
155
+ const canvasRect = boxed.reduce((a, b) => (b.rect.h < a.rect.h ? b : a)).rect;
156
+ const a = canvasRect.w / canvasRect.h;
157
+ return timeline.map((s) => even(Math.min(s.rect.w, s.rect.h * a) / a));
158
+ };
159
+ const windowHeights =
160
+ target === null
161
+ ? rectShapedHeights()
162
+ : timeline.map((s, i) => {
163
+ // An unmeasured segment inherits the median fraction of the segments
164
+ // framed like it (same rect height), so it is sized in ITS OWN class
165
+ // rather than averaged across two different shots.
166
+ const sameClass = timeline.flatMap((o, j) =>
167
+ measured[j] !== null && Math.abs(o.rect.h - s.rect.h) <= 2 ? [measured[j]!] : [],
168
+ );
169
+ const frac = measured[i] ?? (sameClass.length > 0 ? median(sameClass) : median(known));
170
+ return even(clamp((frac * s.rect.h) / target, 16, s.rect.h));
171
+ });
172
+
173
+ // ---- Canvas --------------------------------------------------------------
174
+ // The widest aspect every window can actually hold. Wider than the output's
175
+ // own aspect leaves the stage some horizontal freedom for the face bias;
176
+ // narrower simply means the output crops height, which cover already does.
177
+ const aspect = timeline.reduce(
178
+ (a, s, i) => Math.min(a, s.rect.w / windowHeights[i]!),
179
+ Number.POSITIVE_INFINITY,
180
+ );
181
+ // The smallest window, so baking never upscales — the tightest segment sets
182
+ // the resolution and every other one is downscaled into it.
183
+ const canvasHeight = even(Math.min(...windowHeights));
184
+ const canvas = { width: even(canvasHeight * aspect), height: canvasHeight };
185
+
186
+ // ---- Face placement inside the window ------------------------------------
187
+ // Taken from the segments whose window IS their rect: their framing is the
188
+ // author's own and survives untouched, so it is the one to reproduce.
189
+ let wx = 0;
190
+ let wy = 0;
191
+ let weight = 0;
192
+ timeline.forEach((seg, i) => {
193
+ const f = faces[i];
194
+ if (!f || windowHeights[i]! < seg.rect.h - 2) return;
195
+ const dur = Math.max(1e-6, seg.endSec - seg.startSec);
196
+ wx += f.centerXFrac * dur;
197
+ wy += f.centerYFrac * dur;
198
+ weight += dur;
199
+ });
200
+ const targetX = weight > 0 ? wx / weight : 0.5;
201
+ const targetY = weight > 0 ? wy / weight : 0.45;
202
+
203
+ const segments: NormalizeSegment[] = timeline.map((seg, i) => {
204
+ const r = seg.rect;
205
+ const wH = windowHeights[i]!;
206
+ const wW = even(Math.min(r.w, wH * aspect));
207
+ const f = faces[i];
208
+ // Face position in source px; the rect centre when nothing was measured.
209
+ const faceX = r.x + (f ? f.centerXFrac : 0.5) * r.w;
210
+ const faceY = r.y + (f ? f.centerYFrac : 0.5) * r.h;
211
+ const x = even(clamp(faceX - targetX * wW, r.x, r.x + r.w - wW));
212
+
213
+ let y = clamp(faceY - targetY * wH, r.y, r.y + r.h - wH);
214
+ // Then slide — never resize — so the whole HEAD is inside the window at
215
+ // the segment's largest face, since the aesthetic anchor above is about
216
+ // where the face sits, and this is about not amputating it. Bounded by
217
+ // the rect: if the head genuinely runs past the source's own edge there
218
+ // is nothing to slide toward, and the clamp leaves it where it was.
219
+ if (f) {
220
+ const maxFace = (f.sizeFracMax ?? f.sizeFrac) * r.h;
221
+ const headTop = faceY - HEAD_ABOVE * maxFace;
222
+ const headBottom = faceY + HEAD_BELOW * maxFace;
223
+ if (headBottom - headTop <= wH) {
224
+ y = clamp(y, headBottom - wH, headTop);
225
+ } else {
226
+ // Head taller than the window: centre it, so what is lost is shared
227
+ // between crown and chin instead of taking the whole bite off one end.
228
+ y = (headTop + headBottom) / 2 - wH / 2;
229
+ }
230
+ y = clamp(y, r.y, r.y + r.h - wH);
231
+ }
232
+ return {
233
+ startSec: seg.startSec,
234
+ endSec: seg.endSec,
235
+ window: { x, y: even(y), w: wW, h: wH },
236
+ };
237
+ });
238
+
239
+ // Cover the output with the canvas: for a canvas wider than the output's
240
+ // aspect the height binds, otherwise the width does.
241
+ const coverUpscale =
242
+ canvas.width / canvas.height > output.width / output.height
243
+ ? output.height / canvas.height
244
+ : output.width / canvas.width;
245
+
246
+ // What each segment actually achieved once its window is scaled to the
247
+ // canvas — measured from the plan, never assumed to equal the target: a
248
+ // segment clamped at its own rect lands wherever its rect put it.
249
+ const faceFracOfCanvas = timeline.map((seg, i) => {
250
+ const frac = measured[i];
251
+ if (frac === null || frac === undefined) return target ?? 0;
252
+ return (frac * seg.rect.h) / segments[i]!.window.h;
253
+ });
254
+
255
+ return {
256
+ canvas,
257
+ segments,
258
+ faceFracOfCanvas,
259
+ coverUpscale,
260
+ ok: coverUpscale <= MAX_NORMALIZE_UPSCALE,
261
+ };
262
+ }
263
+
264
+ /** A cue's video slot, in output pixels — `layoutSlots(cue.layout).video.rect`. */
265
+ export interface CueSlot {
266
+ id: string;
267
+ layout: string;
268
+ /** SOURCE seconds, so this intersects the content timeline directly. */
269
+ startSec: number;
270
+ endSec: number;
271
+ slot: { width: number; height: number };
272
+ }
273
+
274
+ export interface FramingIssue {
275
+ cueId: string;
276
+ layout: string;
277
+ /** Face height over the SLOT's height, after cover crops the canvas to it. */
278
+ faceFracOfSlot: number;
279
+ /** The whole head, under the idle zoom. Above 1 it does not fit. */
280
+ headFracOfSlot: number;
281
+ }
282
+
283
+ /**
284
+ * How each scene's slot actually frames the speaker (plan step D).
285
+ *
286
+ * A slot WIDER than the canvas gets cover-cropped vertically, so it shows only
287
+ * `canvasAspect / slotAspect` of the canvas height — and the face grows by the
288
+ * inverse of that. `video-top` is a 1080x806 band against a portrait canvas:
289
+ * it shows ~42% of the canvas height, so a face occupying 44% of the canvas
290
+ * occupies 105% of the band. That is the crown trimming, and it is a property
291
+ * of the LAYOUT, not of the source or of any global constant.
292
+ *
293
+ * This reports it per cue rather than fixing it, deliberately. It is not
294
+ * fixable by cropping: the pixels a wide band wants do not exist in a portrait
295
+ * close-up. It is fixable by not putting a wide band on that moment — which is
296
+ * a producer decision, and this is the evidence it needs.
297
+ */
298
+ /**
299
+ * Face height over a SLOT's height, once cover crops the canvas into it.
300
+ *
301
+ * A slot WIDER than the canvas is cover-cropped vertically and shows only
302
+ * `canvasAspect / slotAspect` of the canvas height, so the face grows by the
303
+ * inverse; a slot no wider than the canvas crops width, and the face's height
304
+ * fraction is unchanged. THE arithmetic behind every framing judgement —
305
+ * per-cue assessment, the producer's brief, and the layout repair pass all
306
+ * call this one function, so they cannot disagree.
307
+ */
308
+ export function faceFracInSlot(
309
+ faceFracOfCanvas: number,
310
+ canvasAspect: number,
311
+ slotAspect: number,
312
+ ): number {
313
+ if (canvasAspect <= 0 || slotAspect <= 0) return faceFracOfCanvas;
314
+ return faceFracOfCanvas / Math.min(1, canvasAspect / slotAspect);
315
+ }
316
+
317
+ /** The whole head, under the idle zoom — above 1 the crop trims it. */
318
+ export function headFracInSlot(
319
+ faceFracOfCanvas: number,
320
+ canvasAspect: number,
321
+ slotAspect: number,
322
+ zoom: number,
323
+ ): number {
324
+ return HEAD_PER_FACE * faceFracInSlot(faceFracOfCanvas, canvasAspect, slotAspect) * zoom;
325
+ }
326
+
327
+ export function assessCueFraming(
328
+ cues: readonly CueSlot[],
329
+ segments: ReadonlyArray<{ startSec: number; endSec: number }>,
330
+ faceFracOfCanvas: readonly number[],
331
+ canvas: { width: number; height: number },
332
+ zoom: number,
333
+ ): FramingIssue[] {
334
+ if (canvas.width <= 0 || canvas.height <= 0) return [];
335
+ const canvasAspect = canvas.width / canvas.height;
336
+ const out: FramingIssue[] = [];
337
+ for (const cue of cues) {
338
+ // The worst framing the cue is on screen for — a cue spanning a boundary
339
+ // is judged by its tightest moment, not by an average of the two.
340
+ let frac = 0;
341
+ segments.forEach((seg, i) => {
342
+ if (seg.startSec < cue.endSec && seg.endSec > cue.startSec) {
343
+ frac = Math.max(frac, faceFracOfCanvas[i] ?? 0);
344
+ }
345
+ });
346
+ if (frac <= 0 || cue.slot.width <= 0 || cue.slot.height <= 0) continue;
347
+ const slotAspect = cue.slot.width / cue.slot.height;
348
+ out.push({
349
+ cueId: cue.id,
350
+ layout: cue.layout,
351
+ faceFracOfSlot: faceFracInSlot(frac, canvasAspect, slotAspect),
352
+ headFracOfSlot: headFracInSlot(frac, canvasAspect, slotAspect, zoom),
353
+ });
354
+ }
355
+ return out;
356
+ }
357
+
358
+ /**
359
+ * The ffmpeg filter graph baking the plan: each segment trimmed, cropped to
360
+ * its window, scaled to the canvas, and the pieces concatenated back into one
361
+ * continuous stream. The segment boundaries partition the source exactly, so
362
+ * the output timeline equals the input's and the untouched audio stays in
363
+ * sync.
364
+ */
365
+ export function normalizationFilterGraph(plan: NormalizePlan): string {
366
+ const parts = plan.segments.map((s, i) => {
367
+ const w = s.window;
368
+ return (
369
+ `[0:v]trim=start=${s.startSec.toFixed(3)}:end=${s.endSec.toFixed(3)},` +
370
+ `setpts=PTS-STARTPTS,crop=${w.w}:${w.h}:${w.x}:${w.y},` +
371
+ `scale=${plan.canvas.width}:${plan.canvas.height}[v${i}]`
372
+ );
373
+ });
374
+ const inputs = plan.segments.map((_, i) => `[v${i}]`).join("");
375
+ return `${parts.join(";")};${inputs}concat=n=${plan.segments.length}:v=1:a=0[v]`;
376
+ }
377
+
378
+ /**
379
+ * Bake the normalized source. Encoded with the mezzanine's own settings
380
+ * (dense keyframes) because it REPLACES the mezzanine — normalizing and then
381
+ * re-encoding for seekability would be two generations of loss for nothing.
382
+ */
383
+ export async function bakeNormalizedSource(
384
+ tools: { ffmpegPath: string },
385
+ input: string,
386
+ plan: NormalizePlan,
387
+ outPath: string,
388
+ ): Promise<void> {
389
+ await run(tools.ffmpegPath, [
390
+ "-y", "-i", input,
391
+ "-filter_complex", normalizationFilterGraph(plan),
392
+ "-map", "[v]", "-map", "0:a?",
393
+ "-c:v", "libx264", "-preset", "veryfast", "-crf", "18", "-g", "30", "-pix_fmt", "yuv420p",
394
+ "-c:a", "aac", "-b:a", "192k",
395
+ outPath,
396
+ ]);
397
+ }