@ossclip/core 0.1.24 → 0.1.26
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/fonts/NotoNastaliqUrdu-Bold.ttf +0 -0
- package/assets/fonts/OFL.txt +93 -0
- package/assets/fonts/README.md +15 -0
- package/package.json +2 -1
- package/src/blooper.ts +91 -8
- package/src/browser.ts +8 -0
- package/src/captions.ts +45 -2
- package/src/concat.ts +95 -8
- package/src/config.ts +110 -0
- package/src/content-rect-detect.ts +16 -4
- package/src/content-rect.ts +211 -0
- package/src/cover.ts +21 -5
- package/src/cutlist.ts +38 -6
- package/src/dictionary.ts +56 -0
- package/src/export-premiere-project.ts +26 -9
- package/src/fonts.ts +17 -0
- package/src/index.ts +3 -0
- package/src/ingest.ts +88 -2
- package/src/normalize.ts +273 -127
- package/src/overrides.ts +728 -0
- package/src/phonetics.ts +101 -1
- package/src/producer/index.ts +1 -0
- package/src/producer/repair.ts +55 -23
- package/src/producer/youtube.ts +434 -0
- package/src/recut.ts +61 -0
- package/src/retake.ts +104 -2
- package/src/scene-schema.ts +10 -0
- package/src/thumbnail.ts +412 -0
- package/src/transcribe.ts +93 -5
- package/src/zoom.ts +63 -12
package/src/fonts.ts
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import { fileURLToPath } from "node:url";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Node-side half of the bundled caption font (see `captions.ts` for the
|
|
5
|
+
* browser-safe constants and the reason the font ships at all): the absolute
|
|
6
|
+
* path produce copies into the render's public dir. Split from captions.ts
|
|
7
|
+
* because `node:url` must never enter the Remotion bundle — captions.ts is
|
|
8
|
+
* on the `@ossclip/core/browser` surface.
|
|
9
|
+
*
|
|
10
|
+
* The URL below is face.ts's pico-cascade load shape, and the packaging test
|
|
11
|
+
* (R22 §111) scans this source for exactly that shape to assert the tarball
|
|
12
|
+
* carries the file — which is also why this comment doesn't spell the
|
|
13
|
+
* pattern out literally: the scanner reads comments too.
|
|
14
|
+
*/
|
|
15
|
+
export function nastaliqFontFile(): string {
|
|
16
|
+
return fileURLToPath(new URL("../assets/fonts/NotoNastaliqUrdu-Bold.ttf", import.meta.url));
|
|
17
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -16,6 +16,8 @@ export * from "./clip";
|
|
|
16
16
|
export * from "./blooper";
|
|
17
17
|
export * from "./retake";
|
|
18
18
|
export * from "./captions";
|
|
19
|
+
export * from "./fonts";
|
|
20
|
+
export * from "./dictionary";
|
|
19
21
|
export * from "./zoom";
|
|
20
22
|
export * from "./grounding";
|
|
21
23
|
export * from "./cta";
|
|
@@ -25,6 +27,7 @@ export * from "./normalize";
|
|
|
25
27
|
export * from "./framing";
|
|
26
28
|
export * from "./face";
|
|
27
29
|
export * from "./cover";
|
|
30
|
+
export * from "./thumbnail";
|
|
28
31
|
export * from "./source-text";
|
|
29
32
|
export * from "./report";
|
|
30
33
|
export * from "./export-markers";
|
package/src/ingest.ts
CHANGED
|
@@ -85,22 +85,108 @@ export async function extractAudio(tools: IngestTools, src: string, outWav: stri
|
|
|
85
85
|
]);
|
|
86
86
|
}
|
|
87
87
|
|
|
88
|
+
/**
|
|
89
|
+
* Headroom over the exact displayed size so a zoomed span never renders from
|
|
90
|
+
* below-native pixels (2026-08-17 render-speed pass). The two motion drivers
|
|
91
|
+
* stack to at most ZOOM_MAX_SCALE (1.05) × FACE_PUNCH_SCALE (1.015) ≈ 1.066
|
|
92
|
+
* on any frame a new run emits, so 1.1 covers the worst momentary
|
|
93
|
+
* magnification with margin. (The legacy punch-less contract renders 1.07 —
|
|
94
|
+
* still under 1.1 — and pre-existing render-props keep their own full-res
|
|
95
|
+
* mezzanine anyway; see `mezzanineFileName`.)
|
|
96
|
+
*/
|
|
97
|
+
export const MEZZANINE_SCALE_MARGIN = 1.1;
|
|
98
|
+
|
|
99
|
+
export interface MezzanineScale {
|
|
100
|
+
width: number;
|
|
101
|
+
height: number;
|
|
102
|
+
fps: number;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** Nearest even dimension — yuv420 chroma subsampling needs both axes even. */
|
|
106
|
+
function evenDim(v: number): number {
|
|
107
|
+
return Math.max(2, 2 * Math.round(v / 2));
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* The size and rate the mezzanine should be encoded at, or null when the
|
|
112
|
+
* source is already no larger than the render needs (2026-08-17 render-speed
|
|
113
|
+
* pass). Remotion's OffthreadVideo extracts EVERY sampled frame via ffmpeg
|
|
114
|
+
* on the CPU, so decode cost scales with pixels × fps — a 3456x2234@60
|
|
115
|
+
* source feeding a 1920x1080@30 render pays ~4.6× the pixels and 2× the
|
|
116
|
+
* frames the render ever shows.
|
|
117
|
+
*
|
|
118
|
+
* The target is the size at which the source is DISPLAYED: for `cover` the
|
|
119
|
+
* larger frame/source axis ratio (overflow is cropped, not shown), for
|
|
120
|
+
* `contain` the smaller (the whole frame fits inside). That target gets
|
|
121
|
+
* MEZZANINE_SCALE_MARGIN of headroom for the motion drivers, is rounded
|
|
122
|
+
* even for yuv420, and is capped at native — scaling UP would soften every
|
|
123
|
+
* frame for zero decode saved.
|
|
124
|
+
*
|
|
125
|
+
* fps: min(source, output) — frames the render never samples are pure decode
|
|
126
|
+
* waste. Safe because EDL `srcIn`/`srcOut` are SECONDS, not frame indexes:
|
|
127
|
+
* a 60→30 resample moves a cut boundary by at most 1/60s, the same
|
|
128
|
+
* magnitude whisper's word stamps already jitter by.
|
|
129
|
+
*/
|
|
130
|
+
export function mezzanineScale(
|
|
131
|
+
source: { width: number; height: number; fps: number },
|
|
132
|
+
frame: { width: number; height: number; fps: number },
|
|
133
|
+
sourceFit: "cover" | "contain",
|
|
134
|
+
): MezzanineScale | null {
|
|
135
|
+
if (source.width <= 0 || source.height <= 0) return null;
|
|
136
|
+
const displayed =
|
|
137
|
+
sourceFit === "contain"
|
|
138
|
+
? Math.min(frame.width / source.width, frame.height / source.height)
|
|
139
|
+
: Math.max(frame.width / source.width, frame.height / source.height);
|
|
140
|
+
const k = Math.min(1, displayed * MEZZANINE_SCALE_MARGIN);
|
|
141
|
+
// At the cap, keep the source's exact dims — even-rounding a size that is
|
|
142
|
+
// not being resampled would manufacture a 1px no-op rescale.
|
|
143
|
+
const width = k < 1 ? evenDim(source.width * k) : source.width;
|
|
144
|
+
const height = k < 1 ? evenDim(source.height * k) : source.height;
|
|
145
|
+
const fps = Math.min(source.fps, frame.fps);
|
|
146
|
+
if (width === source.width && height === source.height && fps >= source.fps) return null;
|
|
147
|
+
return { width, height, fps };
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* The mezzanine's filename, which IS its cache key: mezzanines are
|
|
152
|
+
* existence-keyed in the workdir, so the scale decision must live in the
|
|
153
|
+
* name — a pre-pass full-res `mezzanine.mp4` must never satisfy a run that
|
|
154
|
+
* will emit mezzanine-sized framing windows (they would land on a file with
|
|
155
|
+
* ~1.6× their pixel space and crop the wrong picture). Unscaled runs keep
|
|
156
|
+
* the legacy names so existing workdir caches stay valid; a scaled run
|
|
157
|
+
* rebuilds once under its own name and old workdirs' render-props keep
|
|
158
|
+
* referencing (and rendering from) the file they were emitted against.
|
|
159
|
+
*/
|
|
160
|
+
export function mezzanineFileName(cropped: boolean, scale: MezzanineScale | null): string {
|
|
161
|
+
const base = cropped ? "mezzanine-content" : "mezzanine";
|
|
162
|
+
if (!scale) return `${base}.mp4`;
|
|
163
|
+
return `${base}-${scale.width}x${scale.height}@${Math.round(scale.fps)}.mp4`;
|
|
164
|
+
}
|
|
165
|
+
|
|
88
166
|
/**
|
|
89
167
|
* Re-encode with dense keyframes so EDL playback (<OffthreadVideo> with many
|
|
90
168
|
* small trims) seeks fast. Optional — most sources play fine untouched —
|
|
91
169
|
* EXCEPT when the source is letterboxed: then this pass also trims the baked
|
|
92
170
|
* bars (`crop`), so everything downstream sees the picture, not picture+bars
|
|
93
171
|
* (PLAN Task 7), and the pass stops being optional.
|
|
172
|
+
*
|
|
173
|
+
* `scale` (from `mezzanineScale`) downsizes to display size in the SAME
|
|
174
|
+
* pass, crop first — the scale dims are computed on the post-crop picture.
|
|
94
175
|
*/
|
|
95
176
|
export async function makeMezzanine(
|
|
96
177
|
tools: IngestTools,
|
|
97
178
|
src: string,
|
|
98
179
|
out: string,
|
|
99
|
-
opts: { cropVf?: string } = {},
|
|
180
|
+
opts: { cropVf?: string; scale?: MezzanineScale } = {},
|
|
100
181
|
): Promise<void> {
|
|
182
|
+
const vf = [
|
|
183
|
+
...(opts.cropVf ? [opts.cropVf] : []),
|
|
184
|
+
...(opts.scale ? [`scale=${opts.scale.width}:${opts.scale.height}`] : []),
|
|
185
|
+
].join(",");
|
|
101
186
|
await run(tools.ffmpegPath, [
|
|
102
187
|
"-y", "-i", src,
|
|
103
|
-
...(
|
|
188
|
+
...(vf ? ["-vf", vf] : []),
|
|
189
|
+
...(opts.scale ? ["-r", String(opts.scale.fps)] : []),
|
|
104
190
|
"-c:v", "libx264", "-preset", "veryfast", "-crf", "18", "-g", "30", "-pix_fmt", "yuv420p",
|
|
105
191
|
"-c:a", "aac", "-b:a", "192k",
|
|
106
192
|
out,
|
package/src/normalize.ts
CHANGED
|
@@ -1,6 +1,4 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import { run } from "./exec";
|
|
3
|
-
import type { ContentRectSegment } from "./content-rect";
|
|
1
|
+
import { MIN_FRAMING_CLASS_FRAC, type ContentRectSegment } from "./content-rect";
|
|
4
2
|
import type { WindowFace } from "./face";
|
|
5
3
|
|
|
6
4
|
/**
|
|
@@ -14,21 +12,29 @@ import type { WindowFace } from "./face";
|
|
|
14
12
|
* Smoothing the boundaries cannot fix that; the output would still alternate
|
|
15
13
|
* between two shots.
|
|
16
14
|
*
|
|
17
|
-
* The fix is editorial
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
*
|
|
15
|
+
* The fix is editorial: pick ONE field of view — the tightest the source ever
|
|
16
|
+
* shows, i.e. the strip, since the strip's pixels are all those stretches
|
|
17
|
+
* have — and crop every other segment down to a window of that same shape,
|
|
18
|
+
* placed on the measured face. One consistent apparent framing, by choice.
|
|
19
|
+
*
|
|
20
|
+
* The plan used to be BAKED into a re-encoded file. That ended with the
|
|
21
|
+
* 2026-08-16 incident: an over-eager plan destroyed 55% of a screen
|
|
22
|
+
* recording's picture and the only undo was deleting the baked mp4 — the
|
|
23
|
+
* editor could not even see the crop had happened. The plan is now emitted
|
|
24
|
+
* into render-props.json as `framingTimeline` and applied at render time as a
|
|
25
|
+
* transform (packages/scenes/src/content-crop.ts), fully visible to and
|
|
26
|
+
* counteractable from the editor. The plan's GEOMETRY is unchanged by that
|
|
27
|
+
* move: every window shares one aspect, so per-segment render-time cover
|
|
28
|
+
* shows the same apparent framing on both sides of every boundary — the
|
|
29
|
+
* invariant the bake existed to enforce (144bbfb).
|
|
24
30
|
*
|
|
25
31
|
* When even the strip cannot cover the output frame without excessive
|
|
26
|
-
* upscaling
|
|
27
|
-
*
|
|
28
|
-
* fake-zoomed (option (b)).
|
|
32
|
+
* upscaling — or the plan would discard too much picture — normalization
|
|
33
|
+
* refuses (`ok: false`) and the caller falls back to render-time FIT — the
|
|
34
|
+
* strip shown at its natural size rather than fake-zoomed (option (b)).
|
|
29
35
|
*/
|
|
30
36
|
|
|
31
|
-
/** One
|
|
37
|
+
/** One planned stretch: this window of the source, covering the slot. */
|
|
32
38
|
export interface NormalizeSegment {
|
|
33
39
|
startSec: number;
|
|
34
40
|
endSec: number;
|
|
@@ -53,6 +59,28 @@ export interface NormalizePlan {
|
|
|
53
59
|
* soft fake is worse than an honest fit.
|
|
54
60
|
*/
|
|
55
61
|
coverUpscale: number;
|
|
62
|
+
/**
|
|
63
|
+
* Duration-weighted mean of the picture area each window discards from its
|
|
64
|
+
* segment's rect (`1 - windowArea / rectArea`). The other half of the
|
|
65
|
+
* quality gate, and the number the refusal log reports: coverUpscale alone
|
|
66
|
+
* measured softness, never loss (2026-08-16 incident).
|
|
67
|
+
*/
|
|
68
|
+
areaDiscardWeighted: number;
|
|
69
|
+
/**
|
|
70
|
+
* Per timeline segment, what the window is anchored on. "face" means the
|
|
71
|
+
* segment passed `segmentIsFaceOnly` and its window is sized and placed on
|
|
72
|
+
* the measured face; "screen" means the window is the segment's own rect,
|
|
73
|
+
* centered and clipped to the shared aspect — the picture, not the person,
|
|
74
|
+
* is the subject there (2026-08-16 incident: the PiP was not the subject).
|
|
75
|
+
*/
|
|
76
|
+
subject: ("face" | "screen")[];
|
|
77
|
+
/**
|
|
78
|
+
* Per timeline segment, where the subject sits INSIDE its window, both axes
|
|
79
|
+
* in 0..1. For a face segment this is the measured face centre relative to
|
|
80
|
+
* the final window, so a render-time cover crop can keep the head where the
|
|
81
|
+
* plan put it; a screen segment's subject is the whole picture, so 0.5/0.5.
|
|
82
|
+
*/
|
|
83
|
+
bias: { x: number; y: number }[];
|
|
56
84
|
ok: boolean;
|
|
57
85
|
}
|
|
58
86
|
|
|
@@ -64,6 +92,52 @@ export interface NormalizePlan {
|
|
|
64
92
|
*/
|
|
65
93
|
export const MAX_NORMALIZE_UPSCALE = 2.6;
|
|
66
94
|
|
|
95
|
+
/**
|
|
96
|
+
* Ceiling on the duration-weighted mean fraction of picture area a plan may
|
|
97
|
+
* throw away. coverUpscale 0.77 passed while 37% of the frame area was being
|
|
98
|
+
* thrown away — the gate measured softness, never loss (2026-08-16 incident).
|
|
99
|
+
*/
|
|
100
|
+
export const MAX_MEAN_AREA_DISCARD = 0.5;
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Ceiling on the picture area a SCREEN-subject segment's window may discard
|
|
104
|
+
* from its own rect. A screen segment's window is its rect clipped to the
|
|
105
|
+
* shared aspect — losing more than this means the shared aspect is genuinely
|
|
106
|
+
* fighting that segment's shape, and cropping screen content slides text and
|
|
107
|
+
* UI out of frame. Face segments are exempt: a face crop discards area by
|
|
108
|
+
* design. Applies to MATERIAL segments only, mirroring the aspect vote — a
|
|
109
|
+
* sliver class gets no vote on the aspect, so it cannot veto the plan for
|
|
110
|
+
* being clipped to it either (the 1.1% dark segment of the 2026-08-16
|
|
111
|
+
* incident loses ~16% to the shared aspect; the duration-weighted mean gate
|
|
112
|
+
* is what bounds slivers).
|
|
113
|
+
*/
|
|
114
|
+
export const MAX_SCREEN_AREA_DISCARD = 0.1;
|
|
115
|
+
|
|
116
|
+
/**
|
|
117
|
+
* Smallest face (box height over segment-rect height) that makes a segment
|
|
118
|
+
* "just a face". Equals DEFAULT_FACE.sizeFrac (stage.ts): a face smaller than
|
|
119
|
+
* the assumed arm's-length selfie box is not the frame's subject. 2026-08-16
|
|
120
|
+
* incident: the camera PiP measured 0.119 and the crop chased it, discarding
|
|
121
|
+
* the screen content that WAS the subject; a real talking head measures 0.28+.
|
|
122
|
+
*/
|
|
123
|
+
export const FACE_ONLY_MIN_FRAC = 0.22;
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* Below this fraction of sampled frames with a detection, the measurement is
|
|
127
|
+
* not confident enough to reframe on — a face seen in under half the looks is
|
|
128
|
+
* as likely a false positive or an occasional glance at a webcam.
|
|
129
|
+
*/
|
|
130
|
+
export const FACE_MIN_DETECTION_RATIO = 0.5;
|
|
131
|
+
|
|
132
|
+
/**
|
|
133
|
+
* Extra breathing room, as a fraction of the window height, kept above the
|
|
134
|
+
* crown and below the chin when the window slides to contain the head. The
|
|
135
|
+
* user's rule (2026-08-16): the ENTIRE head including hair stays in frame,
|
|
136
|
+
* with ~1% of margin — touching the frame edge reads as a crop even when
|
|
137
|
+
* nothing is technically cut.
|
|
138
|
+
*/
|
|
139
|
+
export const HEAD_WINDOW_MARGIN = 0.01;
|
|
140
|
+
|
|
67
141
|
/**
|
|
68
142
|
* A head is about 1.55x the detector's face box tall — the box bounds eyes,
|
|
69
143
|
* nose and mouth, and `stage.ts` models the crown at 0.35x above it and the
|
|
@@ -96,6 +170,25 @@ const median = (xs: number[]): number => {
|
|
|
96
170
|
return s.length % 2 ? s[m]! : (s[m - 1]! + s[m]!) / 2;
|
|
97
171
|
};
|
|
98
172
|
|
|
173
|
+
/**
|
|
174
|
+
* Is this segment essentially JUST a face — the only case where a
|
|
175
|
+
* face-anchored reframe is allowed (user decision, 2026-08-16)?
|
|
176
|
+
*
|
|
177
|
+
* Classified on the representative `sizeFrac`, not `sizeFracMax`: a segment
|
|
178
|
+
* is face-only by what it looks like most of the time, not at its one biggest
|
|
179
|
+
* lean-in (sizing, once classified, still uses the max — see planNormalization).
|
|
180
|
+
* The detection ratio guards against reframing on a face the detector barely
|
|
181
|
+
* ever saw.
|
|
182
|
+
*/
|
|
183
|
+
export function segmentIsFaceOnly(face: WindowFace | null): boolean {
|
|
184
|
+
if (!face) return false;
|
|
185
|
+
if (face.sizeFrac < FACE_ONLY_MIN_FRAC) return false;
|
|
186
|
+
return (
|
|
187
|
+
face.framesSampled > 0 &&
|
|
188
|
+
face.framesDetected / face.framesSampled >= FACE_MIN_DETECTION_RATIO
|
|
189
|
+
);
|
|
190
|
+
}
|
|
191
|
+
|
|
99
192
|
/**
|
|
100
193
|
* Decide the canvas and each segment's crop window.
|
|
101
194
|
*
|
|
@@ -119,6 +212,12 @@ const median = (xs: number[]): number => {
|
|
|
119
212
|
* zoom out as far as its own rect — clamping there rather than inventing
|
|
120
213
|
* pixels. The median (not the max) is what keeps that clamping rare and the
|
|
121
214
|
* upscale inside the quality gate.
|
|
215
|
+
*
|
|
216
|
+
* All of that applies ONLY to segments that are essentially just a face
|
|
217
|
+
* (`segmentIsFaceOnly`). Anything else — a screen share, a face-and-screen
|
|
218
|
+
* mix, a PiP — keeps its whole rect, centered and clipped to the shared
|
|
219
|
+
* aspect: the 2026-08-16 incident chased a 0.119 PiP and cropped away the
|
|
220
|
+
* screen content that was the actual subject.
|
|
122
221
|
*/
|
|
123
222
|
export function planNormalization(
|
|
124
223
|
timeline: readonly ContentRectSegment[],
|
|
@@ -133,10 +232,22 @@ export function planNormalization(
|
|
|
133
232
|
segments: [],
|
|
134
233
|
faceFracOfCanvas: [],
|
|
135
234
|
coverUpscale: Infinity,
|
|
235
|
+
// Nothing was planned, so nothing was discarded; `ok: false` is the
|
|
236
|
+
// refusal signal, not this number.
|
|
237
|
+
areaDiscardWeighted: 0,
|
|
238
|
+
subject: [],
|
|
239
|
+
bias: [],
|
|
136
240
|
ok: false,
|
|
137
241
|
};
|
|
138
242
|
}
|
|
139
243
|
|
|
244
|
+
// Face-anchored sizing and placement ONLY where the frame is essentially
|
|
245
|
+
// just a face (user decision, 2026-08-16). Everything else is a "screen"
|
|
246
|
+
// subject: its window is its own rect, centered and clipped to the shared
|
|
247
|
+
// aspect — the incident's PiP (sizeFrac 0.119) must never drag the crop
|
|
248
|
+
// to the bottom-right corner of a screen recording again.
|
|
249
|
+
const faceOnly = timeline.map((_, i) => segmentIsFaceOnly(faces[i] ?? null));
|
|
250
|
+
|
|
140
251
|
/**
|
|
141
252
|
* The LARGEST face fraction in each segment, not the median. A window sized
|
|
142
253
|
* on the median is correct only at the median moment: the author's clip
|
|
@@ -144,39 +255,60 @@ export function planNormalization(
|
|
|
144
255
|
* the frame edge whenever they leaned in — which is precisely the frame they
|
|
145
256
|
* flagged. Sizing on the maximum makes the tightest moment the safe one and
|
|
146
257
|
* every other moment merely roomier.
|
|
258
|
+
*
|
|
259
|
+
* Only FACE-ONLY segments contribute a measurement: a PiP-sized face must
|
|
260
|
+
* not drag the target down for the real talking heads (2026-08-16 incident).
|
|
147
261
|
*/
|
|
148
|
-
const measured = timeline.map((_, i) =>
|
|
262
|
+
const measured = timeline.map((_, i) =>
|
|
263
|
+
faceOnly[i] ? (faces[i]!.sizeFracMax ?? faces[i]!.sizeFrac) : null,
|
|
264
|
+
);
|
|
149
265
|
const known = measured.filter((v): v is number => v !== null);
|
|
150
266
|
|
|
151
267
|
// ---- Window heights ------------------------------------------------------
|
|
152
|
-
// Without a single
|
|
153
|
-
// rect-shaped
|
|
268
|
+
// Without a single face-only segment there is no subject to hold constant,
|
|
269
|
+
// and the whole plan degrades to rect-shaped windows: the tightest field of
|
|
270
|
+
// view, uniformly, with nothing anchored on a face.
|
|
154
271
|
const target = known.length > 0 ? Math.min(median(known), MAX_FACE_FRACTION) : null;
|
|
155
|
-
const
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
const windowHeights =
|
|
161
|
-
target === null
|
|
162
|
-
?
|
|
163
|
-
:
|
|
164
|
-
|
|
165
|
-
// framed like it (same rect height), so it is sized in ITS OWN class
|
|
166
|
-
// rather than averaged across two different shots.
|
|
167
|
-
const sameClass = timeline.flatMap((o, j) =>
|
|
168
|
-
measured[j] !== null && Math.abs(o.rect.h - s.rect.h) <= 2 ? [measured[j]!] : [],
|
|
169
|
-
);
|
|
170
|
-
const frac = measured[i] ?? (sameClass.length > 0 ? median(sameClass) : median(known));
|
|
171
|
-
return even(clamp((frac * s.rect.h) / target, 16, s.rect.h));
|
|
172
|
-
});
|
|
272
|
+
const canvasRect = boxed.reduce((a, b) => (b.rect.h < a.rect.h ? b : a)).rect;
|
|
273
|
+
const canvasRectAspect = canvasRect.w / canvasRect.h;
|
|
274
|
+
/** The tallest window of the tightest rect's shape that fits in `s.rect`. */
|
|
275
|
+
const rectShapedHeight = (s: ContentRectSegment): number =>
|
|
276
|
+
even(Math.min(s.rect.w, s.rect.h * canvasRectAspect) / canvasRectAspect);
|
|
277
|
+
const windowHeights = timeline.map((s, i) =>
|
|
278
|
+
target === null || !faceOnly[i]
|
|
279
|
+
? rectShapedHeight(s)
|
|
280
|
+
: even(clamp((measured[i]! * s.rect.h) / target, 16, s.rect.h)),
|
|
281
|
+
);
|
|
173
282
|
|
|
174
283
|
// ---- Canvas --------------------------------------------------------------
|
|
175
|
-
// The widest aspect every window can actually hold. Wider than the
|
|
176
|
-
// own aspect leaves the stage some horizontal freedom for the face
|
|
177
|
-
// narrower simply means the output crops height, which cover already
|
|
284
|
+
// The widest aspect every MATERIAL window can actually hold. Wider than the
|
|
285
|
+
// output's own aspect leaves the stage some horizontal freedom for the face
|
|
286
|
+
// bias; narrower simply means the output crops height, which cover already
|
|
287
|
+
// does.
|
|
288
|
+
//
|
|
289
|
+
// Material = the segment's framing class (same rect within 2px) totals at
|
|
290
|
+
// least
|
|
291
|
+
// MIN_FRAMING_CLASS_FRAC of the runtime — the constant content-rect's
|
|
292
|
+
// materiality filter uses, belt-and-braces with it. A sliver class still gets a
|
|
293
|
+
// window — the rect clamp below bounds it — it just gets no vote here: one
|
|
294
|
+
// 15.4s segment (1.1% of a 1435s take, rect 2848x2234) set canvas aspect
|
|
295
|
+
// 1.2748 for the whole video and baked away 28% of source width
|
|
296
|
+
// (2026-08-16 incident).
|
|
297
|
+
const totalDur = timeline.reduce((a, s) => a + Math.max(0, s.endSec - s.startSec), 0);
|
|
298
|
+
const classDur = timeline.map((s) =>
|
|
299
|
+
timeline.reduce(
|
|
300
|
+
(a, o) =>
|
|
301
|
+
Math.abs(o.rect.w - s.rect.w) <= 2 && Math.abs(o.rect.h - s.rect.h) <= 2
|
|
302
|
+
? a + Math.max(0, o.endSec - o.startSec)
|
|
303
|
+
: a,
|
|
304
|
+
0,
|
|
305
|
+
),
|
|
306
|
+
);
|
|
307
|
+
const material = classDur.map((d) => totalDur > 0 && d / totalDur >= MIN_FRAMING_CLASS_FRAC);
|
|
308
|
+
// If every class is a sliver there is no majority to defer to — all vote.
|
|
309
|
+
const votes = material.some(Boolean) ? material : material.map(() => true);
|
|
178
310
|
const aspect = timeline.reduce(
|
|
179
|
-
(a, s, i) => Math.min(a, s.rect.w / windowHeights[i]!),
|
|
311
|
+
(a, s, i) => (votes[i] ? Math.min(a, s.rect.w / windowHeights[i]!) : a),
|
|
180
312
|
Number.POSITIVE_INFINITY,
|
|
181
313
|
);
|
|
182
314
|
// The smallest window, so baking never upscales — the tightest segment sets
|
|
@@ -185,14 +317,16 @@ export function planNormalization(
|
|
|
185
317
|
const canvas = { width: even(canvasHeight * aspect), height: canvasHeight };
|
|
186
318
|
|
|
187
319
|
// ---- Face placement inside the window ------------------------------------
|
|
188
|
-
// Taken from the segments whose window IS their rect: their
|
|
189
|
-
// author's own and survives untouched, so it is the one to
|
|
320
|
+
// Taken from the FACE-ONLY segments whose window IS their rect: their
|
|
321
|
+
// framing is the author's own and survives untouched, so it is the one to
|
|
322
|
+
// reproduce. Screen segments get no say — the incident's PiP at 0.88/0.76
|
|
323
|
+
// would anchor every window bottom-right.
|
|
190
324
|
let wx = 0;
|
|
191
325
|
let wy = 0;
|
|
192
326
|
let weight = 0;
|
|
193
327
|
timeline.forEach((seg, i) => {
|
|
194
328
|
const f = faces[i];
|
|
195
|
-
if (!f || windowHeights[i]! < seg.rect.h - 2) return;
|
|
329
|
+
if (!f || !faceOnly[i] || windowHeights[i]! < seg.rect.h - 2) return;
|
|
196
330
|
const dur = Math.max(1e-6, seg.endSec - seg.startSec);
|
|
197
331
|
wx += f.centerXFrac * dur;
|
|
198
332
|
wy += f.centerYFrac * dur;
|
|
@@ -205,31 +339,54 @@ export function planNormalization(
|
|
|
205
339
|
const r = seg.rect;
|
|
206
340
|
const wH = windowHeights[i]!;
|
|
207
341
|
const wW = even(Math.min(r.w, wH * aspect));
|
|
208
|
-
|
|
209
|
-
//
|
|
210
|
-
|
|
211
|
-
|
|
342
|
+
|
|
343
|
+
// Screen subject: no face math at all. The window is the segment's own
|
|
344
|
+
// rect clipped to the shared aspect, centered on BOTH axes — the picture
|
|
345
|
+
// is the subject, and a centered clip is the only placement that does not
|
|
346
|
+
// pick a corner of it to sacrifice (2026-08-16 incident).
|
|
347
|
+
if (!faceOnly[i]) {
|
|
348
|
+
return {
|
|
349
|
+
startSec: seg.startSec,
|
|
350
|
+
endSec: seg.endSec,
|
|
351
|
+
window: {
|
|
352
|
+
x: even(r.x + (r.w - wW) / 2),
|
|
353
|
+
y: even(r.y + (r.h - wH) / 2),
|
|
354
|
+
w: wW,
|
|
355
|
+
h: wH,
|
|
356
|
+
},
|
|
357
|
+
};
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
const f = faces[i]!;
|
|
361
|
+
// Face position in source px.
|
|
362
|
+
const faceX = r.x + f.centerXFrac * r.w;
|
|
363
|
+
const faceY = r.y + f.centerYFrac * r.h;
|
|
212
364
|
const x = even(clamp(faceX - targetX * wW, r.x, r.x + r.w - wW));
|
|
213
365
|
|
|
214
366
|
let y = clamp(faceY - targetY * wH, r.y, r.y + r.h - wH);
|
|
215
367
|
// Then slide — never resize — so the whole HEAD is inside the window at
|
|
216
368
|
// the segment's largest face, since the aesthetic anchor above is about
|
|
217
|
-
// where the face sits, and this is about not amputating it.
|
|
218
|
-
// the
|
|
219
|
-
//
|
|
220
|
-
if
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
y = clamp(y,
|
|
369
|
+
// where the face sits, and this is about not amputating it. The margin is
|
|
370
|
+
// the user's ~1% rule (2026-08-16): the entire head including hair stays
|
|
371
|
+
// in frame with a hair of breathing room — a crown touching the edge
|
|
372
|
+
// reads as cropped. Bounded by the rect: if the head genuinely runs past
|
|
373
|
+
// the source's own edge there is nothing to slide toward, and the clamp
|
|
374
|
+
// leaves it where it was.
|
|
375
|
+
const maxFace = (f.sizeFracMax ?? f.sizeFrac) * r.h;
|
|
376
|
+
const headTop = faceY - HEAD_ABOVE * maxFace;
|
|
377
|
+
const headBottom = faceY + HEAD_BELOW * maxFace;
|
|
378
|
+
const margin = HEAD_WINDOW_MARGIN * wH;
|
|
379
|
+
if (headBottom - headTop + 2 * margin <= wH) {
|
|
380
|
+
y = clamp(y, headBottom + margin - wH, headTop - margin);
|
|
381
|
+
} else if (headBottom - headTop <= wH) {
|
|
382
|
+
// Head fits but its margin does not: keep the head, split the shortfall.
|
|
383
|
+
y = clamp(y, headBottom - wH, headTop);
|
|
384
|
+
} else {
|
|
385
|
+
// Head taller than the window: centre it, so what is lost is shared
|
|
386
|
+
// between crown and chin instead of taking the whole bite off one end.
|
|
387
|
+
y = (headTop + headBottom) / 2 - wH / 2;
|
|
232
388
|
}
|
|
389
|
+
y = clamp(y, r.y, r.y + r.h - wH);
|
|
233
390
|
return {
|
|
234
391
|
startSec: seg.startSec,
|
|
235
392
|
endSec: seg.endSec,
|
|
@@ -237,6 +394,23 @@ export function planNormalization(
|
|
|
237
394
|
};
|
|
238
395
|
});
|
|
239
396
|
|
|
397
|
+
const subject = timeline.map((_, i): "face" | "screen" => (faceOnly[i] ? "face" : "screen"));
|
|
398
|
+
// Where the subject sits inside its FINAL window (post-clamp, post-even):
|
|
399
|
+
// the render-time cover crop uses this to keep the head where the plan put
|
|
400
|
+
// it. Clamped to 0..1 because a rect-bounded window can leave the face
|
|
401
|
+
// centre outside it in the degenerate edge cases the clamps above allow.
|
|
402
|
+
const bias = timeline.map((seg, i) => {
|
|
403
|
+
if (!faceOnly[i]) return { x: 0.5, y: 0.5 };
|
|
404
|
+
const f = faces[i]!;
|
|
405
|
+
const w = segments[i]!.window;
|
|
406
|
+
const faceX = seg.rect.x + f.centerXFrac * seg.rect.w;
|
|
407
|
+
const faceY = seg.rect.y + f.centerYFrac * seg.rect.h;
|
|
408
|
+
return {
|
|
409
|
+
x: clamp(w.w > 0 ? (faceX - w.x) / w.w : 0.5, 0, 1),
|
|
410
|
+
y: clamp(w.h > 0 ? (faceY - w.y) / w.h : 0.5, 0, 1),
|
|
411
|
+
};
|
|
412
|
+
});
|
|
413
|
+
|
|
240
414
|
// Cover the output with the canvas: for a canvas wider than the output's
|
|
241
415
|
// aspect the height binds, otherwise the width does.
|
|
242
416
|
const coverUpscale =
|
|
@@ -246,19 +420,54 @@ export function planNormalization(
|
|
|
246
420
|
|
|
247
421
|
// What each segment actually achieved once its window is scaled to the
|
|
248
422
|
// canvas — measured from the plan, never assumed to equal the target: a
|
|
249
|
-
// segment clamped at its own rect lands wherever its rect put it.
|
|
423
|
+
// segment clamped at its own rect lands wherever its rect put it. A screen
|
|
424
|
+
// segment reports 0: it has no face SUBJECT, and downstream framing advice
|
|
425
|
+
// (assessCueFraming, framing.ts) rightly skips zeros rather than warning
|
|
426
|
+
// about the head of a PiP nobody is framing on.
|
|
250
427
|
const faceFracOfCanvas = timeline.map((seg, i) => {
|
|
251
428
|
const frac = measured[i];
|
|
252
|
-
if (frac === null || frac === undefined) return
|
|
429
|
+
if (frac === null || frac === undefined) return 0;
|
|
253
430
|
return (frac * seg.rect.h) / segments[i]!.window.h;
|
|
254
431
|
});
|
|
255
432
|
|
|
433
|
+
/** Picture area the window throws away from its segment's rect. */
|
|
434
|
+
const discardFrac = (i: number): number => {
|
|
435
|
+
const r = timeline[i]!.rect;
|
|
436
|
+
const w = segments[i]!.window;
|
|
437
|
+
return r.w * r.h > 0 ? 1 - (w.w * w.h) / (r.w * r.h) : 0;
|
|
438
|
+
};
|
|
439
|
+
|
|
440
|
+
// How much of each segment's picture the windows throw away, weighted by
|
|
441
|
+
// how long the viewer looks at it. coverUpscale 0.77 passed while 37% of
|
|
442
|
+
// the frame area was being thrown away — the gate measured softness, never
|
|
443
|
+
// loss (2026-08-16 incident).
|
|
444
|
+
const areaDiscardWeighted = timeline.reduce((acc, seg, i) => {
|
|
445
|
+
const dur = Math.max(0, seg.endSec - seg.startSec);
|
|
446
|
+
return acc + (totalDur > 0 ? dur / totalDur : 0) * discardFrac(i);
|
|
447
|
+
}, 0);
|
|
448
|
+
|
|
449
|
+
// Per-segment bound for SCREEN subjects: their window is meant to be
|
|
450
|
+
// (essentially) their whole rect, so a material screen segment losing more
|
|
451
|
+
// than MAX_SCREEN_AREA_DISCARD means the plan is cropping content nobody
|
|
452
|
+
// asked it to reframe — refuse and let render-time fit show it honestly.
|
|
453
|
+
// Face segments are exempt (a face crop discards area by design); sliver
|
|
454
|
+
// segments are exempt for the same reason they get no aspect vote.
|
|
455
|
+
const screenLossOk = timeline.every(
|
|
456
|
+
(_, i) => faceOnly[i] || !material[i] || discardFrac(i) <= MAX_SCREEN_AREA_DISCARD,
|
|
457
|
+
);
|
|
458
|
+
|
|
256
459
|
return {
|
|
257
460
|
canvas,
|
|
258
461
|
segments,
|
|
259
462
|
faceFracOfCanvas,
|
|
260
463
|
coverUpscale,
|
|
261
|
-
|
|
464
|
+
areaDiscardWeighted,
|
|
465
|
+
subject,
|
|
466
|
+
bias,
|
|
467
|
+
ok:
|
|
468
|
+
coverUpscale <= MAX_NORMALIZE_UPSCALE &&
|
|
469
|
+
areaDiscardWeighted <= MAX_MEAN_AREA_DISCARD &&
|
|
470
|
+
screenLossOk,
|
|
262
471
|
};
|
|
263
472
|
}
|
|
264
473
|
|
|
@@ -355,66 +564,3 @@ export function assessCueFraming(
|
|
|
355
564
|
}
|
|
356
565
|
return out;
|
|
357
566
|
}
|
|
358
|
-
|
|
359
|
-
/**
|
|
360
|
-
* The ffmpeg filter graph baking the plan: each segment trimmed, cropped to
|
|
361
|
-
* its window, scaled to the canvas, and the pieces concatenated back into one
|
|
362
|
-
* continuous stream. The segment boundaries partition the source exactly, so
|
|
363
|
-
* the output timeline equals the input's and the untouched audio stays in
|
|
364
|
-
* sync.
|
|
365
|
-
*/
|
|
366
|
-
export function normalizationFilterGraph(plan: NormalizePlan): string {
|
|
367
|
-
const parts = plan.segments.map((s, i) => {
|
|
368
|
-
const w = s.window;
|
|
369
|
-
return (
|
|
370
|
-
`[0:v]trim=start=${s.startSec.toFixed(3)}:end=${s.endSec.toFixed(3)},` +
|
|
371
|
-
`setpts=PTS-STARTPTS,crop=${w.w}:${w.h}:${w.x}:${w.y},` +
|
|
372
|
-
// setsar=1 is load-bearing, not tidiness (R27 §125). Every segment is
|
|
373
|
-
// scaled to the SAME canvas, but from a DIFFERENT crop, and ffmpeg
|
|
374
|
-
// derives a sample aspect from that ratio: a 946x1682 crop yields SAR
|
|
375
|
-
// 1683:1682 and a 932x1660 crop 1377:1376. `concat` requires identical
|
|
376
|
-
// SAR across inputs and aborts the whole bake when they disagree, so a
|
|
377
|
-
// take whose framing varies — exactly the take normalization exists
|
|
378
|
-
// for — failed to render at all.
|
|
379
|
-
`scale=${plan.canvas.width}:${plan.canvas.height},setsar=1[v${i}]`
|
|
380
|
-
);
|
|
381
|
-
});
|
|
382
|
-
const inputs = plan.segments.map((_, i) => `[v${i}]`).join("");
|
|
383
|
-
return `${parts.join(";")};${inputs}concat=n=${plan.segments.length}:v=1:a=0[v]`;
|
|
384
|
-
}
|
|
385
|
-
|
|
386
|
-
/**
|
|
387
|
-
* Bake the normalized source. Encoded with the mezzanine's own settings
|
|
388
|
-
* (dense keyframes) because it REPLACES the mezzanine — normalizing and then
|
|
389
|
-
* re-encoding for seekability would be two generations of loss for nothing.
|
|
390
|
-
*/
|
|
391
|
-
export async function bakeNormalizedSource(
|
|
392
|
-
tools: { ffmpegPath: string },
|
|
393
|
-
input: string,
|
|
394
|
-
plan: NormalizePlan,
|
|
395
|
-
outPath: string,
|
|
396
|
-
): Promise<void> {
|
|
397
|
-
// Encode to a sibling temp path and rename only on success (R27 §125).
|
|
398
|
-
// ffmpeg writes the container header as it goes, so a bake that dies
|
|
399
|
-
// mid-graph leaves a file with no `moov` atom — and the cache upstream keys
|
|
400
|
-
// on EXISTENCE, so that corpse is then reused as a valid normalized source
|
|
401
|
-
// on every later run. The failure surfaces as "moov atom not found" from a
|
|
402
|
-
// step that never ran, and deleting the workdir is the only way out. Rename
|
|
403
|
-
// is atomic on a POSIX filesystem, so the cache can only ever see a file
|
|
404
|
-
// ffmpeg finished writing.
|
|
405
|
-
const partial = `${outPath}.partial.mp4`;
|
|
406
|
-
try {
|
|
407
|
-
await run(tools.ffmpegPath, [
|
|
408
|
-
"-y", "-i", input,
|
|
409
|
-
"-filter_complex", normalizationFilterGraph(plan),
|
|
410
|
-
"-map", "[v]", "-map", "0:a?",
|
|
411
|
-
"-c:v", "libx264", "-preset", "veryfast", "-crf", "18", "-g", "30", "-pix_fmt", "yuv420p",
|
|
412
|
-
"-c:a", "aac", "-b:a", "192k",
|
|
413
|
-
partial,
|
|
414
|
-
]);
|
|
415
|
-
await rename(partial, outPath);
|
|
416
|
-
} catch (err) {
|
|
417
|
-
await rm(partial, { force: true });
|
|
418
|
-
throw err;
|
|
419
|
-
}
|
|
420
|
-
}
|