@ossclip/core 0.1.24 → 0.1.25

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -153,6 +153,75 @@ export interface ContentRectSegment {
153
153
  rect: ContentRect;
154
154
  }
155
155
 
156
+ /**
157
+ * One stretch of the RENDER-TIME framing plan — the props-based successor to
158
+ * the destructive normalization bake (2026-08-16 incident: the bake's crop
159
+ * could only be undone by deleting the re-encoded file; expressed as data,
160
+ * the same window renders as a transform the editor can see and counteract).
161
+ * Emitted into render-props.json as `framingTimeline`; absent means "no plan"
162
+ * and every pre-existing render-props renders unchanged.
163
+ */
164
+ export interface FramingSegment {
165
+ /** Source seconds — intersects the content timeline and kept spans directly. */
166
+ startSec: number;
167
+ endSec: number;
168
+ /** Crop window in SOURCE pixels, all windows sharing one aspect. */
169
+ window: { x: number; y: number; w: number; h: number };
170
+ /** What the window is anchored on; "screen" windows are centered clips. */
171
+ subject: "face" | "screen";
172
+ /** The subject's anchor point inside the window, 0..1 both axes. */
173
+ bias: { x: number; y: number };
174
+ }
175
+
176
+ /**
177
+ * Rescale framing windows from TRUE source pixels into the pixel space of a
178
+ * display-sized mezzanine (2026-08-17 render-speed pass). `planNormalization`
179
+ * keeps working in true source pixels — analysis runs on the source — and
180
+ * the render-props emission scales the windows so window space === the space
181
+ * of the file the render actually plays. Per-axis factors, not one scalar:
182
+ * yuv420 even-rounding makes the two axes' ratios differ by a fraction of a
183
+ * percent, and scaling both by one axis's factor could push a right-edge
184
+ * window past the scaled file's width. Times, subject and bias are
185
+ * scale-invariant and pass through untouched.
186
+ */
187
+ export function scaleFramingWindows(
188
+ timeline: readonly FramingSegment[],
189
+ factor: { x: number; y: number },
190
+ ): FramingSegment[] {
191
+ return timeline.map((s) => ({
192
+ ...s,
193
+ window: {
194
+ x: s.window.x * factor.x,
195
+ y: s.window.y * factor.y,
196
+ w: s.window.w * factor.x,
197
+ h: s.window.h * factor.y,
198
+ },
199
+ }));
200
+ }
201
+
202
+ /**
203
+ * The same rescale for the fit-fallback's content timeline — its rects are
204
+ * source pixels too, and they reach the renderer only alongside a
205
+ * `sourceSize` that must describe the played file (see the render-props
206
+ * emission in produce.ts). `full` survives the scale: it means "this rect IS
207
+ * the whole frame", which a uniform resample does not change.
208
+ */
209
+ export function scaleContentTimeline(
210
+ timeline: readonly ContentRectSegment[],
211
+ factor: { x: number; y: number },
212
+ ): ContentRectSegment[] {
213
+ return timeline.map((s) => ({
214
+ ...s,
215
+ rect: {
216
+ ...s.rect,
217
+ x: s.rect.x * factor.x,
218
+ y: s.rect.y * factor.y,
219
+ w: s.rect.w * factor.x,
220
+ h: s.rect.h * factor.y,
221
+ },
222
+ }));
223
+ }
224
+
156
225
  /**
157
226
  * A framing run must survive BOTH of these to be believed. Together they are
158
227
  * what replaces the union rule's protection (PLAN Task C, step C2).
@@ -270,6 +339,148 @@ export function contentRectTimeline(
270
339
  }));
271
340
  }
272
341
 
342
+ /**
343
+ * Materiality thresholds (2026-08-16 landscape screen-recording over-crop — a
344
+ * 72px dark strip (2.08% inset, 1px past SNAP_FRAC's 69.1px tolerance) plus a
345
+ * 15.4s/1435s outlier segment destroyed 55% of the picture).
346
+ *
347
+ * SNAP_FRAC decides whether a bar is worth TRIMMING; this bound decides
348
+ * whether a rect difference is worth calling a FRAMING CHANGE — a far more
349
+ * destructive claim, because a mixed-framing verdict sends the whole take
350
+ * into normalization. The incident's jitter measured 2.0–2.3% per side, so
351
+ * the bound sits at 3.5%: comfortably above jitter, comfortably below any
352
+ * real letterbox (the motivating 144bbfb strip insets 34% per side).
353
+ */
354
+ export const MATERIAL_INSET_FRAC = 0.035;
355
+
356
+ /**
357
+ * A rect keeping at least this share of the frame's area, at (within
358
+ * MATERIAL_ASPECT_TOL of) the frame's own aspect, is the frame minus
359
+ * measurement noise — treating it as a distinct framing would re-crop every
360
+ * downstream geometry to chase pixels nobody can see missing.
361
+ */
362
+ export const MATERIAL_AREA_FRAC = 0.92;
363
+ const MATERIAL_ASPECT_TOL = 0.05;
364
+
365
+ /**
366
+ * A framing class — every segment sharing one rect, contiguous or not —
367
+ * totalling under this share of the runtime is an anomaly, not a framing
368
+ * change. MIN_RUN_SEC only guards a single run; the incident's 15.4s outlier
369
+ * (1.1% of 1435s) sailed past it and single-handedly set the canvas aspect
370
+ * for the whole take.
371
+ */
372
+ export const MIN_FRAMING_CLASS_FRAC = 0.05;
373
+
374
+ /**
375
+ * Re-judge a measured timeline against what a framing change must AMOUNT TO
376
+ * before it is believed (2026-08-16 incident above).
377
+ *
378
+ * Two passes. First, per segment: a rect whose every side inset is under
379
+ * MATERIAL_INSET_FRAC, or that keeps MATERIAL_AREA_FRAC of the frame at the
380
+ * frame's own aspect, is reclassified as the full frame — it differs from the
381
+ * frame by less than a framing change is worth. Second, per CLASS: framing
382
+ * classes totalling under MIN_FRAMING_CLASS_FRAC of the runtime are absorbed
383
+ * into their longer neighbour, same shape as the run-absorption loop in
384
+ * `contentRectTimeline` (drop one, retry until stable, then merge agreeing
385
+ * neighbours). A uniform source comes out as exactly one segment.
386
+ *
387
+ * Pure — applied on top of the cached raw timeline, so an already-measured
388
+ * source re-classifies on replay without a cache version bump.
389
+ */
390
+ export function materializeTimeline(
391
+ timeline: readonly ContentRectSegment[],
392
+ width: number,
393
+ height: number,
394
+ durationSec: number,
395
+ ): ContentRectSegment[] {
396
+ if (timeline.length === 0 || width <= 0 || height <= 0) {
397
+ return timeline.map((s) => ({ ...s }));
398
+ }
399
+ const whole: ContentRect = { x: 0, y: 0, w: width, h: height, full: true };
400
+ const frameAspect = width / height;
401
+
402
+ const immaterial = (r: ContentRect): boolean => {
403
+ if (r.full) return true;
404
+ const insets = [
405
+ r.x / width,
406
+ r.y / height,
407
+ (width - (r.x + r.w)) / width,
408
+ (height - (r.y + r.h)) / height,
409
+ ];
410
+ if (insets.every((f) => f < MATERIAL_INSET_FRAC)) return true;
411
+ const areaFrac = (r.w * r.h) / (width * height);
412
+ const aspectDelta = Math.abs(r.w / r.h - frameAspect) / frameAspect;
413
+ return areaFrac >= MATERIAL_AREA_FRAC && aspectDelta < MATERIAL_ASPECT_TOL;
414
+ };
415
+
416
+ const segs = timeline.map((s) => ({
417
+ startSec: s.startSec,
418
+ endSec: s.endSec,
419
+ rect: immaterial(s.rect) ? whole : s.rect,
420
+ }));
421
+
422
+ // Duration of each segment's whole CLASS — the outlier that motivated this
423
+ // appeared as one contiguous run, but a strip flickering in and out would
424
+ // split into several, and each alone dodging the threshold must not let the
425
+ // class as a whole survive.
426
+ const classTotals = (list: typeof segs): number[] => {
427
+ const reps: ContentRect[] = [];
428
+ const totals: number[] = [];
429
+ const ids = list.map((s) => {
430
+ let id = reps.findIndex((r) => sameFraming(r, s.rect, width, height));
431
+ if (id === -1) {
432
+ id = reps.length;
433
+ reps.push(s.rect);
434
+ totals.push(0);
435
+ }
436
+ totals[id]! += s.endSec - s.startSec;
437
+ return id;
438
+ });
439
+ return ids.map((id) => totals[id]!);
440
+ };
441
+
442
+ // Absorb, then retry until stable: dropping one segment changes its
443
+ // neighbours' class totals and can make them adjacent and mergeable.
444
+ let changed = true;
445
+ while (changed && segs.length > 1) {
446
+ changed = false;
447
+ const totals = classTotals(segs);
448
+ for (let i = 0; i < segs.length; i++) {
449
+ if (totals[i]! >= MIN_FRAMING_CLASS_FRAC * durationSec) continue;
450
+ const r = segs[i]!;
451
+ // Absorb into the LONGER neighbour — the one more likely to be the truth.
452
+ const prev = segs[i - 1];
453
+ const next = segs[i + 1];
454
+ const into = !prev
455
+ ? next
456
+ : !next
457
+ ? prev
458
+ : prev.endSec - prev.startSec >= next.endSec - next.startSec
459
+ ? prev
460
+ : next;
461
+ if (!into) continue;
462
+ into.startSec = Math.min(into.startSec, r.startSec);
463
+ into.endSec = Math.max(into.endSec, r.endSec);
464
+ segs.splice(i, 1);
465
+ changed = true;
466
+ break;
467
+ }
468
+ }
469
+
470
+ // Merge neighbours that now agree, so a source whose every segment was
471
+ // reclassified collapses to the single-segment (uniform) shape callers test.
472
+ const merged: typeof segs = [];
473
+ for (const s of segs) {
474
+ const last = merged[merged.length - 1];
475
+ if (last && sameFraming(last.rect, s.rect, width, height)) {
476
+ last.endSec = Math.max(last.endSec, s.endSec);
477
+ } else {
478
+ merged.push({ ...s });
479
+ }
480
+ }
481
+ return merged;
482
+ }
483
+
273
484
  /**
274
485
  * The exact framing-change instant inside a densely-sampled window
275
486
  * (NORMALIZE plan, boundary refinement).
package/src/cover.ts CHANGED
@@ -136,9 +136,17 @@ export function laplacianVariance(pixels: Uint8Array, w: number, h: number): num
136
136
  }
137
137
 
138
138
  /**
139
- * Score a candidate. A face is close to mandatory — a cover without the
140
- * speaker is a cover for a different video — and among frames that have one,
141
- * sharpness decides. Earlier frames win ties so the cover matches the opening.
139
+ * Score a candidate. On a "face" take a face is close to mandatory — a cover
140
+ * without the speaker is a cover for a different video — and among frames
141
+ * that have one, sharpness decides. Earlier frames win ties so the cover
142
+ * matches the opening.
143
+ *
144
+ * On a "screen" take the face weight drops to ZERO (2026-08-16): a Facebook
145
+ * reel playing inside a 21-minute screen recording put a STRANGER'S face on
146
+ * the cover, because face×2 hunts any face at all and on a screen-subject
147
+ * take every face in frame is content, not the speaker. Sharpness and
148
+ * earliness alone pick that cover. Default "face" so portrait/talking-head
149
+ * runs score exactly as before.
142
150
  */
143
151
  export function scoreCandidate(c: {
144
152
  timeSec: number;
@@ -146,11 +154,13 @@ export function scoreCandidate(c: {
146
154
  sharpness: number;
147
155
  hasFace: boolean;
148
156
  maxSharpness: number;
157
+ subject?: "face" | "screen";
149
158
  }): number {
159
+ const faceWeight = c.subject === "screen" ? 0 : 2;
150
160
  const face = c.hasFace ? 1 : 0;
151
161
  const sharp = c.maxSharpness > 0 ? c.sharpness / c.maxSharpness : 0;
152
162
  const earliness = 1 - Math.min(1, c.timeSec / Math.max(1e-6, c.durationSec));
153
- return face * 2 + sharp + earliness * 0.3;
163
+ return face * faceWeight + sharp + earliness * 0.3;
154
164
  }
155
165
 
156
166
  export interface PickCoverOptions {
@@ -172,6 +182,12 @@ export interface PickCoverOptions {
172
182
  * canvas that is two-thirds baked-in black bar.
173
183
  */
174
184
  cropVf?: string;
185
+ /**
186
+ * The take's whole-frame subject (`faceSubject`'s verdict). "screen" zeroes
187
+ * the face weight in `scoreCandidate` — see its doc comment for the
188
+ * 2026-08-16 stranger's-face incident. Default "face".
189
+ */
190
+ subject?: "face" | "screen";
175
191
  }
176
192
 
177
193
  /**
@@ -229,7 +245,7 @@ export async function pickCoverFrame(
229
245
  for (const r of raw) {
230
246
  candidates.push({
231
247
  ...r,
232
- score: scoreCandidate({ ...r, durationSec: window, maxSharpness }),
248
+ score: scoreCandidate({ ...r, durationSec: window, maxSharpness, subject: opts.subject }),
233
249
  });
234
250
  }
235
251
  candidates.sort((a, b) => b.score - a.score);
package/src/cutlist.ts CHANGED
@@ -42,6 +42,18 @@ const SILENCE_PAD_IN = 0.06;
42
42
  const SILENCE_PAD_OUT = 0.1;
43
43
  /** Kept fragments shorter than this (holding no word) are folded into the cut. */
44
44
  const MIN_KEEP = 0.25;
45
+ /**
46
+ * A WORDED kept gap up to this long is still folded when a flanking removal
47
+ * is a `retake` cut (2026-08-16 incident): whisper stamped a debris "And" at
48
+ * 670.0–670.4 between a marker-terminated blooper cut and the silence cut
49
+ * that followed, and `hasProtectedWordInside` — correctly — refused the
50
+ * wordless fold, so a 0.4s one-word sliver of an abandoned sentence shipped
51
+ * as a choppy blip. A retake cut already means "this whole attempt is
52
+ * unusable", so a sub-0.35s word fragment glued to it is debris of that same
53
+ * attempt, not content. Silence/silence neighbors get NO such license: their
54
+ * worded gaps stay protected exactly as §124's follow-up demands.
55
+ */
56
+ const MIN_KEEP_AFTER_RETAKE = 0.35;
45
57
 
46
58
  interface Removal {
47
59
  start: number;
@@ -74,8 +86,19 @@ export interface BuildCutlistArgs {
74
86
  * still has no judgement of its own about what a bad take looks like, it
75
87
  * just folds whichever spans two independent detectors handed it into the
76
88
  * one partition.
89
+ *
90
+ * `confidence` optionally overrides the 0.9 default below — the
91
+ * exact-prefix restart rule (retake.ts, RESTART_PREFIX_CONFIDENCE) earns
92
+ * 0.85, one notch less than a full similarity match, and the caller is the
93
+ * one who knows which rule found the span.
77
94
  */
78
- retakes?: readonly { startWord: number; endWord: number; startSec: number; endSec: number }[];
95
+ retakes?: readonly {
96
+ startWord: number;
97
+ endWord: number;
98
+ startSec: number;
99
+ endSec: number;
100
+ confidence?: number;
101
+ }[];
79
102
  }
80
103
 
81
104
  export function buildCutlist({
@@ -88,9 +111,9 @@ export function buildCutlist({
88
111
  }: BuildCutlistArgs): Segment[] {
89
112
  const keepAll: Segment[] = [{ srcIn: 0, srcOut: duration, kind: "keep" }];
90
113
  // `exact` means exact: it is the escape hatch for "touch nothing", and a
91
- // blooper or retake cut is still a cut. --blooper-marker or
92
- // --collapse-retakes with --cleanup exact is a contradiction, and the flag
93
- // the user typed second does not get to win.
114
+ // blooper or retake cut is still a cut. --blooper-marker (which also
115
+ // switches on retake collapse, 2026-08-16 gate) with --cleanup exact is a
116
+ // contradiction, and the flag the user typed second does not get to win.
94
117
  if (level === "exact") return keepAll;
95
118
  const policy = POLICIES[level];
96
119
  const words = transcript.words;
@@ -126,7 +149,7 @@ export function buildCutlist({
126
149
  start: r.startSec,
127
150
  end: r.endSec,
128
151
  reason: "retake",
129
- confidence: 0.9,
152
+ confidence: r.confidence ?? 0.9,
130
153
  source: "acoustic",
131
154
  });
132
155
  }
@@ -268,7 +291,16 @@ export function buildCutlist({
268
291
  const overlapping = prev !== undefined && gap <= 1e-6;
269
292
  const wordless = prev !== undefined && !hasProtectedWordInside(prev.end, r.start);
270
293
  const foldableGap = wordless && gap <= Math.max(MIN_KEEP, policy.pauseMin);
271
- if (prev && (overlapping || foldableGap)) {
294
+ // The one exception to word protection (MIN_KEEP_AFTER_RETAKE's why): a
295
+ // worded sliver this short survives only between two cuts, and when one
296
+ // of those cuts is a retake the sliver is debris of the discarded
297
+ // attempt. Requires the retake FLANK, not just the length — a worded gap
298
+ // between two silence removals must never fold, whatever its size.
299
+ const retakeSliver =
300
+ prev !== undefined &&
301
+ gap <= MIN_KEEP_AFTER_RETAKE &&
302
+ (prev.reason === "retake" || r.reason === "retake");
303
+ if (prev && (overlapping || foldableGap || retakeSliver)) {
272
304
  const prevDur = prev.end - prev.start;
273
305
  const curDur = r.end - r.start;
274
306
  prev.end = Math.max(prev.end, r.end);
@@ -0,0 +1,56 @@
1
+ import type { Transcript } from "./schema";
2
+
3
+ /**
4
+ * Deterministic casing for the user's dictionary (F4, 2026-08-16).
5
+ *
6
+ * Whisper biasing and the LLM repair pass get a word's SPELLING right; what
7
+ * neither guarantees is its CASE — a decoder nudged into "json" has still
8
+ * lost the acronym. This pass is the deterministic last word: any token that
9
+ * IS a dictionary term (exact case-insensitive match, punctuation aside)
10
+ * takes the term's canonical casing.
11
+ *
12
+ * Deliberately exact-match only: "Jason" is NEVER touched, even with "JSON"
13
+ * in the dictionary — deciding that a different word is a mishearing of a
14
+ * term is phonetic judgement, and that is the LLM repair pass's job, behind
15
+ * its guards. A deterministic pass that rewrote near-misses would be the
16
+ * un-gated rewrite pass §17 exists to forbid.
17
+ *
18
+ * Not on the browser surface: only `produce` runs it, and keeping it out of
19
+ * browser.ts keeps the Remotion bundle's import graph untouched.
20
+ */
21
+
22
+ /**
23
+ * The same leading/trailing bounds `normalizeToken` (analyze.ts) strips, but
24
+ * keeping the pieces: the punctuation must survive the swap ("json." →
25
+ * "JSON.", quotes and commas intact), so the strip is a split here, not a
26
+ * deletion.
27
+ */
28
+ const TOKEN_BOUNDS = /^([^\p{L}\p{N}]*)([\s\S]*?)([^\p{L}\p{N}-]*)$/u;
29
+
30
+ export function canonicalizeDictionaryCasing(
31
+ transcript: Transcript,
32
+ dictionary: readonly string[],
33
+ ): Transcript {
34
+ // Keyed on the term's own lowercase, valued with its typed casing. A later
35
+ // duplicate (differing only in case) loses to the first — the user's list
36
+ // order is the only precedence signal there is.
37
+ const canonical = new Map<string, string>();
38
+ for (const raw of dictionary) {
39
+ const term = raw.trim();
40
+ const key = term.toLowerCase();
41
+ if (term && !canonical.has(key)) canonical.set(key, term);
42
+ }
43
+ if (canonical.size === 0) return transcript;
44
+ return {
45
+ ...transcript,
46
+ words: transcript.words.map((w) => {
47
+ const [, lead = "", core = "", trail = ""] = TOKEN_BOUNDS.exec(w.text) ?? [];
48
+ const want = canonical.get(core.toLowerCase());
49
+ // `want !== core` also guards the degenerate empty core; strict
50
+ // equality-of-lowercase above means this is a CASING change only —
51
+ // never a respelling, never added punctuation.
52
+ if (want === undefined || want === core) return w;
53
+ return { ...w, text: `${lead}${want}${trail}` };
54
+ }),
55
+ };
56
+ }
@@ -39,22 +39,34 @@ export interface PremiereProjectInput {
39
39
  zoomPlan: readonly ZoomSegment[];
40
40
  /** render-props' flag: true disables BOTH motion layers (zoom AND punch). */
41
41
  staticCamera?: boolean;
42
+ /**
43
+ * render-props' `punch` — the face-only jump-cut plan (2026-08-16, scenes'
44
+ * punch-plan.ts): its scale replaces the legacy 1.07 and its mask gates
45
+ * which spans punch. Absent means the LEGACY contract, so a pre-feature
46
+ * workdir exports exactly the camera its render had.
47
+ */
48
+ punch?: { scale: number; allowed: readonly boolean[] } | null;
42
49
  /** The OUTPUT frame (production.render), e.g. 1080×1920. */
43
50
  frame: { width: number; height: number };
44
51
  }
45
52
 
46
53
  /**
47
- * The render's jump-cut concealer, replicated from EdlVideo.tsx:41-51 — the
48
- * alternating toggle, the srcIn−prev.srcOut gap, and the INCLUSIVE >=
49
- * threshold must all match, or the exported project punches different clips
50
- * than the render did. Defaults mirror EdlVideoProps; they are parameters so
51
- * a drift in either place shows up as a failing hand-computed test, not a
52
- * silent divergence.
54
+ * The render's jump-cut concealer, replicated from scenes' `punchScalesFor`
55
+ * (punch-plan.ts, the loop EdlVideo renders) — the alternating toggle, the
56
+ * srcIn−prev.srcOut gap, and the INCLUSIVE >= threshold must all match, or
57
+ * the exported project punches different clips than the render did. The
58
+ * `allowed` mask must match too, down to the parity rule: the toggle flips
59
+ * on EVERY qualifying gap, masked spans included (stable indexing), and a
60
+ * masked span renders its punched turn at 1. `allowed[i] !== false` so a
61
+ * short mask reads as allowed, same as no mask at all. Defaults mirror
62
+ * EdlVideoProps; they are parameters so a drift in either place shows up as
63
+ * a failing hand-computed test, not a silent divergence.
53
64
  */
54
65
  export function punchScales(
55
66
  spans: readonly KeptSpan[],
56
67
  punchInScale = 1.07,
57
68
  punchThresholdSec = 0.15,
69
+ allowed?: readonly boolean[] | null,
58
70
  ): number[] {
59
71
  const out: number[] = [];
60
72
  let punched = false;
@@ -62,7 +74,7 @@ export function punchScales(
62
74
  const prev = spans[i - 1];
63
75
  const gap = prev ? spans[i]!.srcIn - prev.srcOut : 0;
64
76
  if (i > 0 && gap >= punchThresholdSec) punched = !punched;
65
- out.push(punched ? punchInScale : 1);
77
+ out.push(punched && (!allowed || allowed[i] !== false) ? punchInScale : 1);
66
78
  }
67
79
  return out;
68
80
  }
@@ -174,7 +186,7 @@ export function srtFromCaptionLines(lines: readonly SrtLine[]): string {
174
186
  const fmt = (n: number): string => String(Number(n.toFixed(4)));
175
187
 
176
188
  export function buildPremiereProject(input: PremiereProjectInput): { xml: string; srt: string } {
177
- const { production, spans, captionLines, zoomPlan, staticCamera, frame } = input;
189
+ const { production, spans, captionLines, zoomPlan, staticCamera, punch, frame } = input;
178
190
  const { path, probe } = production.source;
179
191
  // The whole document runs at the SOURCE rate (the marker exporter's
180
192
  // precedent): clip in/out and sequence start/end share one timebase, so no
@@ -189,7 +201,12 @@ export function buildPremiereProject(input: PremiereProjectInput): { xml: string
189
201
  const cover = coverTransform(probe, production.source.face, frame);
190
202
  // staticCamera kills BOTH motion drivers (produce's render-props contract:
191
203
  // zoomPlan alone can't reach the punch) — cover-crop is all that remains.
192
- const punches = staticCamera ? spans.map(() => 1) : punchScales(spans);
204
+ // Otherwise the punch plan's scale and mask apply when present; `punch?.`
205
+ // collapses both absent and null to undefined, which punchScales reads as
206
+ // the legacy 1.07-everywhere the render itself falls back to.
207
+ const punches = staticCamera
208
+ ? spans.map(() => 1)
209
+ : punchScales(spans, punch?.scale, undefined, punch?.allowed);
193
210
 
194
211
  // Per-clip frame geometry. A sub-frame span still occupies one frame on
195
212
  // both axes — a zero-length clipitem is undefined importer behavior, same
package/src/fonts.ts ADDED
@@ -0,0 +1,17 @@
1
+ import { fileURLToPath } from "node:url";
2
+
3
+ /**
4
+ * Node-side half of the bundled caption font (see `captions.ts` for the
5
+ * browser-safe constants and the reason the font ships at all): the absolute
6
+ * path produce copies into the render's public dir. Split from captions.ts
7
+ * because `node:url` must never enter the Remotion bundle — captions.ts is
8
+ * on the `@ossclip/core/browser` surface.
9
+ *
10
+ * The URL below is face.ts's pico-cascade load shape, and the packaging test
11
+ * (R22 §111) scans this source for exactly that shape to assert the tarball
12
+ * carries the file — which is also why this comment doesn't spell the
13
+ * pattern out literally: the scanner reads comments too.
14
+ */
15
+ export function nastaliqFontFile(): string {
16
+ return fileURLToPath(new URL("../assets/fonts/NotoNastaliqUrdu-Bold.ttf", import.meta.url));
17
+ }
package/src/index.ts CHANGED
@@ -16,6 +16,8 @@ export * from "./clip";
16
16
  export * from "./blooper";
17
17
  export * from "./retake";
18
18
  export * from "./captions";
19
+ export * from "./fonts";
20
+ export * from "./dictionary";
19
21
  export * from "./zoom";
20
22
  export * from "./grounding";
21
23
  export * from "./cta";
@@ -25,6 +27,7 @@ export * from "./normalize";
25
27
  export * from "./framing";
26
28
  export * from "./face";
27
29
  export * from "./cover";
30
+ export * from "./thumbnail";
28
31
  export * from "./source-text";
29
32
  export * from "./report";
30
33
  export * from "./export-markers";
package/src/ingest.ts CHANGED
@@ -85,22 +85,108 @@ export async function extractAudio(tools: IngestTools, src: string, outWav: stri
85
85
  ]);
86
86
  }
87
87
 
88
+ /**
89
+ * Headroom over the exact displayed size so a zoomed span never renders from
90
+ * below-native pixels (2026-08-17 render-speed pass). The two motion drivers
91
+ * stack to at most ZOOM_MAX_SCALE (1.05) × FACE_PUNCH_SCALE (1.015) ≈ 1.066
92
+ * on any frame a new run emits, so 1.1 covers the worst momentary
93
+ * magnification with margin. (The legacy punch-less contract renders 1.07 —
94
+ * still under 1.1 — and pre-existing render-props keep their own full-res
95
+ * mezzanine anyway; see `mezzanineFileName`.)
96
+ */
97
+ export const MEZZANINE_SCALE_MARGIN = 1.1;
98
+
99
+ export interface MezzanineScale {
100
+ width: number;
101
+ height: number;
102
+ fps: number;
103
+ }
104
+
105
+ /** Nearest even dimension — yuv420 chroma subsampling needs both axes even. */
106
+ function evenDim(v: number): number {
107
+ return Math.max(2, 2 * Math.round(v / 2));
108
+ }
109
+
110
+ /**
111
+ * The size and rate the mezzanine should be encoded at, or null when the
112
+ * source is already no larger than the render needs (2026-08-17 render-speed
113
+ * pass). Remotion's OffthreadVideo extracts EVERY sampled frame via ffmpeg
114
+ * on the CPU, so decode cost scales with pixels × fps — a 3456x2234@60
115
+ * source feeding a 1920x1080@30 render pays ~4.6× the pixels and 2× the
116
+ * frames the render ever shows.
117
+ *
118
+ * The target is the size at which the source is DISPLAYED: for `cover` the
119
+ * larger frame/source axis ratio (overflow is cropped, not shown), for
120
+ * `contain` the smaller (the whole frame fits inside). That target gets
121
+ * MEZZANINE_SCALE_MARGIN of headroom for the motion drivers, is rounded
122
+ * even for yuv420, and is capped at native — scaling UP would soften every
123
+ * frame for zero decode saved.
124
+ *
125
+ * fps: min(source, output) — frames the render never samples are pure decode
126
+ * waste. Safe because EDL `srcIn`/`srcOut` are SECONDS, not frame indexes:
127
+ * a 60→30 resample moves a cut boundary by at most 1/60s, the same
128
+ * magnitude whisper's word stamps already jitter by.
129
+ */
130
+ export function mezzanineScale(
131
+ source: { width: number; height: number; fps: number },
132
+ frame: { width: number; height: number; fps: number },
133
+ sourceFit: "cover" | "contain",
134
+ ): MezzanineScale | null {
135
+ if (source.width <= 0 || source.height <= 0) return null;
136
+ const displayed =
137
+ sourceFit === "contain"
138
+ ? Math.min(frame.width / source.width, frame.height / source.height)
139
+ : Math.max(frame.width / source.width, frame.height / source.height);
140
+ const k = Math.min(1, displayed * MEZZANINE_SCALE_MARGIN);
141
+ // At the cap, keep the source's exact dims — even-rounding a size that is
142
+ // not being resampled would manufacture a 1px no-op rescale.
143
+ const width = k < 1 ? evenDim(source.width * k) : source.width;
144
+ const height = k < 1 ? evenDim(source.height * k) : source.height;
145
+ const fps = Math.min(source.fps, frame.fps);
146
+ if (width === source.width && height === source.height && fps >= source.fps) return null;
147
+ return { width, height, fps };
148
+ }
149
+
150
+ /**
151
+ * The mezzanine's filename, which IS its cache key: mezzanines are
152
+ * existence-keyed in the workdir, so the scale decision must live in the
153
+ * name — a pre-pass full-res `mezzanine.mp4` must never satisfy a run that
154
+ * will emit mezzanine-sized framing windows (they would land on a file with
155
+ * ~1.6× their pixel space and crop the wrong picture). Unscaled runs keep
156
+ * the legacy names so existing workdir caches stay valid; a scaled run
157
+ * rebuilds once under its own name and old workdirs' render-props keep
158
+ * referencing (and rendering from) the file they were emitted against.
159
+ */
160
+ export function mezzanineFileName(cropped: boolean, scale: MezzanineScale | null): string {
161
+ const base = cropped ? "mezzanine-content" : "mezzanine";
162
+ if (!scale) return `${base}.mp4`;
163
+ return `${base}-${scale.width}x${scale.height}@${Math.round(scale.fps)}.mp4`;
164
+ }
165
+
88
166
  /**
89
167
  * Re-encode with dense keyframes so EDL playback (<OffthreadVideo> with many
90
168
  * small trims) seeks fast. Optional — most sources play fine untouched —
91
169
  * EXCEPT when the source is letterboxed: then this pass also trims the baked
92
170
  * bars (`crop`), so everything downstream sees the picture, not picture+bars
93
171
  * (PLAN Task 7), and the pass stops being optional.
172
+ *
173
+ * `scale` (from `mezzanineScale`) downsizes to display size in the SAME
174
+ * pass, crop first — the scale dims are computed on the post-crop picture.
94
175
  */
95
176
  export async function makeMezzanine(
96
177
  tools: IngestTools,
97
178
  src: string,
98
179
  out: string,
99
- opts: { cropVf?: string } = {},
180
+ opts: { cropVf?: string; scale?: MezzanineScale } = {},
100
181
  ): Promise<void> {
182
+ const vf = [
183
+ ...(opts.cropVf ? [opts.cropVf] : []),
184
+ ...(opts.scale ? [`scale=${opts.scale.width}:${opts.scale.height}`] : []),
185
+ ].join(",");
101
186
  await run(tools.ffmpegPath, [
102
187
  "-y", "-i", src,
103
- ...(opts.cropVf ? ["-vf", opts.cropVf] : []),
188
+ ...(vf ? ["-vf", vf] : []),
189
+ ...(opts.scale ? ["-r", String(opts.scale.fps)] : []),
104
190
  "-c:v", "libx264", "-preset", "veryfast", "-crf", "18", "-g", "30", "-pix_fmt", "yuv420p",
105
191
  "-c:a", "aac", "-b:a", "192k",
106
192
  out,