@ossclip/core 0.1.24 → 0.1.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/fonts/NotoNastaliqUrdu-Bold.ttf +0 -0
- package/assets/fonts/OFL.txt +93 -0
- package/assets/fonts/README.md +15 -0
- package/package.json +2 -1
- package/src/blooper.ts +91 -8
- package/src/browser.ts +8 -0
- package/src/captions.ts +45 -2
- package/src/concat.ts +5 -5
- package/src/config.ts +110 -0
- package/src/content-rect-detect.ts +16 -4
- package/src/content-rect.ts +211 -0
- package/src/cover.ts +21 -5
- package/src/cutlist.ts +38 -6
- package/src/dictionary.ts +56 -0
- package/src/export-premiere-project.ts +26 -9
- package/src/fonts.ts +17 -0
- package/src/index.ts +3 -0
- package/src/ingest.ts +88 -2
- package/src/normalize.ts +273 -127
- package/src/producer/index.ts +1 -0
- package/src/producer/repair.ts +36 -13
- package/src/producer/youtube.ts +434 -0
- package/src/retake.ts +104 -2
- package/src/scene-schema.ts +10 -0
- package/src/thumbnail.ts +412 -0
- package/src/transcribe.ts +22 -0
- package/src/zoom.ts +63 -12
package/src/content-rect.ts
CHANGED
|
@@ -153,6 +153,75 @@ export interface ContentRectSegment {
|
|
|
153
153
|
rect: ContentRect;
|
|
154
154
|
}
|
|
155
155
|
|
|
156
|
+
/**
|
|
157
|
+
* One stretch of the RENDER-TIME framing plan — the props-based successor to
|
|
158
|
+
* the destructive normalization bake (2026-08-16 incident: the bake's crop
|
|
159
|
+
* could only be undone by deleting the re-encoded file; expressed as data,
|
|
160
|
+
* the same window renders as a transform the editor can see and counteract).
|
|
161
|
+
* Emitted into render-props.json as `framingTimeline`; absent means "no plan"
|
|
162
|
+
* and every pre-existing render-props renders unchanged.
|
|
163
|
+
*/
|
|
164
|
+
export interface FramingSegment {
|
|
165
|
+
/** Source seconds — intersects the content timeline and kept spans directly. */
|
|
166
|
+
startSec: number;
|
|
167
|
+
endSec: number;
|
|
168
|
+
/** Crop window in SOURCE pixels, all windows sharing one aspect. */
|
|
169
|
+
window: { x: number; y: number; w: number; h: number };
|
|
170
|
+
/** What the window is anchored on; "screen" windows are centered clips. */
|
|
171
|
+
subject: "face" | "screen";
|
|
172
|
+
/** The subject's anchor point inside the window, 0..1 both axes. */
|
|
173
|
+
bias: { x: number; y: number };
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* Rescale framing windows from TRUE source pixels into the pixel space of a
|
|
178
|
+
* display-sized mezzanine (2026-08-17 render-speed pass). `planNormalization`
|
|
179
|
+
* keeps working in true source pixels — analysis runs on the source — and
|
|
180
|
+
* the render-props emission scales the windows so window space === the space
|
|
181
|
+
* of the file the render actually plays. Per-axis factors, not one scalar:
|
|
182
|
+
* yuv420 even-rounding makes the two axes' ratios differ by a fraction of a
|
|
183
|
+
* percent, and scaling both by one axis's factor could push a right-edge
|
|
184
|
+
* window past the scaled file's width. Times, subject and bias are
|
|
185
|
+
* scale-invariant and pass through untouched.
|
|
186
|
+
*/
|
|
187
|
+
export function scaleFramingWindows(
|
|
188
|
+
timeline: readonly FramingSegment[],
|
|
189
|
+
factor: { x: number; y: number },
|
|
190
|
+
): FramingSegment[] {
|
|
191
|
+
return timeline.map((s) => ({
|
|
192
|
+
...s,
|
|
193
|
+
window: {
|
|
194
|
+
x: s.window.x * factor.x,
|
|
195
|
+
y: s.window.y * factor.y,
|
|
196
|
+
w: s.window.w * factor.x,
|
|
197
|
+
h: s.window.h * factor.y,
|
|
198
|
+
},
|
|
199
|
+
}));
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* The same rescale for the fit-fallback's content timeline — its rects are
|
|
204
|
+
* source pixels too, and they reach the renderer only alongside a
|
|
205
|
+
* `sourceSize` that must describe the played file (see the render-props
|
|
206
|
+
* emission in produce.ts). `full` survives the scale: it means "this rect IS
|
|
207
|
+
* the whole frame", which a uniform resample does not change.
|
|
208
|
+
*/
|
|
209
|
+
export function scaleContentTimeline(
|
|
210
|
+
timeline: readonly ContentRectSegment[],
|
|
211
|
+
factor: { x: number; y: number },
|
|
212
|
+
): ContentRectSegment[] {
|
|
213
|
+
return timeline.map((s) => ({
|
|
214
|
+
...s,
|
|
215
|
+
rect: {
|
|
216
|
+
...s.rect,
|
|
217
|
+
x: s.rect.x * factor.x,
|
|
218
|
+
y: s.rect.y * factor.y,
|
|
219
|
+
w: s.rect.w * factor.x,
|
|
220
|
+
h: s.rect.h * factor.y,
|
|
221
|
+
},
|
|
222
|
+
}));
|
|
223
|
+
}
|
|
224
|
+
|
|
156
225
|
/**
|
|
157
226
|
* A framing run must survive BOTH of these to be believed. Together they are
|
|
158
227
|
* what replaces the union rule's protection (PLAN Task C, step C2).
|
|
@@ -270,6 +339,148 @@ export function contentRectTimeline(
|
|
|
270
339
|
}));
|
|
271
340
|
}
|
|
272
341
|
|
|
342
|
+
/**
|
|
343
|
+
* Materiality thresholds (2026-08-16 landscape screen-recording over-crop — a
|
|
344
|
+
* 72px dark strip (2.08% inset, 1px past SNAP_FRAC's 69.1px tolerance) plus a
|
|
345
|
+
* 15.4s/1435s outlier segment destroyed 55% of the picture).
|
|
346
|
+
*
|
|
347
|
+
* SNAP_FRAC decides whether a bar is worth TRIMMING; this bound decides
|
|
348
|
+
* whether a rect difference is worth calling a FRAMING CHANGE — a far more
|
|
349
|
+
* destructive claim, because a mixed-framing verdict sends the whole take
|
|
350
|
+
* into normalization. The incident's jitter measured 2.0–2.3% per side, so
|
|
351
|
+
* the bound sits at 3.5%: comfortably above jitter, comfortably below any
|
|
352
|
+
* real letterbox (the motivating 144bbfb strip insets 34% per side).
|
|
353
|
+
*/
|
|
354
|
+
export const MATERIAL_INSET_FRAC = 0.035;
|
|
355
|
+
|
|
356
|
+
/**
|
|
357
|
+
* A rect keeping at least this share of the frame's area, at (within
|
|
358
|
+
* MATERIAL_ASPECT_TOL of) the frame's own aspect, is the frame minus
|
|
359
|
+
* measurement noise — treating it as a distinct framing would re-crop every
|
|
360
|
+
* downstream geometry to chase pixels nobody can see missing.
|
|
361
|
+
*/
|
|
362
|
+
export const MATERIAL_AREA_FRAC = 0.92;
|
|
363
|
+
const MATERIAL_ASPECT_TOL = 0.05;
|
|
364
|
+
|
|
365
|
+
/**
|
|
366
|
+
* A framing class — every segment sharing one rect, contiguous or not —
|
|
367
|
+
* totalling under this share of the runtime is an anomaly, not a framing
|
|
368
|
+
* change. MIN_RUN_SEC only guards a single run; the incident's 15.4s outlier
|
|
369
|
+
* (1.1% of 1435s) sailed past it and single-handedly set the canvas aspect
|
|
370
|
+
* for the whole take.
|
|
371
|
+
*/
|
|
372
|
+
export const MIN_FRAMING_CLASS_FRAC = 0.05;
|
|
373
|
+
|
|
374
|
+
/**
|
|
375
|
+
* Re-judge a measured timeline against what a framing change must AMOUNT TO
|
|
376
|
+
* before it is believed (2026-08-16 incident above).
|
|
377
|
+
*
|
|
378
|
+
* Two passes. First, per segment: a rect whose every side inset is under
|
|
379
|
+
* MATERIAL_INSET_FRAC, or that keeps MATERIAL_AREA_FRAC of the frame at the
|
|
380
|
+
* frame's own aspect, is reclassified as the full frame — it differs from the
|
|
381
|
+
* frame by less than a framing change is worth. Second, per CLASS: framing
|
|
382
|
+
* classes totalling under MIN_FRAMING_CLASS_FRAC of the runtime are absorbed
|
|
383
|
+
* into their longer neighbour, same shape as the run-absorption loop in
|
|
384
|
+
* `contentRectTimeline` (drop one, retry until stable, then merge agreeing
|
|
385
|
+
* neighbours). A uniform source comes out as exactly one segment.
|
|
386
|
+
*
|
|
387
|
+
* Pure — applied on top of the cached raw timeline, so an already-measured
|
|
388
|
+
* source re-classifies on replay without a cache version bump.
|
|
389
|
+
*/
|
|
390
|
+
export function materializeTimeline(
|
|
391
|
+
timeline: readonly ContentRectSegment[],
|
|
392
|
+
width: number,
|
|
393
|
+
height: number,
|
|
394
|
+
durationSec: number,
|
|
395
|
+
): ContentRectSegment[] {
|
|
396
|
+
if (timeline.length === 0 || width <= 0 || height <= 0) {
|
|
397
|
+
return timeline.map((s) => ({ ...s }));
|
|
398
|
+
}
|
|
399
|
+
const whole: ContentRect = { x: 0, y: 0, w: width, h: height, full: true };
|
|
400
|
+
const frameAspect = width / height;
|
|
401
|
+
|
|
402
|
+
const immaterial = (r: ContentRect): boolean => {
|
|
403
|
+
if (r.full) return true;
|
|
404
|
+
const insets = [
|
|
405
|
+
r.x / width,
|
|
406
|
+
r.y / height,
|
|
407
|
+
(width - (r.x + r.w)) / width,
|
|
408
|
+
(height - (r.y + r.h)) / height,
|
|
409
|
+
];
|
|
410
|
+
if (insets.every((f) => f < MATERIAL_INSET_FRAC)) return true;
|
|
411
|
+
const areaFrac = (r.w * r.h) / (width * height);
|
|
412
|
+
const aspectDelta = Math.abs(r.w / r.h - frameAspect) / frameAspect;
|
|
413
|
+
return areaFrac >= MATERIAL_AREA_FRAC && aspectDelta < MATERIAL_ASPECT_TOL;
|
|
414
|
+
};
|
|
415
|
+
|
|
416
|
+
const segs = timeline.map((s) => ({
|
|
417
|
+
startSec: s.startSec,
|
|
418
|
+
endSec: s.endSec,
|
|
419
|
+
rect: immaterial(s.rect) ? whole : s.rect,
|
|
420
|
+
}));
|
|
421
|
+
|
|
422
|
+
// Duration of each segment's whole CLASS — the outlier that motivated this
|
|
423
|
+
// appeared as one contiguous run, but a strip flickering in and out would
|
|
424
|
+
// split into several, and each alone dodging the threshold must not let the
|
|
425
|
+
// class as a whole survive.
|
|
426
|
+
const classTotals = (list: typeof segs): number[] => {
|
|
427
|
+
const reps: ContentRect[] = [];
|
|
428
|
+
const totals: number[] = [];
|
|
429
|
+
const ids = list.map((s) => {
|
|
430
|
+
let id = reps.findIndex((r) => sameFraming(r, s.rect, width, height));
|
|
431
|
+
if (id === -1) {
|
|
432
|
+
id = reps.length;
|
|
433
|
+
reps.push(s.rect);
|
|
434
|
+
totals.push(0);
|
|
435
|
+
}
|
|
436
|
+
totals[id]! += s.endSec - s.startSec;
|
|
437
|
+
return id;
|
|
438
|
+
});
|
|
439
|
+
return ids.map((id) => totals[id]!);
|
|
440
|
+
};
|
|
441
|
+
|
|
442
|
+
// Absorb, then retry until stable: dropping one segment changes its
|
|
443
|
+
// neighbours' class totals and can make them adjacent and mergeable.
|
|
444
|
+
let changed = true;
|
|
445
|
+
while (changed && segs.length > 1) {
|
|
446
|
+
changed = false;
|
|
447
|
+
const totals = classTotals(segs);
|
|
448
|
+
for (let i = 0; i < segs.length; i++) {
|
|
449
|
+
if (totals[i]! >= MIN_FRAMING_CLASS_FRAC * durationSec) continue;
|
|
450
|
+
const r = segs[i]!;
|
|
451
|
+
// Absorb into the LONGER neighbour — the one more likely to be the truth.
|
|
452
|
+
const prev = segs[i - 1];
|
|
453
|
+
const next = segs[i + 1];
|
|
454
|
+
const into = !prev
|
|
455
|
+
? next
|
|
456
|
+
: !next
|
|
457
|
+
? prev
|
|
458
|
+
: prev.endSec - prev.startSec >= next.endSec - next.startSec
|
|
459
|
+
? prev
|
|
460
|
+
: next;
|
|
461
|
+
if (!into) continue;
|
|
462
|
+
into.startSec = Math.min(into.startSec, r.startSec);
|
|
463
|
+
into.endSec = Math.max(into.endSec, r.endSec);
|
|
464
|
+
segs.splice(i, 1);
|
|
465
|
+
changed = true;
|
|
466
|
+
break;
|
|
467
|
+
}
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
// Merge neighbours that now agree, so a source whose every segment was
|
|
471
|
+
// reclassified collapses to the single-segment (uniform) shape callers test.
|
|
472
|
+
const merged: typeof segs = [];
|
|
473
|
+
for (const s of segs) {
|
|
474
|
+
const last = merged[merged.length - 1];
|
|
475
|
+
if (last && sameFraming(last.rect, s.rect, width, height)) {
|
|
476
|
+
last.endSec = Math.max(last.endSec, s.endSec);
|
|
477
|
+
} else {
|
|
478
|
+
merged.push({ ...s });
|
|
479
|
+
}
|
|
480
|
+
}
|
|
481
|
+
return merged;
|
|
482
|
+
}
|
|
483
|
+
|
|
273
484
|
/**
|
|
274
485
|
* The exact framing-change instant inside a densely-sampled window
|
|
275
486
|
* (NORMALIZE plan, boundary refinement).
|
package/src/cover.ts
CHANGED
|
@@ -136,9 +136,17 @@ export function laplacianVariance(pixels: Uint8Array, w: number, h: number): num
|
|
|
136
136
|
}
|
|
137
137
|
|
|
138
138
|
/**
|
|
139
|
-
* Score a candidate.
|
|
140
|
-
* speaker is a cover for a different video — and among frames
|
|
141
|
-
* sharpness decides. Earlier frames win ties so the cover
|
|
139
|
+
* Score a candidate. On a "face" take a face is close to mandatory — a cover
|
|
140
|
+
* without the speaker is a cover for a different video — and among frames
|
|
141
|
+
* that have one, sharpness decides. Earlier frames win ties so the cover
|
|
142
|
+
* matches the opening.
|
|
143
|
+
*
|
|
144
|
+
* On a "screen" take the face weight drops to ZERO (2026-08-16): a Facebook
|
|
145
|
+
* reel playing inside a 21-minute screen recording put a STRANGER'S face on
|
|
146
|
+
* the cover, because face×2 hunts any face at all and on a screen-subject
|
|
147
|
+
* take every face in frame is content, not the speaker. Sharpness and
|
|
148
|
+
* earliness alone pick that cover. Default "face" so portrait/talking-head
|
|
149
|
+
* runs score exactly as before.
|
|
142
150
|
*/
|
|
143
151
|
export function scoreCandidate(c: {
|
|
144
152
|
timeSec: number;
|
|
@@ -146,11 +154,13 @@ export function scoreCandidate(c: {
|
|
|
146
154
|
sharpness: number;
|
|
147
155
|
hasFace: boolean;
|
|
148
156
|
maxSharpness: number;
|
|
157
|
+
subject?: "face" | "screen";
|
|
149
158
|
}): number {
|
|
159
|
+
const faceWeight = c.subject === "screen" ? 0 : 2;
|
|
150
160
|
const face = c.hasFace ? 1 : 0;
|
|
151
161
|
const sharp = c.maxSharpness > 0 ? c.sharpness / c.maxSharpness : 0;
|
|
152
162
|
const earliness = 1 - Math.min(1, c.timeSec / Math.max(1e-6, c.durationSec));
|
|
153
|
-
return face *
|
|
163
|
+
return face * faceWeight + sharp + earliness * 0.3;
|
|
154
164
|
}
|
|
155
165
|
|
|
156
166
|
export interface PickCoverOptions {
|
|
@@ -172,6 +182,12 @@ export interface PickCoverOptions {
|
|
|
172
182
|
* canvas that is two-thirds baked-in black bar.
|
|
173
183
|
*/
|
|
174
184
|
cropVf?: string;
|
|
185
|
+
/**
|
|
186
|
+
* The take's whole-frame subject (`faceSubject`'s verdict). "screen" zeroes
|
|
187
|
+
* the face weight in `scoreCandidate` — see its doc comment for the
|
|
188
|
+
* 2026-08-16 stranger's-face incident. Default "face".
|
|
189
|
+
*/
|
|
190
|
+
subject?: "face" | "screen";
|
|
175
191
|
}
|
|
176
192
|
|
|
177
193
|
/**
|
|
@@ -229,7 +245,7 @@ export async function pickCoverFrame(
|
|
|
229
245
|
for (const r of raw) {
|
|
230
246
|
candidates.push({
|
|
231
247
|
...r,
|
|
232
|
-
score: scoreCandidate({ ...r, durationSec: window, maxSharpness }),
|
|
248
|
+
score: scoreCandidate({ ...r, durationSec: window, maxSharpness, subject: opts.subject }),
|
|
233
249
|
});
|
|
234
250
|
}
|
|
235
251
|
candidates.sort((a, b) => b.score - a.score);
|
package/src/cutlist.ts
CHANGED
|
@@ -42,6 +42,18 @@ const SILENCE_PAD_IN = 0.06;
|
|
|
42
42
|
const SILENCE_PAD_OUT = 0.1;
|
|
43
43
|
/** Kept fragments shorter than this (holding no word) are folded into the cut. */
|
|
44
44
|
const MIN_KEEP = 0.25;
|
|
45
|
+
/**
|
|
46
|
+
* A WORDED kept gap up to this long is still folded when a flanking removal
|
|
47
|
+
* is a `retake` cut (2026-08-16 incident): whisper stamped a debris "And" at
|
|
48
|
+
* 670.0–670.4 between a marker-terminated blooper cut and the silence cut
|
|
49
|
+
* that followed, and `hasProtectedWordInside` — correctly — refused the
|
|
50
|
+
* wordless fold, so a 0.4s one-word sliver of an abandoned sentence shipped
|
|
51
|
+
* as a choppy blip. A retake cut already means "this whole attempt is
|
|
52
|
+
* unusable", so a sub-0.35s word fragment glued to it is debris of that same
|
|
53
|
+
* attempt, not content. Silence/silence neighbors get NO such license: their
|
|
54
|
+
* worded gaps stay protected exactly as §124's follow-up demands.
|
|
55
|
+
*/
|
|
56
|
+
const MIN_KEEP_AFTER_RETAKE = 0.35;
|
|
45
57
|
|
|
46
58
|
interface Removal {
|
|
47
59
|
start: number;
|
|
@@ -74,8 +86,19 @@ export interface BuildCutlistArgs {
|
|
|
74
86
|
* still has no judgement of its own about what a bad take looks like, it
|
|
75
87
|
* just folds whichever spans two independent detectors handed it into the
|
|
76
88
|
* one partition.
|
|
89
|
+
*
|
|
90
|
+
* `confidence` optionally overrides the 0.9 default below — the
|
|
91
|
+
* exact-prefix restart rule (retake.ts, RESTART_PREFIX_CONFIDENCE) earns
|
|
92
|
+
* 0.85, one notch less than a full similarity match, and the caller is the
|
|
93
|
+
* one who knows which rule found the span.
|
|
77
94
|
*/
|
|
78
|
-
retakes?: readonly {
|
|
95
|
+
retakes?: readonly {
|
|
96
|
+
startWord: number;
|
|
97
|
+
endWord: number;
|
|
98
|
+
startSec: number;
|
|
99
|
+
endSec: number;
|
|
100
|
+
confidence?: number;
|
|
101
|
+
}[];
|
|
79
102
|
}
|
|
80
103
|
|
|
81
104
|
export function buildCutlist({
|
|
@@ -88,9 +111,9 @@ export function buildCutlist({
|
|
|
88
111
|
}: BuildCutlistArgs): Segment[] {
|
|
89
112
|
const keepAll: Segment[] = [{ srcIn: 0, srcOut: duration, kind: "keep" }];
|
|
90
113
|
// `exact` means exact: it is the escape hatch for "touch nothing", and a
|
|
91
|
-
// blooper or retake cut is still a cut. --blooper-marker
|
|
92
|
-
//
|
|
93
|
-
// the user typed second does not get to win.
|
|
114
|
+
// blooper or retake cut is still a cut. --blooper-marker (which also
|
|
115
|
+
// switches on retake collapse, 2026-08-16 gate) with --cleanup exact is a
|
|
116
|
+
// contradiction, and the flag the user typed second does not get to win.
|
|
94
117
|
if (level === "exact") return keepAll;
|
|
95
118
|
const policy = POLICIES[level];
|
|
96
119
|
const words = transcript.words;
|
|
@@ -126,7 +149,7 @@ export function buildCutlist({
|
|
|
126
149
|
start: r.startSec,
|
|
127
150
|
end: r.endSec,
|
|
128
151
|
reason: "retake",
|
|
129
|
-
confidence: 0.9,
|
|
152
|
+
confidence: r.confidence ?? 0.9,
|
|
130
153
|
source: "acoustic",
|
|
131
154
|
});
|
|
132
155
|
}
|
|
@@ -268,7 +291,16 @@ export function buildCutlist({
|
|
|
268
291
|
const overlapping = prev !== undefined && gap <= 1e-6;
|
|
269
292
|
const wordless = prev !== undefined && !hasProtectedWordInside(prev.end, r.start);
|
|
270
293
|
const foldableGap = wordless && gap <= Math.max(MIN_KEEP, policy.pauseMin);
|
|
271
|
-
|
|
294
|
+
// The one exception to word protection (MIN_KEEP_AFTER_RETAKE's why): a
|
|
295
|
+
// worded sliver this short survives only between two cuts, and when one
|
|
296
|
+
// of those cuts is a retake the sliver is debris of the discarded
|
|
297
|
+
// attempt. Requires the retake FLANK, not just the length — a worded gap
|
|
298
|
+
// between two silence removals must never fold, whatever its size.
|
|
299
|
+
const retakeSliver =
|
|
300
|
+
prev !== undefined &&
|
|
301
|
+
gap <= MIN_KEEP_AFTER_RETAKE &&
|
|
302
|
+
(prev.reason === "retake" || r.reason === "retake");
|
|
303
|
+
if (prev && (overlapping || foldableGap || retakeSliver)) {
|
|
272
304
|
const prevDur = prev.end - prev.start;
|
|
273
305
|
const curDur = r.end - r.start;
|
|
274
306
|
prev.end = Math.max(prev.end, r.end);
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import type { Transcript } from "./schema";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Deterministic casing for the user's dictionary (F4, 2026-08-16).
|
|
5
|
+
*
|
|
6
|
+
* Whisper biasing and the LLM repair pass get a word's SPELLING right; what
|
|
7
|
+
* neither guarantees is its CASE — a decoder nudged into "json" has still
|
|
8
|
+
* lost the acronym. This pass is the deterministic last word: any token that
|
|
9
|
+
* IS a dictionary term (exact case-insensitive match, punctuation aside)
|
|
10
|
+
* takes the term's canonical casing.
|
|
11
|
+
*
|
|
12
|
+
* Deliberately exact-match only: "Jason" is NEVER touched, even with "JSON"
|
|
13
|
+
* in the dictionary — deciding that a different word is a mishearing of a
|
|
14
|
+
* term is phonetic judgement, and that is the LLM repair pass's job, behind
|
|
15
|
+
* its guards. A deterministic pass that rewrote near-misses would be the
|
|
16
|
+
* un-gated rewrite pass §17 exists to forbid.
|
|
17
|
+
*
|
|
18
|
+
* Not on the browser surface: only `produce` runs it, and keeping it out of
|
|
19
|
+
* browser.ts keeps the Remotion bundle's import graph untouched.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* The same leading/trailing bounds `normalizeToken` (analyze.ts) strips, but
|
|
24
|
+
* keeping the pieces: the punctuation must survive the swap ("json." →
|
|
25
|
+
* "JSON.", quotes and commas intact), so the strip is a split here, not a
|
|
26
|
+
* deletion.
|
|
27
|
+
*/
|
|
28
|
+
const TOKEN_BOUNDS = /^([^\p{L}\p{N}]*)([\s\S]*?)([^\p{L}\p{N}-]*)$/u;
|
|
29
|
+
|
|
30
|
+
export function canonicalizeDictionaryCasing(
|
|
31
|
+
transcript: Transcript,
|
|
32
|
+
dictionary: readonly string[],
|
|
33
|
+
): Transcript {
|
|
34
|
+
// Keyed on the term's own lowercase, valued with its typed casing. A later
|
|
35
|
+
// duplicate (differing only in case) loses to the first — the user's list
|
|
36
|
+
// order is the only precedence signal there is.
|
|
37
|
+
const canonical = new Map<string, string>();
|
|
38
|
+
for (const raw of dictionary) {
|
|
39
|
+
const term = raw.trim();
|
|
40
|
+
const key = term.toLowerCase();
|
|
41
|
+
if (term && !canonical.has(key)) canonical.set(key, term);
|
|
42
|
+
}
|
|
43
|
+
if (canonical.size === 0) return transcript;
|
|
44
|
+
return {
|
|
45
|
+
...transcript,
|
|
46
|
+
words: transcript.words.map((w) => {
|
|
47
|
+
const [, lead = "", core = "", trail = ""] = TOKEN_BOUNDS.exec(w.text) ?? [];
|
|
48
|
+
const want = canonical.get(core.toLowerCase());
|
|
49
|
+
// `want !== core` also guards the degenerate empty core; strict
|
|
50
|
+
// equality-of-lowercase above means this is a CASING change only —
|
|
51
|
+
// never a respelling, never added punctuation.
|
|
52
|
+
if (want === undefined || want === core) return w;
|
|
53
|
+
return { ...w, text: `${lead}${want}${trail}` };
|
|
54
|
+
}),
|
|
55
|
+
};
|
|
56
|
+
}
|
|
@@ -39,22 +39,34 @@ export interface PremiereProjectInput {
|
|
|
39
39
|
zoomPlan: readonly ZoomSegment[];
|
|
40
40
|
/** render-props' flag: true disables BOTH motion layers (zoom AND punch). */
|
|
41
41
|
staticCamera?: boolean;
|
|
42
|
+
/**
|
|
43
|
+
* render-props' `punch` — the face-only jump-cut plan (2026-08-16, scenes'
|
|
44
|
+
* punch-plan.ts): its scale replaces the legacy 1.07 and its mask gates
|
|
45
|
+
* which spans punch. Absent means the LEGACY contract, so a pre-feature
|
|
46
|
+
* workdir exports exactly the camera its render had.
|
|
47
|
+
*/
|
|
48
|
+
punch?: { scale: number; allowed: readonly boolean[] } | null;
|
|
42
49
|
/** The OUTPUT frame (production.render), e.g. 1080×1920. */
|
|
43
50
|
frame: { width: number; height: number };
|
|
44
51
|
}
|
|
45
52
|
|
|
46
53
|
/**
|
|
47
|
-
* The render's jump-cut concealer, replicated from
|
|
48
|
-
*
|
|
49
|
-
* threshold must all match, or
|
|
50
|
-
*
|
|
51
|
-
*
|
|
52
|
-
*
|
|
54
|
+
* The render's jump-cut concealer, replicated from scenes' `punchScalesFor`
|
|
55
|
+
* (punch-plan.ts, the loop EdlVideo renders) — the alternating toggle, the
|
|
56
|
+
* srcIn−prev.srcOut gap, and the INCLUSIVE >= threshold must all match, or
|
|
57
|
+
* the exported project punches different clips than the render did. The
|
|
58
|
+
* `allowed` mask must match too, down to the parity rule: the toggle flips
|
|
59
|
+
* on EVERY qualifying gap, masked spans included (stable indexing), and a
|
|
60
|
+
* masked span renders its punched turn at 1. `allowed[i] !== false` so a
|
|
61
|
+
* short mask reads as allowed, same as no mask at all. Defaults mirror
|
|
62
|
+
* EdlVideoProps; they are parameters so a drift in either place shows up as
|
|
63
|
+
* a failing hand-computed test, not a silent divergence.
|
|
53
64
|
*/
|
|
54
65
|
export function punchScales(
|
|
55
66
|
spans: readonly KeptSpan[],
|
|
56
67
|
punchInScale = 1.07,
|
|
57
68
|
punchThresholdSec = 0.15,
|
|
69
|
+
allowed?: readonly boolean[] | null,
|
|
58
70
|
): number[] {
|
|
59
71
|
const out: number[] = [];
|
|
60
72
|
let punched = false;
|
|
@@ -62,7 +74,7 @@ export function punchScales(
|
|
|
62
74
|
const prev = spans[i - 1];
|
|
63
75
|
const gap = prev ? spans[i]!.srcIn - prev.srcOut : 0;
|
|
64
76
|
if (i > 0 && gap >= punchThresholdSec) punched = !punched;
|
|
65
|
-
out.push(punched ? punchInScale : 1);
|
|
77
|
+
out.push(punched && (!allowed || allowed[i] !== false) ? punchInScale : 1);
|
|
66
78
|
}
|
|
67
79
|
return out;
|
|
68
80
|
}
|
|
@@ -174,7 +186,7 @@ export function srtFromCaptionLines(lines: readonly SrtLine[]): string {
|
|
|
174
186
|
const fmt = (n: number): string => String(Number(n.toFixed(4)));
|
|
175
187
|
|
|
176
188
|
export function buildPremiereProject(input: PremiereProjectInput): { xml: string; srt: string } {
|
|
177
|
-
const { production, spans, captionLines, zoomPlan, staticCamera, frame } = input;
|
|
189
|
+
const { production, spans, captionLines, zoomPlan, staticCamera, punch, frame } = input;
|
|
178
190
|
const { path, probe } = production.source;
|
|
179
191
|
// The whole document runs at the SOURCE rate (the marker exporter's
|
|
180
192
|
// precedent): clip in/out and sequence start/end share one timebase, so no
|
|
@@ -189,7 +201,12 @@ export function buildPremiereProject(input: PremiereProjectInput): { xml: string
|
|
|
189
201
|
const cover = coverTransform(probe, production.source.face, frame);
|
|
190
202
|
// staticCamera kills BOTH motion drivers (produce's render-props contract:
|
|
191
203
|
// zoomPlan alone can't reach the punch) — cover-crop is all that remains.
|
|
192
|
-
|
|
204
|
+
// Otherwise the punch plan's scale and mask apply when present; `punch?.`
|
|
205
|
+
// collapses both absent and null to undefined, which punchScales reads as
|
|
206
|
+
// the legacy 1.07-everywhere the render itself falls back to.
|
|
207
|
+
const punches = staticCamera
|
|
208
|
+
? spans.map(() => 1)
|
|
209
|
+
: punchScales(spans, punch?.scale, undefined, punch?.allowed);
|
|
193
210
|
|
|
194
211
|
// Per-clip frame geometry. A sub-frame span still occupies one frame on
|
|
195
212
|
// both axes — a zero-length clipitem is undefined importer behavior, same
|
package/src/fonts.ts
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import { fileURLToPath } from "node:url";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Node-side half of the bundled caption font (see `captions.ts` for the
|
|
5
|
+
* browser-safe constants and the reason the font ships at all): the absolute
|
|
6
|
+
* path produce copies into the render's public dir. Split from captions.ts
|
|
7
|
+
* because `node:url` must never enter the Remotion bundle — captions.ts is
|
|
8
|
+
* on the `@ossclip/core/browser` surface.
|
|
9
|
+
*
|
|
10
|
+
* The URL below is face.ts's pico-cascade load shape, and the packaging test
|
|
11
|
+
* (R22 §111) scans this source for exactly that shape to assert the tarball
|
|
12
|
+
* carries the file — which is also why this comment doesn't spell the
|
|
13
|
+
* pattern out literally: the scanner reads comments too.
|
|
14
|
+
*/
|
|
15
|
+
export function nastaliqFontFile(): string {
|
|
16
|
+
return fileURLToPath(new URL("../assets/fonts/NotoNastaliqUrdu-Bold.ttf", import.meta.url));
|
|
17
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -16,6 +16,8 @@ export * from "./clip";
|
|
|
16
16
|
export * from "./blooper";
|
|
17
17
|
export * from "./retake";
|
|
18
18
|
export * from "./captions";
|
|
19
|
+
export * from "./fonts";
|
|
20
|
+
export * from "./dictionary";
|
|
19
21
|
export * from "./zoom";
|
|
20
22
|
export * from "./grounding";
|
|
21
23
|
export * from "./cta";
|
|
@@ -25,6 +27,7 @@ export * from "./normalize";
|
|
|
25
27
|
export * from "./framing";
|
|
26
28
|
export * from "./face";
|
|
27
29
|
export * from "./cover";
|
|
30
|
+
export * from "./thumbnail";
|
|
28
31
|
export * from "./source-text";
|
|
29
32
|
export * from "./report";
|
|
30
33
|
export * from "./export-markers";
|
package/src/ingest.ts
CHANGED
|
@@ -85,22 +85,108 @@ export async function extractAudio(tools: IngestTools, src: string, outWav: stri
|
|
|
85
85
|
]);
|
|
86
86
|
}
|
|
87
87
|
|
|
88
|
+
/**
|
|
89
|
+
* Headroom over the exact displayed size so a zoomed span never renders from
|
|
90
|
+
* below-native pixels (2026-08-17 render-speed pass). The two motion drivers
|
|
91
|
+
* stack to at most ZOOM_MAX_SCALE (1.05) × FACE_PUNCH_SCALE (1.015) ≈ 1.066
|
|
92
|
+
* on any frame a new run emits, so 1.1 covers the worst momentary
|
|
93
|
+
* magnification with margin. (The legacy punch-less contract renders 1.07 —
|
|
94
|
+
* still under 1.1 — and pre-existing render-props keep their own full-res
|
|
95
|
+
* mezzanine anyway; see `mezzanineFileName`.)
|
|
96
|
+
*/
|
|
97
|
+
export const MEZZANINE_SCALE_MARGIN = 1.1;
|
|
98
|
+
|
|
99
|
+
export interface MezzanineScale {
|
|
100
|
+
width: number;
|
|
101
|
+
height: number;
|
|
102
|
+
fps: number;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** Nearest even dimension — yuv420 chroma subsampling needs both axes even. */
|
|
106
|
+
function evenDim(v: number): number {
|
|
107
|
+
return Math.max(2, 2 * Math.round(v / 2));
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* The size and rate the mezzanine should be encoded at, or null when the
|
|
112
|
+
* source is already no larger than the render needs (2026-08-17 render-speed
|
|
113
|
+
* pass). Remotion's OffthreadVideo extracts EVERY sampled frame via ffmpeg
|
|
114
|
+
* on the CPU, so decode cost scales with pixels × fps — a 3456x2234@60
|
|
115
|
+
* source feeding a 1920x1080@30 render pays ~4.6× the pixels and 2× the
|
|
116
|
+
* frames the render ever shows.
|
|
117
|
+
*
|
|
118
|
+
* The target is the size at which the source is DISPLAYED: for `cover` the
|
|
119
|
+
* larger frame/source axis ratio (overflow is cropped, not shown), for
|
|
120
|
+
* `contain` the smaller (the whole frame fits inside). That target gets
|
|
121
|
+
* MEZZANINE_SCALE_MARGIN of headroom for the motion drivers, is rounded
|
|
122
|
+
* even for yuv420, and is capped at native — scaling UP would soften every
|
|
123
|
+
* frame for zero decode saved.
|
|
124
|
+
*
|
|
125
|
+
* fps: min(source, output) — frames the render never samples are pure decode
|
|
126
|
+
* waste. Safe because EDL `srcIn`/`srcOut` are SECONDS, not frame indexes:
|
|
127
|
+
* a 60→30 resample moves a cut boundary by at most 1/60s, the same
|
|
128
|
+
* magnitude whisper's word stamps already jitter by.
|
|
129
|
+
*/
|
|
130
|
+
export function mezzanineScale(
|
|
131
|
+
source: { width: number; height: number; fps: number },
|
|
132
|
+
frame: { width: number; height: number; fps: number },
|
|
133
|
+
sourceFit: "cover" | "contain",
|
|
134
|
+
): MezzanineScale | null {
|
|
135
|
+
if (source.width <= 0 || source.height <= 0) return null;
|
|
136
|
+
const displayed =
|
|
137
|
+
sourceFit === "contain"
|
|
138
|
+
? Math.min(frame.width / source.width, frame.height / source.height)
|
|
139
|
+
: Math.max(frame.width / source.width, frame.height / source.height);
|
|
140
|
+
const k = Math.min(1, displayed * MEZZANINE_SCALE_MARGIN);
|
|
141
|
+
// At the cap, keep the source's exact dims — even-rounding a size that is
|
|
142
|
+
// not being resampled would manufacture a 1px no-op rescale.
|
|
143
|
+
const width = k < 1 ? evenDim(source.width * k) : source.width;
|
|
144
|
+
const height = k < 1 ? evenDim(source.height * k) : source.height;
|
|
145
|
+
const fps = Math.min(source.fps, frame.fps);
|
|
146
|
+
if (width === source.width && height === source.height && fps >= source.fps) return null;
|
|
147
|
+
return { width, height, fps };
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* The mezzanine's filename, which IS its cache key: mezzanines are
|
|
152
|
+
* existence-keyed in the workdir, so the scale decision must live in the
|
|
153
|
+
* name — a pre-pass full-res `mezzanine.mp4` must never satisfy a run that
|
|
154
|
+
* will emit mezzanine-sized framing windows (they would land on a file with
|
|
155
|
+
* ~1.6× their pixel space and crop the wrong picture). Unscaled runs keep
|
|
156
|
+
* the legacy names so existing workdir caches stay valid; a scaled run
|
|
157
|
+
* rebuilds once under its own name and old workdirs' render-props keep
|
|
158
|
+
* referencing (and rendering from) the file they were emitted against.
|
|
159
|
+
*/
|
|
160
|
+
export function mezzanineFileName(cropped: boolean, scale: MezzanineScale | null): string {
|
|
161
|
+
const base = cropped ? "mezzanine-content" : "mezzanine";
|
|
162
|
+
if (!scale) return `${base}.mp4`;
|
|
163
|
+
return `${base}-${scale.width}x${scale.height}@${Math.round(scale.fps)}.mp4`;
|
|
164
|
+
}
|
|
165
|
+
|
|
88
166
|
/**
|
|
89
167
|
* Re-encode with dense keyframes so EDL playback (<OffthreadVideo> with many
|
|
90
168
|
* small trims) seeks fast. Optional — most sources play fine untouched —
|
|
91
169
|
* EXCEPT when the source is letterboxed: then this pass also trims the baked
|
|
92
170
|
* bars (`crop`), so everything downstream sees the picture, not picture+bars
|
|
93
171
|
* (PLAN Task 7), and the pass stops being optional.
|
|
172
|
+
*
|
|
173
|
+
* `scale` (from `mezzanineScale`) downsizes to display size in the SAME
|
|
174
|
+
* pass, crop first — the scale dims are computed on the post-crop picture.
|
|
94
175
|
*/
|
|
95
176
|
export async function makeMezzanine(
|
|
96
177
|
tools: IngestTools,
|
|
97
178
|
src: string,
|
|
98
179
|
out: string,
|
|
99
|
-
opts: { cropVf?: string } = {},
|
|
180
|
+
opts: { cropVf?: string; scale?: MezzanineScale } = {},
|
|
100
181
|
): Promise<void> {
|
|
182
|
+
const vf = [
|
|
183
|
+
...(opts.cropVf ? [opts.cropVf] : []),
|
|
184
|
+
...(opts.scale ? [`scale=${opts.scale.width}:${opts.scale.height}`] : []),
|
|
185
|
+
].join(",");
|
|
101
186
|
await run(tools.ffmpegPath, [
|
|
102
187
|
"-y", "-i", src,
|
|
103
|
-
...(
|
|
188
|
+
...(vf ? ["-vf", vf] : []),
|
|
189
|
+
...(opts.scale ? ["-r", String(opts.scale.fps)] : []),
|
|
104
190
|
"-c:v", "libx264", "-preset", "veryfast", "-crf", "18", "-g", "30", "-pix_fmt", "yuv420p",
|
|
105
191
|
"-c:a", "aac", "-b:a", "192k",
|
|
106
192
|
out,
|