@ossclip/core 0.1.24 → 0.1.25

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/normalize.ts CHANGED
@@ -1,6 +1,4 @@
1
- import { rename, rm } from "node:fs/promises";
2
- import { run } from "./exec";
3
- import type { ContentRectSegment } from "./content-rect";
1
+ import { MIN_FRAMING_CLASS_FRAC, type ContentRectSegment } from "./content-rect";
4
2
  import type { WindowFace } from "./face";
5
3
 
6
4
  /**
@@ -14,21 +12,29 @@ import type { WindowFace } from "./face";
14
12
  * Smoothing the boundaries cannot fix that; the output would still alternate
15
13
  * between two shots.
16
14
  *
17
- * The fix is editorial, applied at bake time: pick ONE field of view — the
18
- * tightest the source ever shows, i.e. the strip, since the strip's pixels
19
- * are all those stretches have — and crop every other segment down to a
20
- * window of that same shape, placed on the measured face. The result is a
21
- * single, uniform landscape source with constant apparent framing, and the
22
- * ENTIRE downstream pipeline (face bias, source-text, cover, layouts, zoom)
23
- * runs its ordinary uniform-source path on it. Tight but stable, by choice.
15
+ * The fix is editorial: pick ONE field of view — the tightest the source ever
16
+ * shows, i.e. the strip, since the strip's pixels are all those stretches
17
+ * have — and crop every other segment down to a window of that same shape,
18
+ * placed on the measured face. One consistent apparent framing, by choice.
19
+ *
20
+ * The plan used to be BAKED into a re-encoded file. That ended with the
21
+ * 2026-08-16 incident: an over-eager plan destroyed 55% of a screen
22
+ * recording's picture and the only undo was deleting the baked mp4 — the
23
+ * editor could not even see the crop had happened. The plan is now emitted
24
+ * into render-props.json as `framingTimeline` and applied at render time as a
25
+ * transform (packages/scenes/src/content-crop.ts), fully visible to and
26
+ * counteractable from the editor. The plan's GEOMETRY is unchanged by that
27
+ * move: every window shares one aspect, so per-segment render-time cover
28
+ * shows the same apparent framing on both sides of every boundary — the
29
+ * invariant the bake existed to enforce (144bbfb).
24
30
  *
25
31
  * When even the strip cannot cover the output frame without excessive
26
- * upscaling, normalization refuses (`ok: false`) and the caller falls back to
27
- * render-time FIT — the strip shown at its natural size rather than
28
- * fake-zoomed (option (b)).
32
+ * upscaling — or the plan would discard too much picture — normalization
33
+ * refuses (`ok: false`) and the caller falls back to render-time FIT — the
34
+ * strip shown at its natural size rather than fake-zoomed (option (b)).
29
35
  */
30
36
 
31
- /** One baked stretch: this window of the source, scaled to the canvas. */
37
+ /** One planned stretch: this window of the source, covering the slot. */
32
38
  export interface NormalizeSegment {
33
39
  startSec: number;
34
40
  endSec: number;
@@ -53,6 +59,28 @@ export interface NormalizePlan {
53
59
  * soft fake is worse than an honest fit.
54
60
  */
55
61
  coverUpscale: number;
62
+ /**
63
+ * Duration-weighted mean of the picture area each window discards from its
64
+ * segment's rect (`1 - windowArea / rectArea`). The other half of the
65
+ * quality gate, and the number the refusal log reports: coverUpscale alone
66
+ * measured softness, never loss (2026-08-16 incident).
67
+ */
68
+ areaDiscardWeighted: number;
69
+ /**
70
+ * Per timeline segment, what the window is anchored on. "face" means the
71
+ * segment passed `segmentIsFaceOnly` and its window is sized and placed on
72
+ * the measured face; "screen" means the window is the segment's own rect,
73
+ * centered and clipped to the shared aspect — the picture, not the person,
74
+ * is the subject there (2026-08-16 incident: the PiP was not the subject).
75
+ */
76
+ subject: ("face" | "screen")[];
77
+ /**
78
+ * Per timeline segment, where the subject sits INSIDE its window, both axes
79
+ * in 0..1. For a face segment this is the measured face centre relative to
80
+ * the final window, so a render-time cover crop can keep the head where the
81
+ * plan put it; a screen segment's subject is the whole picture, so 0.5/0.5.
82
+ */
83
+ bias: { x: number; y: number }[];
56
84
  ok: boolean;
57
85
  }
58
86
 
@@ -64,6 +92,52 @@ export interface NormalizePlan {
64
92
  */
65
93
  export const MAX_NORMALIZE_UPSCALE = 2.6;
66
94
 
95
+ /**
96
+ * Ceiling on the duration-weighted mean fraction of picture area a plan may
97
+ * throw away. coverUpscale 0.77 passed while 37% of the frame area was being
98
+ * thrown away — the gate measured softness, never loss (2026-08-16 incident).
99
+ */
100
+ export const MAX_MEAN_AREA_DISCARD = 0.5;
101
+
102
+ /**
103
+ * Ceiling on the picture area a SCREEN-subject segment's window may discard
104
+ * from its own rect. A screen segment's window is its rect clipped to the
105
+ * shared aspect — losing more than this means the shared aspect is genuinely
106
+ * fighting that segment's shape, and cropping screen content slides text and
107
+ * UI out of frame. Face segments are exempt: a face crop discards area by
108
+ * design. Applies to MATERIAL segments only, mirroring the aspect vote — a
109
+ * sliver class gets no vote on the aspect, so it cannot veto the plan for
110
+ * being clipped to it either (the 1.1% dark segment of the 2026-08-16
111
+ * incident loses ~16% to the shared aspect; the duration-weighted mean gate
112
+ * is what bounds slivers).
113
+ */
114
+ export const MAX_SCREEN_AREA_DISCARD = 0.1;
115
+
116
+ /**
117
+ * Smallest face (box height over segment-rect height) that makes a segment
118
+ * "just a face". Equals DEFAULT_FACE.sizeFrac (stage.ts): a face smaller than
119
+ * the assumed arm's-length selfie box is not the frame's subject. 2026-08-16
120
+ * incident: the camera PiP measured 0.119 and the crop chased it, discarding
121
+ * the screen content that WAS the subject; a real talking head measures 0.28+.
122
+ */
123
+ export const FACE_ONLY_MIN_FRAC = 0.22;
124
+
125
+ /**
126
+ * Below this fraction of sampled frames with a detection, the measurement is
127
+ * not confident enough to reframe on — a face seen in under half the looks is
128
+ * as likely a false positive or an occasional glance at a webcam.
129
+ */
130
+ export const FACE_MIN_DETECTION_RATIO = 0.5;
131
+
132
+ /**
133
+ * Extra breathing room, as a fraction of the window height, kept above the
134
+ * crown and below the chin when the window slides to contain the head. The
135
+ * user's rule (2026-08-16): the ENTIRE head including hair stays in frame,
136
+ * with ~1% of margin — touching the frame edge reads as a crop even when
137
+ * nothing is technically cut.
138
+ */
139
+ export const HEAD_WINDOW_MARGIN = 0.01;
140
+
67
141
  /**
68
142
  * A head is about 1.55x the detector's face box tall — the box bounds eyes,
69
143
  * nose and mouth, and `stage.ts` models the crown at 0.35x above it and the
@@ -96,6 +170,25 @@ const median = (xs: number[]): number => {
96
170
  return s.length % 2 ? s[m]! : (s[m - 1]! + s[m]!) / 2;
97
171
  };
98
172
 
173
+ /**
174
+ * Is this segment essentially JUST a face — the only case where a
175
+ * face-anchored reframe is allowed (user decision, 2026-08-16)?
176
+ *
177
+ * Classified on the representative `sizeFrac`, not `sizeFracMax`: a segment
178
+ * is face-only by what it looks like most of the time, not at its one biggest
179
+ * lean-in (sizing, once classified, still uses the max — see planNormalization).
180
+ * The detection ratio guards against reframing on a face the detector barely
181
+ * ever saw.
182
+ */
183
+ export function segmentIsFaceOnly(face: WindowFace | null): boolean {
184
+ if (!face) return false;
185
+ if (face.sizeFrac < FACE_ONLY_MIN_FRAC) return false;
186
+ return (
187
+ face.framesSampled > 0 &&
188
+ face.framesDetected / face.framesSampled >= FACE_MIN_DETECTION_RATIO
189
+ );
190
+ }
191
+
99
192
  /**
100
193
  * Decide the canvas and each segment's crop window.
101
194
  *
@@ -119,6 +212,12 @@ const median = (xs: number[]): number => {
119
212
  * zoom out as far as its own rect — clamping there rather than inventing
120
213
  * pixels. The median (not the max) is what keeps that clamping rare and the
121
214
  * upscale inside the quality gate.
215
+ *
216
+ * All of that applies ONLY to segments that are essentially just a face
217
+ * (`segmentIsFaceOnly`). Anything else — a screen share, a face-and-screen
218
+ * mix, a PiP — keeps its whole rect, centered and clipped to the shared
219
+ * aspect: the 2026-08-16 incident chased a 0.119 PiP and cropped away the
220
+ * screen content that was the actual subject.
122
221
  */
123
222
  export function planNormalization(
124
223
  timeline: readonly ContentRectSegment[],
@@ -133,10 +232,22 @@ export function planNormalization(
133
232
  segments: [],
134
233
  faceFracOfCanvas: [],
135
234
  coverUpscale: Infinity,
235
+ // Nothing was planned, so nothing was discarded; `ok: false` is the
236
+ // refusal signal, not this number.
237
+ areaDiscardWeighted: 0,
238
+ subject: [],
239
+ bias: [],
136
240
  ok: false,
137
241
  };
138
242
  }
139
243
 
244
+ // Face-anchored sizing and placement ONLY where the frame is essentially
245
+ // just a face (user decision, 2026-08-16). Everything else is a "screen"
246
+ // subject: its window is its own rect, centered and clipped to the shared
247
+ // aspect — the incident's PiP (sizeFrac 0.119) must never drag the crop
248
+ // to the bottom-right corner of a screen recording again.
249
+ const faceOnly = timeline.map((_, i) => segmentIsFaceOnly(faces[i] ?? null));
250
+
140
251
  /**
141
252
  * The LARGEST face fraction in each segment, not the median. A window sized
142
253
  * on the median is correct only at the median moment: the author's clip
@@ -144,39 +255,60 @@ export function planNormalization(
144
255
  * the frame edge whenever they leaned in — which is precisely the frame they
145
256
  * flagged. Sizing on the maximum makes the tightest moment the safe one and
146
257
  * every other moment merely roomier.
258
+ *
259
+ * Only FACE-ONLY segments contribute a measurement: a PiP-sized face must
260
+ * not drag the target down for the real talking heads (2026-08-16 incident).
147
261
  */
148
- const measured = timeline.map((_, i) => faces[i]?.sizeFracMax ?? faces[i]?.sizeFrac ?? null);
262
+ const measured = timeline.map((_, i) =>
263
+ faceOnly[i] ? (faces[i]!.sizeFracMax ?? faces[i]!.sizeFrac) : null,
264
+ );
149
265
  const known = measured.filter((v): v is number => v !== null);
150
266
 
151
267
  // ---- Window heights ------------------------------------------------------
152
- // Without a single measurement there is no subject to hold constant, so the
153
- // rect-shaped fallback stands: the tightest field of view, uniformly.
268
+ // Without a single face-only segment there is no subject to hold constant,
269
+ // and the whole plan degrades to rect-shaped windows: the tightest field of
270
+ // view, uniformly, with nothing anchored on a face.
154
271
  const target = known.length > 0 ? Math.min(median(known), MAX_FACE_FRACTION) : null;
155
- const rectShapedHeights = (): number[] => {
156
- const canvasRect = boxed.reduce((a, b) => (b.rect.h < a.rect.h ? b : a)).rect;
157
- const a = canvasRect.w / canvasRect.h;
158
- return timeline.map((s) => even(Math.min(s.rect.w, s.rect.h * a) / a));
159
- };
160
- const windowHeights =
161
- target === null
162
- ? rectShapedHeights()
163
- : timeline.map((s, i) => {
164
- // An unmeasured segment inherits the median fraction of the segments
165
- // framed like it (same rect height), so it is sized in ITS OWN class
166
- // rather than averaged across two different shots.
167
- const sameClass = timeline.flatMap((o, j) =>
168
- measured[j] !== null && Math.abs(o.rect.h - s.rect.h) <= 2 ? [measured[j]!] : [],
169
- );
170
- const frac = measured[i] ?? (sameClass.length > 0 ? median(sameClass) : median(known));
171
- return even(clamp((frac * s.rect.h) / target, 16, s.rect.h));
172
- });
272
+ const canvasRect = boxed.reduce((a, b) => (b.rect.h < a.rect.h ? b : a)).rect;
273
+ const canvasRectAspect = canvasRect.w / canvasRect.h;
274
+ /** The tallest window of the tightest rect's shape that fits in `s.rect`. */
275
+ const rectShapedHeight = (s: ContentRectSegment): number =>
276
+ even(Math.min(s.rect.w, s.rect.h * canvasRectAspect) / canvasRectAspect);
277
+ const windowHeights = timeline.map((s, i) =>
278
+ target === null || !faceOnly[i]
279
+ ? rectShapedHeight(s)
280
+ : even(clamp((measured[i]! * s.rect.h) / target, 16, s.rect.h)),
281
+ );
173
282
 
174
283
  // ---- Canvas --------------------------------------------------------------
175
- // The widest aspect every window can actually hold. Wider than the output's
176
- // own aspect leaves the stage some horizontal freedom for the face bias;
177
- // narrower simply means the output crops height, which cover already does.
284
+ // The widest aspect every MATERIAL window can actually hold. Wider than the
285
+ // output's own aspect leaves the stage some horizontal freedom for the face
286
+ // bias; narrower simply means the output crops height, which cover already
287
+ // does.
288
+ //
289
+ // Material = the segment's framing class (same rect within 2px) totals at
290
+ // least
291
+ // MIN_FRAMING_CLASS_FRAC of the runtime — the constant content-rect's
292
+ // materiality filter uses, belt-and-braces with it. A sliver class still gets a
293
+ // window — the rect clamp below bounds it — it just gets no vote here: one
294
+ // 15.4s segment (1.1% of a 1435s take, rect 2848x2234) set canvas aspect
295
+ // 1.2748 for the whole video and baked away 28% of source width
296
+ // (2026-08-16 incident).
297
+ const totalDur = timeline.reduce((a, s) => a + Math.max(0, s.endSec - s.startSec), 0);
298
+ const classDur = timeline.map((s) =>
299
+ timeline.reduce(
300
+ (a, o) =>
301
+ Math.abs(o.rect.w - s.rect.w) <= 2 && Math.abs(o.rect.h - s.rect.h) <= 2
302
+ ? a + Math.max(0, o.endSec - o.startSec)
303
+ : a,
304
+ 0,
305
+ ),
306
+ );
307
+ const material = classDur.map((d) => totalDur > 0 && d / totalDur >= MIN_FRAMING_CLASS_FRAC);
308
+ // If every class is a sliver there is no majority to defer to — all vote.
309
+ const votes = material.some(Boolean) ? material : material.map(() => true);
178
310
  const aspect = timeline.reduce(
179
- (a, s, i) => Math.min(a, s.rect.w / windowHeights[i]!),
311
+ (a, s, i) => (votes[i] ? Math.min(a, s.rect.w / windowHeights[i]!) : a),
180
312
  Number.POSITIVE_INFINITY,
181
313
  );
182
314
  // The smallest window, so baking never upscales — the tightest segment sets
@@ -185,14 +317,16 @@ export function planNormalization(
185
317
  const canvas = { width: even(canvasHeight * aspect), height: canvasHeight };
186
318
 
187
319
  // ---- Face placement inside the window ------------------------------------
188
- // Taken from the segments whose window IS their rect: their framing is the
189
- // author's own and survives untouched, so it is the one to reproduce.
320
+ // Taken from the FACE-ONLY segments whose window IS their rect: their
321
+ // framing is the author's own and survives untouched, so it is the one to
322
+ // reproduce. Screen segments get no say — the incident's PiP at 0.88/0.76
323
+ // would anchor every window bottom-right.
190
324
  let wx = 0;
191
325
  let wy = 0;
192
326
  let weight = 0;
193
327
  timeline.forEach((seg, i) => {
194
328
  const f = faces[i];
195
- if (!f || windowHeights[i]! < seg.rect.h - 2) return;
329
+ if (!f || !faceOnly[i] || windowHeights[i]! < seg.rect.h - 2) return;
196
330
  const dur = Math.max(1e-6, seg.endSec - seg.startSec);
197
331
  wx += f.centerXFrac * dur;
198
332
  wy += f.centerYFrac * dur;
@@ -205,31 +339,54 @@ export function planNormalization(
205
339
  const r = seg.rect;
206
340
  const wH = windowHeights[i]!;
207
341
  const wW = even(Math.min(r.w, wH * aspect));
208
- const f = faces[i];
209
- // Face position in source px; the rect centre when nothing was measured.
210
- const faceX = r.x + (f ? f.centerXFrac : 0.5) * r.w;
211
- const faceY = r.y + (f ? f.centerYFrac : 0.5) * r.h;
342
+
343
+ // Screen subject: no face math at all. The window is the segment's own
344
+ // rect clipped to the shared aspect, centered on BOTH axes — the picture
345
+ // is the subject, and a centered clip is the only placement that does not
346
+ // pick a corner of it to sacrifice (2026-08-16 incident).
347
+ if (!faceOnly[i]) {
348
+ return {
349
+ startSec: seg.startSec,
350
+ endSec: seg.endSec,
351
+ window: {
352
+ x: even(r.x + (r.w - wW) / 2),
353
+ y: even(r.y + (r.h - wH) / 2),
354
+ w: wW,
355
+ h: wH,
356
+ },
357
+ };
358
+ }
359
+
360
+ const f = faces[i]!;
361
+ // Face position in source px.
362
+ const faceX = r.x + f.centerXFrac * r.w;
363
+ const faceY = r.y + f.centerYFrac * r.h;
212
364
  const x = even(clamp(faceX - targetX * wW, r.x, r.x + r.w - wW));
213
365
 
214
366
  let y = clamp(faceY - targetY * wH, r.y, r.y + r.h - wH);
215
367
  // Then slide — never resize — so the whole HEAD is inside the window at
216
368
  // the segment's largest face, since the aesthetic anchor above is about
217
- // where the face sits, and this is about not amputating it. Bounded by
218
- // the rect: if the head genuinely runs past the source's own edge there
219
- // is nothing to slide toward, and the clamp leaves it where it was.
220
- if (f) {
221
- const maxFace = (f.sizeFracMax ?? f.sizeFrac) * r.h;
222
- const headTop = faceY - HEAD_ABOVE * maxFace;
223
- const headBottom = faceY + HEAD_BELOW * maxFace;
224
- if (headBottom - headTop <= wH) {
225
- y = clamp(y, headBottom - wH, headTop);
226
- } else {
227
- // Head taller than the window: centre it, so what is lost is shared
228
- // between crown and chin instead of taking the whole bite off one end.
229
- y = (headTop + headBottom) / 2 - wH / 2;
230
- }
231
- y = clamp(y, r.y, r.y + r.h - wH);
369
+ // where the face sits, and this is about not amputating it. The margin is
370
+ // the user's ~1% rule (2026-08-16): the entire head including hair stays
371
+ // in frame with a hair of breathing room — a crown touching the edge
372
+ // reads as cropped. Bounded by the rect: if the head genuinely runs past
373
+ // the source's own edge there is nothing to slide toward, and the clamp
374
+ // leaves it where it was.
375
+ const maxFace = (f.sizeFracMax ?? f.sizeFrac) * r.h;
376
+ const headTop = faceY - HEAD_ABOVE * maxFace;
377
+ const headBottom = faceY + HEAD_BELOW * maxFace;
378
+ const margin = HEAD_WINDOW_MARGIN * wH;
379
+ if (headBottom - headTop + 2 * margin <= wH) {
380
+ y = clamp(y, headBottom + margin - wH, headTop - margin);
381
+ } else if (headBottom - headTop <= wH) {
382
+ // Head fits but its margin does not: keep the head, split the shortfall.
383
+ y = clamp(y, headBottom - wH, headTop);
384
+ } else {
385
+ // Head taller than the window: centre it, so what is lost is shared
386
+ // between crown and chin instead of taking the whole bite off one end.
387
+ y = (headTop + headBottom) / 2 - wH / 2;
232
388
  }
389
+ y = clamp(y, r.y, r.y + r.h - wH);
233
390
  return {
234
391
  startSec: seg.startSec,
235
392
  endSec: seg.endSec,
@@ -237,6 +394,23 @@ export function planNormalization(
237
394
  };
238
395
  });
239
396
 
397
+ const subject = timeline.map((_, i): "face" | "screen" => (faceOnly[i] ? "face" : "screen"));
398
+ // Where the subject sits inside its FINAL window (post-clamp, post-even):
399
+ // the render-time cover crop uses this to keep the head where the plan put
400
+ // it. Clamped to 0..1 because a rect-bounded window can leave the face
401
+ // centre outside it in the degenerate edge cases the clamps above allow.
402
+ const bias = timeline.map((seg, i) => {
403
+ if (!faceOnly[i]) return { x: 0.5, y: 0.5 };
404
+ const f = faces[i]!;
405
+ const w = segments[i]!.window;
406
+ const faceX = seg.rect.x + f.centerXFrac * seg.rect.w;
407
+ const faceY = seg.rect.y + f.centerYFrac * seg.rect.h;
408
+ return {
409
+ x: clamp(w.w > 0 ? (faceX - w.x) / w.w : 0.5, 0, 1),
410
+ y: clamp(w.h > 0 ? (faceY - w.y) / w.h : 0.5, 0, 1),
411
+ };
412
+ });
413
+
240
414
  // Cover the output with the canvas: for a canvas wider than the output's
241
415
  // aspect the height binds, otherwise the width does.
242
416
  const coverUpscale =
@@ -246,19 +420,54 @@ export function planNormalization(
246
420
 
247
421
  // What each segment actually achieved once its window is scaled to the
248
422
  // canvas — measured from the plan, never assumed to equal the target: a
249
- // segment clamped at its own rect lands wherever its rect put it.
423
+ // segment clamped at its own rect lands wherever its rect put it. A screen
424
+ // segment reports 0: it has no face SUBJECT, and downstream framing advice
425
+ // (assessCueFraming, framing.ts) rightly skips zeros rather than warning
426
+ // about the head of a PiP nobody is framing on.
250
427
  const faceFracOfCanvas = timeline.map((seg, i) => {
251
428
  const frac = measured[i];
252
- if (frac === null || frac === undefined) return target ?? 0;
429
+ if (frac === null || frac === undefined) return 0;
253
430
  return (frac * seg.rect.h) / segments[i]!.window.h;
254
431
  });
255
432
 
433
+ /** Picture area the window throws away from its segment's rect. */
434
+ const discardFrac = (i: number): number => {
435
+ const r = timeline[i]!.rect;
436
+ const w = segments[i]!.window;
437
+ return r.w * r.h > 0 ? 1 - (w.w * w.h) / (r.w * r.h) : 0;
438
+ };
439
+
440
+ // How much of each segment's picture the windows throw away, weighted by
441
+ // how long the viewer looks at it. coverUpscale 0.77 passed while 37% of
442
+ // the frame area was being thrown away — the gate measured softness, never
443
+ // loss (2026-08-16 incident).
444
+ const areaDiscardWeighted = timeline.reduce((acc, seg, i) => {
445
+ const dur = Math.max(0, seg.endSec - seg.startSec);
446
+ return acc + (totalDur > 0 ? dur / totalDur : 0) * discardFrac(i);
447
+ }, 0);
448
+
449
+ // Per-segment bound for SCREEN subjects: their window is meant to be
450
+ // (essentially) their whole rect, so a material screen segment losing more
451
+ // than MAX_SCREEN_AREA_DISCARD means the plan is cropping content nobody
452
+ // asked it to reframe — refuse and let render-time fit show it honestly.
453
+ // Face segments are exempt (a face crop discards area by design); sliver
454
+ // segments are exempt for the same reason they get no aspect vote.
455
+ const screenLossOk = timeline.every(
456
+ (_, i) => faceOnly[i] || !material[i] || discardFrac(i) <= MAX_SCREEN_AREA_DISCARD,
457
+ );
458
+
256
459
  return {
257
460
  canvas,
258
461
  segments,
259
462
  faceFracOfCanvas,
260
463
  coverUpscale,
261
- ok: coverUpscale <= MAX_NORMALIZE_UPSCALE,
464
+ areaDiscardWeighted,
465
+ subject,
466
+ bias,
467
+ ok:
468
+ coverUpscale <= MAX_NORMALIZE_UPSCALE &&
469
+ areaDiscardWeighted <= MAX_MEAN_AREA_DISCARD &&
470
+ screenLossOk,
262
471
  };
263
472
  }
264
473
 
@@ -355,66 +564,3 @@ export function assessCueFraming(
355
564
  }
356
565
  return out;
357
566
  }
358
-
359
- /**
360
- * The ffmpeg filter graph baking the plan: each segment trimmed, cropped to
361
- * its window, scaled to the canvas, and the pieces concatenated back into one
362
- * continuous stream. The segment boundaries partition the source exactly, so
363
- * the output timeline equals the input's and the untouched audio stays in
364
- * sync.
365
- */
366
- export function normalizationFilterGraph(plan: NormalizePlan): string {
367
- const parts = plan.segments.map((s, i) => {
368
- const w = s.window;
369
- return (
370
- `[0:v]trim=start=${s.startSec.toFixed(3)}:end=${s.endSec.toFixed(3)},` +
371
- `setpts=PTS-STARTPTS,crop=${w.w}:${w.h}:${w.x}:${w.y},` +
372
- // setsar=1 is load-bearing, not tidiness (R27 §125). Every segment is
373
- // scaled to the SAME canvas, but from a DIFFERENT crop, and ffmpeg
374
- // derives a sample aspect from that ratio: a 946x1682 crop yields SAR
375
- // 1683:1682 and a 932x1660 crop 1377:1376. `concat` requires identical
376
- // SAR across inputs and aborts the whole bake when they disagree, so a
377
- // take whose framing varies — exactly the take normalization exists
378
- // for — failed to render at all.
379
- `scale=${plan.canvas.width}:${plan.canvas.height},setsar=1[v${i}]`
380
- );
381
- });
382
- const inputs = plan.segments.map((_, i) => `[v${i}]`).join("");
383
- return `${parts.join(";")};${inputs}concat=n=${plan.segments.length}:v=1:a=0[v]`;
384
- }
385
-
386
- /**
387
- * Bake the normalized source. Encoded with the mezzanine's own settings
388
- * (dense keyframes) because it REPLACES the mezzanine — normalizing and then
389
- * re-encoding for seekability would be two generations of loss for nothing.
390
- */
391
- export async function bakeNormalizedSource(
392
- tools: { ffmpegPath: string },
393
- input: string,
394
- plan: NormalizePlan,
395
- outPath: string,
396
- ): Promise<void> {
397
- // Encode to a sibling temp path and rename only on success (R27 §125).
398
- // ffmpeg writes the container header as it goes, so a bake that dies
399
- // mid-graph leaves a file with no `moov` atom — and the cache upstream keys
400
- // on EXISTENCE, so that corpse is then reused as a valid normalized source
401
- // on every later run. The failure surfaces as "moov atom not found" from a
402
- // step that never ran, and deleting the workdir is the only way out. Rename
403
- // is atomic on a POSIX filesystem, so the cache can only ever see a file
404
- // ffmpeg finished writing.
405
- const partial = `${outPath}.partial.mp4`;
406
- try {
407
- await run(tools.ffmpegPath, [
408
- "-y", "-i", input,
409
- "-filter_complex", normalizationFilterGraph(plan),
410
- "-map", "[v]", "-map", "0:a?",
411
- "-c:v", "libx264", "-preset", "veryfast", "-crf", "18", "-g", "30", "-pix_fmt", "yuv420p",
412
- "-c:a", "aac", "-b:a", "192k",
413
- partial,
414
- ]);
415
- await rename(partial, outPath);
416
- } catch (err) {
417
- await rm(partial, { force: true });
418
- throw err;
419
- }
420
- }
@@ -25,6 +25,7 @@ import {
25
25
  export * from "./provider";
26
26
  export * from "./usage";
27
27
  export * from "./beats";
28
+ export * from "./youtube";
28
29
  export * from "./scene-props";
29
30
  export * from "./repair";
30
31
  export { AnthropicProvider, DEFAULT_CLAUDE_MODEL } from "./anthropic";
@@ -76,7 +76,11 @@ A correction must sound essentially identical to what was heard. If a span is no
76
76
 
77
77
  For each fix give the word-index span, the exact text you are replacing (\`heard\`), and the corrected text.`;
78
78
 
79
- export function buildRepairUserPrompt(transcript: Transcript, speaker?: string): string {
79
+ export function buildRepairUserPrompt(
80
+ transcript: Transcript,
81
+ speaker?: string,
82
+ dictionary?: readonly string[],
83
+ ): string {
80
84
  const words = transcript.words.map((w, i) => `[${i}]${w.text}`).join(" ");
81
85
  return (
82
86
  (speaker
@@ -89,6 +93,13 @@ export function buildRepairUserPrompt(transcript: Transcript, speaker?: string):
89
93
  `About the speaker (use this to recognise names the recognizer mangled, ` +
90
94
  `never to introduce facts): ${speaker}\n\n`
91
95
  : "") +
96
+ (dictionary && dictionary.length > 0
97
+ ? // The user's dictionary (F4, 2026-08-16): same failure class as the
98
+ // speaker hint — "Jason" for JSON is a lookup once the model knows
99
+ // the term, a guess otherwise.
100
+ `Vouched terms the speaker uses — prefer these spellings when the ` +
101
+ `audio matches: ${dictionary.join(", ")}\n\n`
102
+ : "") +
92
103
  `Word-indexed transcript (indices refer to THIS list):\n${words}\n\n` +
93
104
  `Report only spans that are clearly mishearings.`
94
105
  );
@@ -171,6 +182,13 @@ export interface ApplyRepairsOptions {
171
182
  * phonetic gate. Everything else about it is still checked.
172
183
  */
173
184
  speaker?: string;
185
+ /**
186
+ * The user's dictionary (F4, 2026-08-16) — vouched the same way the
187
+ * speaker hint is: terms the user typed themselves join the vouched set,
188
+ * so a fully-vouched correction ("Jason" → "JSON") may pass the phonetic
189
+ * gate. Every other guard still applies.
190
+ */
191
+ dictionary?: readonly string[];
174
192
  }
175
193
 
176
194
  export function applyRepairs(
@@ -180,18 +198,23 @@ export function applyRepairs(
180
198
  ): { transcript: Transcript; applied: AppliedRepair[] } {
181
199
  const maxIndex = transcript.words.length - 1;
182
200
  /**
183
- * Names the user vouched for via `--speaker`. A recognizer that turned
184
- * "Ahsan" into the initialism "SM" produces a correction no phonetic
185
- * measure will accept — and refusing it leaves the wrong name in the
186
- * captions, which is the failure the hint exists to prevent. So a correction
187
- * built ENTIRELY from words the user supplied is exempt from that one gate.
188
- * Every other guard still applies, and the exemption can only ever
189
- * substitute text the user typed themselves.
201
+ * Words the user vouched for — via `--speaker` and, since F4 (2026-08-16),
202
+ * via the dictionary. A recognizer that turned "Ahsan" into the initialism
203
+ * "SM" produces a correction no phonetic measure will accept — and refusing
204
+ * it leaves the wrong name in the captions, which is the failure the hint
205
+ * exists to prevent. So a correction built ENTIRELY from words the user
206
+ * supplied is exempt from that one gate. Every other guard still applies,
207
+ * and the exemption can only ever substitute text the user typed themselves.
190
208
  */
191
- const speakerWords = new Set(norm(opts.speaker ?? "").split(" ").filter(Boolean));
192
- const speakerVouched = (correction: string): boolean => {
209
+ const vouchedWords = new Set(
210
+ [
211
+ ...norm(opts.speaker ?? "").split(" "),
212
+ ...(opts.dictionary ?? []).flatMap((term) => norm(term).split(" ")),
213
+ ].filter(Boolean),
214
+ );
215
+ const userVouched = (correction: string): boolean => {
193
216
  const tokens = norm(correction).split(" ").filter(Boolean);
194
- return tokens.length > 0 && tokens.every((t) => speakerWords.has(t));
217
+ return tokens.length > 0 && tokens.every((t) => vouchedWords.has(t));
195
218
  };
196
219
  const results: AppliedRepair[] = [];
197
220
  const accepted: Array<{ startWord: number; endWord: number; tokens: string[] }> = [];
@@ -285,7 +308,7 @@ export function applyRepairs(
285
308
  record(`"${r.correction}" is too different in length from "${actual}"`);
286
309
  continue;
287
310
  }
288
- if (!soundsSimilar(actual, r.correction) && !speakerVouched(r.correction)) {
311
+ if (!soundsSimilar(actual, r.correction) && !userVouched(r.correction)) {
289
312
  // The gate that keeps this a repair pass and not a rewrite pass.
290
313
  record(`"${r.correction}" does not sound like "${actual}" — rewrite, not a repair`);
291
314
  continue;
@@ -340,7 +363,7 @@ export async function repairTranscript(
340
363
  try {
341
364
  const result = await provider.complete({
342
365
  system: REPAIR_SYSTEM,
343
- user: buildRepairUserPrompt(transcript, opts.speaker),
366
+ user: buildRepairUserPrompt(transcript, opts.speaker, opts.dictionary),
344
367
  schema: TranscriptRepairSchema,
345
368
  schemaName: "transcript_repair",
346
369
  // EDITORIAL on purpose, despite looking mechanical. Measured on the real