@ossclip/core 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +27 -0
- package/README.md +20 -0
- package/package.json +29 -0
- package/src/analyze.ts +299 -0
- package/src/assemble.ts +124 -0
- package/src/browser.ts +24 -0
- package/src/captions.ts +92 -0
- package/src/clip.ts +306 -0
- package/src/config.ts +66 -0
- package/src/content-rect-detect.ts +162 -0
- package/src/content-rect.ts +324 -0
- package/src/cover.ts +216 -0
- package/src/cta.ts +68 -0
- package/src/cutlist.ts +170 -0
- package/src/exec.ts +36 -0
- package/src/face.ts +519 -0
- package/src/fill.ts +110 -0
- package/src/framing.ts +277 -0
- package/src/grounding.ts +130 -0
- package/src/index.ts +27 -0
- package/src/ingest.ts +83 -0
- package/src/normalize.ts +397 -0
- package/src/overrides.ts +509 -0
- package/src/phonetics.ts +129 -0
- package/src/producer/anthropic.ts +73 -0
- package/src/producer/beats.ts +330 -0
- package/src/producer/claude-cli.ts +150 -0
- package/src/producer/gemini.ts +197 -0
- package/src/producer/index.ts +217 -0
- package/src/producer/mock.ts +101 -0
- package/src/producer/provider.ts +42 -0
- package/src/producer/repair.ts +474 -0
- package/src/producer/scene-props.ts +212 -0
- package/src/producer/tiered.ts +56 -0
- package/src/producer/usage.ts +426 -0
- package/src/report.ts +36 -0
- package/src/scene-registry.ts +246 -0
- package/src/scene-schema.ts +203 -0
- package/src/schema.ts +177 -0
- package/src/source-text.ts +348 -0
- package/src/timemap.ts +115 -0
- package/src/transcribe.ts +67 -0
- package/src/zoom.ts +154 -0
package/src/face.ts
ADDED
|
@@ -0,0 +1,519 @@
|
|
|
1
|
+
import { readFile, unlink, writeFile } from "node:fs/promises";
|
|
2
|
+
import { existsSync } from "node:fs";
|
|
3
|
+
import { join } from "node:path";
|
|
4
|
+
import { run } from "./exec";
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Static face measurement (FINDINGS §13) — the v0 shortcut of BRAINSTORM
|
|
8
|
+
* §4.3's reframing note. A constant crop bias trades cutting the forehead
|
|
9
|
+
* for cutting the mouth depending on how the speaker framed themselves, so
|
|
10
|
+
* the system measures where the face IS: sample frames, detect, take the
|
|
11
|
+
* median box, cache it in the workdir (it is a property of the source, not
|
|
12
|
+
* of a render). Full per-frame tracking stays Phase 4.
|
|
13
|
+
*
|
|
14
|
+
* Detection is the pico cascade (Nenad Markus), ported from picojs — a pure
|
|
15
|
+
* JS decision-tree detector, MIT licensed, no native deps:
|
|
16
|
+
* https://github.com/nenadmarkus/picojs. The pretrained `facefinder`
|
|
17
|
+
* cascade is vendored in assets/ (also MIT, from github.com/nenadmarkus/pico).
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
/** Median face box, as fractions of the SOURCE frame. */
|
|
21
|
+
export interface FaceBox {
|
|
22
|
+
/** Horizontal center, 0..1 of source width. */
|
|
23
|
+
centerXFrac: number;
|
|
24
|
+
/** Vertical center, 0..1 of source height. */
|
|
25
|
+
centerYFrac: number;
|
|
26
|
+
/** Face size (pico's square detection edge), as a fraction of source height. */
|
|
27
|
+
sizeFrac: number;
|
|
28
|
+
framesSampled: number;
|
|
29
|
+
framesDetected: number;
|
|
30
|
+
/** Frames only the tilt sweep found (PLAN Task 8) — absent when none. */
|
|
31
|
+
framesRotated?: number;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
// ---- pico runtime (ported from picojs, MIT) --------------------------------
|
|
35
|
+
|
|
36
|
+
type ClassifyFn = (r: number, c: number, s: number, pixels: Uint8Array, ldim: number) => number;
|
|
37
|
+
|
|
38
|
+
function unpackCascade(bytes: Uint8Array): ClassifyFn {
|
|
39
|
+
const dview = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
40
|
+
// Skip version + training metadata (8 bytes).
|
|
41
|
+
let p = 8;
|
|
42
|
+
const tdepth = dview.getInt32(p, true);
|
|
43
|
+
p += 4;
|
|
44
|
+
const ntrees = dview.getInt32(p, true);
|
|
45
|
+
p += 4;
|
|
46
|
+
const pow2tdepth = 2 ** tdepth;
|
|
47
|
+
const tcodesLs: number[] = [];
|
|
48
|
+
const tpredsLs: number[] = [];
|
|
49
|
+
const threshLs: number[] = [];
|
|
50
|
+
for (let t = 0; t < ntrees; t++) {
|
|
51
|
+
tcodesLs.push(0, 0, 0, 0);
|
|
52
|
+
for (let i = 0; i < 4 * pow2tdepth - 4; i++) tcodesLs.push(dview.getInt8(p + i));
|
|
53
|
+
p += 4 * pow2tdepth - 4;
|
|
54
|
+
for (let i = 0; i < pow2tdepth; i++) {
|
|
55
|
+
tpredsLs.push(dview.getFloat32(p, true));
|
|
56
|
+
p += 4;
|
|
57
|
+
}
|
|
58
|
+
threshLs.push(dview.getFloat32(p, true));
|
|
59
|
+
p += 4;
|
|
60
|
+
}
|
|
61
|
+
const tcodes = new Int8Array(tcodesLs);
|
|
62
|
+
const tpreds = new Float32Array(tpredsLs);
|
|
63
|
+
const thresh = new Float32Array(threshLs);
|
|
64
|
+
|
|
65
|
+
return (r, c, s, pixels, ldim) => {
|
|
66
|
+
r *= 256;
|
|
67
|
+
c *= 256;
|
|
68
|
+
let root = 0;
|
|
69
|
+
let o = 0;
|
|
70
|
+
for (let i = 0; i < ntrees; i++) {
|
|
71
|
+
let idx = 1;
|
|
72
|
+
for (let j = 0; j < tdepth; j++) {
|
|
73
|
+
// '>> 8' is the fixed-point division pico uses for speed.
|
|
74
|
+
const p1 =
|
|
75
|
+
pixels[((r + tcodes[root + 4 * idx + 0]! * s) >> 8) * ldim + ((c + tcodes[root + 4 * idx + 1]! * s) >> 8)]!;
|
|
76
|
+
const p2 =
|
|
77
|
+
pixels[((r + tcodes[root + 4 * idx + 2]! * s) >> 8) * ldim + ((c + tcodes[root + 4 * idx + 3]! * s) >> 8)]!;
|
|
78
|
+
idx = 2 * idx + (p1 <= p2 ? 1 : 0);
|
|
79
|
+
}
|
|
80
|
+
o += tpreds[pow2tdepth * i + idx - pow2tdepth]!;
|
|
81
|
+
if (o <= thresh[i]!) return -1;
|
|
82
|
+
root += 4 * pow2tdepth;
|
|
83
|
+
}
|
|
84
|
+
return o - thresh[ntrees - 1]!;
|
|
85
|
+
};
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** [row, col, scale, score] */
|
|
89
|
+
type Detection = [number, number, number, number];
|
|
90
|
+
|
|
91
|
+
function runCascade(
|
|
92
|
+
pixels: Uint8Array,
|
|
93
|
+
nrows: number,
|
|
94
|
+
ncols: number,
|
|
95
|
+
classify: ClassifyFn,
|
|
96
|
+
params: { shiftfactor: number; minsize: number; maxsize: number; scalefactor: number },
|
|
97
|
+
): Detection[] {
|
|
98
|
+
const detections: Detection[] = [];
|
|
99
|
+
let scale = params.minsize;
|
|
100
|
+
while (scale <= params.maxsize) {
|
|
101
|
+
const step = Math.max(params.shiftfactor * scale, 1) | 0;
|
|
102
|
+
const offset = (scale / 2 + 1) | 0;
|
|
103
|
+
for (let r = offset; r <= nrows - offset; r += step) {
|
|
104
|
+
for (let c = offset; c <= ncols - offset; c += step) {
|
|
105
|
+
const q = classify(r, c, scale, pixels, ncols);
|
|
106
|
+
if (q > 0) detections.push([r, c, scale, q]);
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
scale *= params.scalefactor;
|
|
110
|
+
}
|
|
111
|
+
return detections;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
function clusterDetections(dets: Detection[], iouThreshold: number): Detection[] {
|
|
115
|
+
dets = [...dets].sort((a, b) => b[3] - a[3]);
|
|
116
|
+
const iou = (d1: Detection, d2: Detection): number => {
|
|
117
|
+
const [r1, c1, s1] = d1;
|
|
118
|
+
const [r2, c2, s2] = d2;
|
|
119
|
+
const overR = Math.max(0, Math.min(r1 + s1 / 2, r2 + s2 / 2) - Math.max(r1 - s1 / 2, r2 - s2 / 2));
|
|
120
|
+
const overC = Math.max(0, Math.min(c1 + s1 / 2, c2 + s2 / 2) - Math.max(c1 - s1 / 2, c2 - s2 / 2));
|
|
121
|
+
return (overR * overC) / (s1 * s1 + s2 * s2 - overR * overC);
|
|
122
|
+
};
|
|
123
|
+
const assigned = new Array<boolean>(dets.length).fill(false);
|
|
124
|
+
const clusters: Detection[] = [];
|
|
125
|
+
for (let i = 0; i < dets.length; i++) {
|
|
126
|
+
if (assigned[i]) continue;
|
|
127
|
+
let r = 0,
|
|
128
|
+
c = 0,
|
|
129
|
+
s = 0,
|
|
130
|
+
q = 0,
|
|
131
|
+
n = 0;
|
|
132
|
+
for (let j = i; j < dets.length; j++) {
|
|
133
|
+
if (iou(dets[i]!, dets[j]!) > iouThreshold) {
|
|
134
|
+
assigned[j] = true;
|
|
135
|
+
r += dets[j]![0];
|
|
136
|
+
c += dets[j]![1];
|
|
137
|
+
s += dets[j]![2];
|
|
138
|
+
q += dets[j]![3];
|
|
139
|
+
n++;
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
clusters.push([r / n, c / n, s / n, q]);
|
|
143
|
+
}
|
|
144
|
+
return clusters;
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
// ---- rotation sweep (PLAN Task 8) ------------------------------------------
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Angles tried, in order, when the upright pass finds nothing. The cascade is
|
|
151
|
+
* frontal AND upright: a head tilted past ~±20° falls outside what it was
|
|
152
|
+
* trained on, and one real clip lost all nine samples that way. Rotating the
|
|
153
|
+
* FRAME back to upright (cheap nearest-neighbour remap at detection size)
|
|
154
|
+
* recovers in-plane tilt without any new model asset.
|
|
155
|
+
*
|
|
156
|
+
* What this does NOT recover, stated plainly: a true profile — the head
|
|
157
|
+
* TURNED sideways rather than tilted — is out-of-plane and no image rotation
|
|
158
|
+
* makes it frontal. If the loud-miss log still fires on such footage, the
|
|
159
|
+
* next step is a profile cascade, not more angles here.
|
|
160
|
+
*/
|
|
161
|
+
export const ROTATION_SWEEP_DEG = [-20, 20, -40, 40];
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* The frame rotated by `deg` about its centre; out-of-frame samples are 0.
|
|
165
|
+
* Nearest-neighbour is plenty — the cascade reads coarse luminance relations,
|
|
166
|
+
* not edges — and keeps this allocation-cheap at detection size.
|
|
167
|
+
*/
|
|
168
|
+
export function rotateGray(pixels: Uint8Array, w: number, h: number, deg: number): Uint8Array {
|
|
169
|
+
const rad = (deg * Math.PI) / 180;
|
|
170
|
+
const cos = Math.cos(rad);
|
|
171
|
+
const sin = Math.sin(rad);
|
|
172
|
+
const cx = (w - 1) / 2;
|
|
173
|
+
const cy = (h - 1) / 2;
|
|
174
|
+
const out = new Uint8Array(w * h);
|
|
175
|
+
for (let y = 0; y < h; y++) {
|
|
176
|
+
for (let x = 0; x < w; x++) {
|
|
177
|
+
// Sample the source at the inverse rotation of this output position.
|
|
178
|
+
const dx = x - cx;
|
|
179
|
+
const dy = y - cy;
|
|
180
|
+
const sx = Math.round(cx + cos * dx + sin * dy);
|
|
181
|
+
const sy = Math.round(cy - sin * dx + cos * dy);
|
|
182
|
+
if (sx >= 0 && sx < w && sy >= 0 && sy < h) out[y * w + x] = pixels[sy * w + sx]!;
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
return out;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Map a detection centre found in the ROTATED frame back to the original.
|
|
190
|
+
* Exactly the sampling transform `rotateGray` applies — a rotated-frame
|
|
191
|
+
* position corresponds to the source pixel that was sampled into it.
|
|
192
|
+
*/
|
|
193
|
+
export function rotatePointBack(
|
|
194
|
+
r: number,
|
|
195
|
+
c: number,
|
|
196
|
+
w: number,
|
|
197
|
+
h: number,
|
|
198
|
+
deg: number,
|
|
199
|
+
): { r: number; c: number } {
|
|
200
|
+
const rad = (deg * Math.PI) / 180;
|
|
201
|
+
const cos = Math.cos(rad);
|
|
202
|
+
const sin = Math.sin(rad);
|
|
203
|
+
const cx = (w - 1) / 2;
|
|
204
|
+
const cy = (h - 1) / 2;
|
|
205
|
+
const dx = c - cx;
|
|
206
|
+
const dy = r - cy;
|
|
207
|
+
return { c: cx + cos * dx + sin * dy, r: cy - sin * dx + cos * dy };
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
interface CascadeParams {
|
|
211
|
+
shiftfactor: number;
|
|
212
|
+
minsize: number;
|
|
213
|
+
maxsize: number;
|
|
214
|
+
scalefactor: number;
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
/** Best clustered detection above the score floor, or null. */
|
|
218
|
+
function bestDetection(
|
|
219
|
+
pixels: Uint8Array,
|
|
220
|
+
h: number,
|
|
221
|
+
w: number,
|
|
222
|
+
classify: ClassifyFn,
|
|
223
|
+
params: CascadeParams,
|
|
224
|
+
): Detection | null {
|
|
225
|
+
const dets = clusterDetections(runCascade(pixels, h, w, classify, params), 0.2).filter(
|
|
226
|
+
(d) => d[3] >= MIN_CLUSTER_SCORE,
|
|
227
|
+
);
|
|
228
|
+
if (dets.length === 0) return null;
|
|
229
|
+
return dets.reduce((a, b) => (b[3] > a[3] ? b : a));
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/**
|
|
233
|
+
* Upright pass first; on a miss, the rotation sweep. The returned detection is
|
|
234
|
+
* in ORIGINAL frame coordinates whichever pass found it (size is preserved —
|
|
235
|
+
* in-plane rotation does not change scale).
|
|
236
|
+
*/
|
|
237
|
+
export function bestDetectionWithSweep(
|
|
238
|
+
pixels: Uint8Array,
|
|
239
|
+
h: number,
|
|
240
|
+
w: number,
|
|
241
|
+
classify: ClassifyFn,
|
|
242
|
+
params: CascadeParams,
|
|
243
|
+
sweepDeg: readonly number[] = ROTATION_SWEEP_DEG,
|
|
244
|
+
): { det: Detection; angleDeg: number } | null {
|
|
245
|
+
const upright = bestDetection(pixels, h, w, classify, params);
|
|
246
|
+
if (upright) return { det: upright, angleDeg: 0 };
|
|
247
|
+
for (const deg of sweepDeg) {
|
|
248
|
+
const hit = bestDetection(rotateGray(pixels, w, h, deg), h, w, classify, params);
|
|
249
|
+
if (!hit) continue;
|
|
250
|
+
const { r, c } = rotatePointBack(hit[0], hit[1], w, h, deg);
|
|
251
|
+
return { det: [r, c, hit[2], hit[3]], angleDeg: deg };
|
|
252
|
+
}
|
|
253
|
+
return null;
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
// ---- frame sampling + measurement ------------------------------------------
|
|
257
|
+
|
|
258
|
+
/** Detection frame width; the HEIGHT follows the (cropped) source's aspect —
|
|
259
|
+
* the cascade is trained on undistorted faces, so no squeeze into a fixed
|
|
260
|
+
* box. Fractions are of the analyzed frame either way. */
|
|
261
|
+
const DET_W = 360;
|
|
262
|
+
/** Clustered-score floor for a believable face. picojs demos use ~50, but
|
|
263
|
+
* that is summed over a 5-frame memory; a SINGLE frame at this resolution
|
|
264
|
+
* clears ~5-10 on a real face. Robustness against a lucky wall-poster hit
|
|
265
|
+
* comes from requiring detections in a majority of sampled frames, not from
|
|
266
|
+
* this per-frame floor. */
|
|
267
|
+
const MIN_CLUSTER_SCORE = 5;
|
|
268
|
+
|
|
269
|
+
/**
|
|
270
|
+
* A reusable detector over raw grayscale frames — the cascade is unpacked
|
|
271
|
+
* once and the closure reused, so callers that score many frames (cover
|
|
272
|
+
* selection, source-text scanning) don't pay for it per frame.
|
|
273
|
+
*/
|
|
274
|
+
export async function createFaceDetector(): Promise<
|
|
275
|
+
(pixels: Uint8Array, width: number, height: number) => Detection | null
|
|
276
|
+
> {
|
|
277
|
+
const cascadeBytes = new Uint8Array(
|
|
278
|
+
await readFile(new URL("../assets/facefinder", import.meta.url)),
|
|
279
|
+
);
|
|
280
|
+
const classify = unpackCascade(cascadeBytes);
|
|
281
|
+
return (pixels, width, height) => {
|
|
282
|
+
if (pixels.length < width * height) return null;
|
|
283
|
+
const hit = bestDetectionWithSweep(pixels, height, width, classify, {
|
|
284
|
+
shiftfactor: 0.1,
|
|
285
|
+
minsize: 60,
|
|
286
|
+
maxsize: height,
|
|
287
|
+
scalefactor: 1.1,
|
|
288
|
+
});
|
|
289
|
+
return hit ? hit.det : null;
|
|
290
|
+
};
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
/** A face measured inside one time-and-crop window, in the WINDOW's fractions. */
|
|
294
|
+
export interface WindowFace {
|
|
295
|
+
centerXFrac: number;
|
|
296
|
+
centerYFrac: number;
|
|
297
|
+
/** Median face height — what the segment looks like most of the time. */
|
|
298
|
+
sizeFrac: number;
|
|
299
|
+
/**
|
|
300
|
+
* The LARGEST face seen in the window (p90, so one bad detection cannot set
|
|
301
|
+
* it). Framing must be sized against this, not the median: a speaker leans
|
|
302
|
+
* in and out, and on the author's clip the face ran 29%-48% of the frame
|
|
303
|
+
* inside a single 12s stretch. A window sized on the median put the head
|
|
304
|
+
* past the frame edge at the moment they leaned in.
|
|
305
|
+
*/
|
|
306
|
+
sizeFracMax: number;
|
|
307
|
+
framesDetected: number;
|
|
308
|
+
framesSampled: number;
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
/**
|
|
312
|
+
* The face per WINDOW — one measurement per (time range, crop) pair, each in
|
|
313
|
+
* that crop's own fractions (NORMALIZE plan, the old Task C5 gap).
|
|
314
|
+
*
|
|
315
|
+
* A mixed-framing source cannot use `measureFace`'s single median: some of its
|
|
316
|
+
* samples are fractions of a letterboxed strip and some of the full frame, and
|
|
317
|
+
* a median across two coordinate systems describes neither — that mispointed
|
|
318
|
+
* crop is exactly what put the eyes at the top of the frame on the author's
|
|
319
|
+
* clip. Normalization instead measures each segment inside its own rect and
|
|
320
|
+
* uses the result to place that segment's crop window.
|
|
321
|
+
*
|
|
322
|
+
* No cache: a handful of frames per window, and the caller's bake output is
|
|
323
|
+
* itself cached by a hash of the plan this feeds.
|
|
324
|
+
*/
|
|
325
|
+
export async function measureFaceInWindows(
|
|
326
|
+
tools: { ffmpegPath: string },
|
|
327
|
+
videoPath: string,
|
|
328
|
+
windows: ReadonlyArray<{ startSec: number; endSec: number; cropVf: string }>,
|
|
329
|
+
opts: { samplesPerWindow?: number; workDir?: string } = {},
|
|
330
|
+
): Promise<Array<WindowFace | null>> {
|
|
331
|
+
const cascadeBytes = new Uint8Array(
|
|
332
|
+
await readFile(new URL("../assets/facefinder", import.meta.url)),
|
|
333
|
+
);
|
|
334
|
+
const classify = unpackCascade(cascadeBytes);
|
|
335
|
+
const out: Array<WindowFace | null> = [];
|
|
336
|
+
|
|
337
|
+
for (const [wi, w] of windows.entries()) {
|
|
338
|
+
const dur = Math.max(0, w.endSec - w.startSec);
|
|
339
|
+
// Roughly one look per 1.5s: the window is sized against the biggest face
|
|
340
|
+
// in the stretch, so under-sampling a long one hides exactly the moment
|
|
341
|
+
// that matters. Bounded so a 60s segment stays cheap.
|
|
342
|
+
const samples = Math.min(
|
|
343
|
+
opts.samplesPerWindow ?? 12,
|
|
344
|
+
Math.max(3, Math.round(dur / 1.5)),
|
|
345
|
+
);
|
|
346
|
+
const centersX: number[] = [];
|
|
347
|
+
const centersY: number[] = [];
|
|
348
|
+
const sizes: number[] = [];
|
|
349
|
+
for (let i = 0; i < samples; i++) {
|
|
350
|
+
// Interior points only — a frame ON the boundary may already be the
|
|
351
|
+
// other framing, which is the confusion this function exists to avoid.
|
|
352
|
+
const t = w.startSec + (dur * (i + 1)) / (samples + 1);
|
|
353
|
+
const framePath = join(opts.workDir ?? ".", `segface-${wi}-${i}.gray`);
|
|
354
|
+
await run(tools.ffmpegPath, [
|
|
355
|
+
"-v", "error",
|
|
356
|
+
"-ss", t.toFixed(3),
|
|
357
|
+
"-i", videoPath,
|
|
358
|
+
"-frames:v", "1",
|
|
359
|
+
"-vf", `${w.cropVf ? `${w.cropVf},` : ""}scale=${DET_W}:-2`,
|
|
360
|
+
"-pix_fmt", "gray",
|
|
361
|
+
"-f", "rawvideo",
|
|
362
|
+
"-y", framePath,
|
|
363
|
+
]);
|
|
364
|
+
const pixels = new Uint8Array(await readFile(framePath));
|
|
365
|
+
await unlink(framePath).catch(() => {});
|
|
366
|
+
const detH = Math.floor(pixels.length / DET_W);
|
|
367
|
+
if (detH < 32) continue;
|
|
368
|
+
const hit = bestDetectionWithSweep(pixels, detH, DET_W, classify, {
|
|
369
|
+
shiftfactor: 0.1,
|
|
370
|
+
minsize: Math.max(24, Math.round(detH * 0.094)),
|
|
371
|
+
maxsize: detH,
|
|
372
|
+
scalefactor: 1.1,
|
|
373
|
+
});
|
|
374
|
+
if (!hit) continue;
|
|
375
|
+
centersY.push(hit.det[0] / detH);
|
|
376
|
+
centersX.push(hit.det[1] / DET_W);
|
|
377
|
+
sizes.push(hit.det[2] / detH);
|
|
378
|
+
}
|
|
379
|
+
const sorted = [...sizes].sort((a, b) => a - b);
|
|
380
|
+
out.push(
|
|
381
|
+
centersY.length >= 2
|
|
382
|
+
? {
|
|
383
|
+
centerXFrac: median(centersX),
|
|
384
|
+
centerYFrac: median(centersY),
|
|
385
|
+
sizeFrac: median(sizes),
|
|
386
|
+
sizeFracMax: sorted[Math.min(sorted.length - 1, Math.floor(sorted.length * 0.9))]!,
|
|
387
|
+
framesDetected: centersY.length,
|
|
388
|
+
framesSampled: samples,
|
|
389
|
+
}
|
|
390
|
+
: null,
|
|
391
|
+
);
|
|
392
|
+
}
|
|
393
|
+
return out;
|
|
394
|
+
}
|
|
395
|
+
|
|
396
|
+
export interface MeasureFaceOptions {
|
|
397
|
+
/** Frames to sample, spread across the middle of the take. */
|
|
398
|
+
samples?: number;
|
|
399
|
+
/** Directory for the cached measurement + temp frames (the workdir). */
|
|
400
|
+
cacheDir?: string;
|
|
401
|
+
/**
|
|
402
|
+
* ffmpeg filter trimming the source to its content rect (PLAN Task 7),
|
|
403
|
+
* prepended before the detection scale. Measuring inside the picture rather
|
|
404
|
+
* than the letterboxed canvas matters twice over: the returned fractions
|
|
405
|
+
* then describe the frame that actually renders, and the face is a large
|
|
406
|
+
* enough share of the searched area for the cascade's scale sweep to find.
|
|
407
|
+
*/
|
|
408
|
+
cropVf?: string;
|
|
409
|
+
/**
|
|
410
|
+
* Extra cache-validity key beyond `cropVf` — set to the measured FILE's
|
|
411
|
+
* identity when it is not the workdir's original source (the normalized
|
|
412
|
+
* bake), so a cache from one geometry is never served for another.
|
|
413
|
+
*/
|
|
414
|
+
cacheTag?: string;
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
function median(xs: number[]): number {
|
|
418
|
+
const s = [...xs].sort((a, b) => a - b);
|
|
419
|
+
const m = Math.floor(s.length / 2);
|
|
420
|
+
return s.length % 2 ? s[m]! : (s[m - 1]! + s[m]!) / 2;
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
/**
|
|
424
|
+
* Sample frames across the take, detect the face in each, return the median
|
|
425
|
+
* box — or null when the take genuinely has no findable face (screen
|
|
426
|
+
* recording, slides), in which case callers fall back to the default bias.
|
|
427
|
+
*/
|
|
428
|
+
export async function measureFace(
|
|
429
|
+
tools: { ffmpegPath: string },
|
|
430
|
+
videoPath: string,
|
|
431
|
+
durationSec: number,
|
|
432
|
+
opts: MeasureFaceOptions = {},
|
|
433
|
+
): Promise<FaceBox | null> {
|
|
434
|
+
const samples = opts.samples ?? 9;
|
|
435
|
+
const cachePath = opts.cacheDir ? join(opts.cacheDir, "face.json") : null;
|
|
436
|
+
if (cachePath && existsSync(cachePath)) {
|
|
437
|
+
const cached = JSON.parse(await readFile(cachePath, "utf8")) as {
|
|
438
|
+
face: FaceBox | null;
|
|
439
|
+
cropVf?: string;
|
|
440
|
+
cacheTag?: string;
|
|
441
|
+
};
|
|
442
|
+
// A measurement made against a different geometry (pre-Task-7 cache, a
|
|
443
|
+
// changed content rect, or a different baked file) describes a frame that
|
|
444
|
+
// no longer renders.
|
|
445
|
+
if (
|
|
446
|
+
(cached.cropVf ?? "") === (opts.cropVf ?? "") &&
|
|
447
|
+
(cached.cacheTag ?? "") === (opts.cacheTag ?? "")
|
|
448
|
+
) {
|
|
449
|
+
return cached.face;
|
|
450
|
+
}
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
const cascadeBytes = new Uint8Array(
|
|
454
|
+
await readFile(new URL("../assets/facefinder", import.meta.url)),
|
|
455
|
+
);
|
|
456
|
+
const classify = unpackCascade(cascadeBytes);
|
|
457
|
+
|
|
458
|
+
const centersX: number[] = [];
|
|
459
|
+
const centersY: number[] = [];
|
|
460
|
+
const sizes: number[] = [];
|
|
461
|
+
let rotated = 0;
|
|
462
|
+
for (let i = 0; i < samples; i++) {
|
|
463
|
+
// Middle 80% — intros/outros are where people lean off-frame.
|
|
464
|
+
const t = durationSec * (0.1 + (0.8 * i) / Math.max(1, samples - 1));
|
|
465
|
+
const framePath = join(opts.cacheDir ?? ".", `face-frame-${i}.gray`);
|
|
466
|
+
await run(tools.ffmpegPath, [
|
|
467
|
+
"-v", "error",
|
|
468
|
+
"-ss", t.toFixed(3),
|
|
469
|
+
"-i", videoPath,
|
|
470
|
+
"-frames:v", "1",
|
|
471
|
+
// Aspect follows the (cropped) source rather than a fixed 9:16 box: the
|
|
472
|
+
// cascade is trained on undistorted faces, and squeezing a landscape
|
|
473
|
+
// content rect into a portrait frame stretches every face past what it
|
|
474
|
+
// can match. -2 keeps the height even.
|
|
475
|
+
"-vf", `${opts.cropVf ? `${opts.cropVf},` : ""}scale=${DET_W}:-2`,
|
|
476
|
+
"-pix_fmt", "gray",
|
|
477
|
+
"-f", "rawvideo",
|
|
478
|
+
"-y", framePath,
|
|
479
|
+
]);
|
|
480
|
+
const pixels = new Uint8Array(await readFile(framePath));
|
|
481
|
+
await unlink(framePath).catch(() => {});
|
|
482
|
+
const detH = Math.floor(pixels.length / DET_W);
|
|
483
|
+
if (detH < 32) continue;
|
|
484
|
+
const hit = bestDetectionWithSweep(pixels, detH, DET_W, classify, {
|
|
485
|
+
shiftfactor: 0.1,
|
|
486
|
+
// The old fixed floor (60px of a 640-tall frame, ~9%) expressed as a
|
|
487
|
+
// ratio, so a shorter landscape frame still sweeps small enough.
|
|
488
|
+
minsize: Math.max(24, Math.round(detH * 0.094)),
|
|
489
|
+
maxsize: detH,
|
|
490
|
+
scalefactor: 1.1,
|
|
491
|
+
});
|
|
492
|
+
if (!hit) continue;
|
|
493
|
+
if (hit.angleDeg !== 0) rotated++;
|
|
494
|
+
centersY.push(hit.det[0] / detH);
|
|
495
|
+
centersX.push(hit.det[1] / DET_W);
|
|
496
|
+
sizes.push(hit.det[2] / detH);
|
|
497
|
+
}
|
|
498
|
+
|
|
499
|
+
// One lucky hit could be a poster in the background; demand a majority-ish.
|
|
500
|
+
const face: FaceBox | null =
|
|
501
|
+
centersY.length >= Math.min(3, samples)
|
|
502
|
+
? {
|
|
503
|
+
centerXFrac: median(centersX),
|
|
504
|
+
centerYFrac: median(centersY),
|
|
505
|
+
sizeFrac: median(sizes),
|
|
506
|
+
framesSampled: samples,
|
|
507
|
+
framesDetected: centersY.length,
|
|
508
|
+
framesRotated: rotated > 0 ? rotated : undefined,
|
|
509
|
+
}
|
|
510
|
+
: null;
|
|
511
|
+
|
|
512
|
+
if (cachePath) {
|
|
513
|
+
await writeFile(
|
|
514
|
+
cachePath,
|
|
515
|
+
JSON.stringify({ face, cropVf: opts.cropVf ?? "", cacheTag: opts.cacheTag ?? "" }, null, 2),
|
|
516
|
+
);
|
|
517
|
+
}
|
|
518
|
+
return face;
|
|
519
|
+
}
|
package/src/fill.ts
ADDED
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
import type { SceneCue } from "./scene-schema";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Fill the gaps between graphic cues with PLAIN cues, one per continuous take
|
|
5
|
+
* (PLAN 2026-07-30 Task A).
|
|
6
|
+
*
|
|
7
|
+
* Cues are deliberately sparse today — a gap renders as the implicit
|
|
8
|
+
* full-bleed talking head — which leaves most of the video unreachable by any
|
|
9
|
+
* per-scene control: `cue.video` framing only applies where a cue is active.
|
|
10
|
+
* Filling the timeline is what makes framing (Task B) work everywhere, and it
|
|
11
|
+
* is why a deleted scene's window (Task C) becomes an editable take instead of
|
|
12
|
+
* a hole.
|
|
13
|
+
*
|
|
14
|
+
* Each gap is SPLIT at every cut boundary strictly inside it, so a plain cue
|
|
15
|
+
* never straddles a cut: the zoom ramp (`zoom.ts`) and `EdlVideo`'s punch-in
|
|
16
|
+
* already key off those same boundaries, and a block that crossed one would
|
|
17
|
+
* pretend two takes are one shot.
|
|
18
|
+
*
|
|
19
|
+
* ID STABILITY, stated honestly: ids derive from (graphic cue windows, clip
|
|
20
|
+
* starts) — `take-<clipIndex>` for a clip's first plain piece, suffixed
|
|
21
|
+
* `take-<clipIndex>-<k>` only for the later pieces graphics split off. So
|
|
22
|
+
* deleting a graphic that had split a take MERGES two pieces back into one:
|
|
23
|
+
* the merged piece keeps the unsuffixed name (any override on it survives),
|
|
24
|
+
* and an override on the vanished suffix becomes an ORPHAN — an
|
|
25
|
+
* already-reported condition (`applyOverrides`), never a silent loss. The
|
|
26
|
+
* alternative, time-keyed ids, would instead drift on every re-cut.
|
|
27
|
+
*/
|
|
28
|
+
|
|
29
|
+
/** Pieces under this are dropped: the assembler's 0.05s breathing gaps must
|
|
30
|
+
* not become blocks, and the renderer's implicit full-bleed already covers
|
|
31
|
+
* them identically. */
|
|
32
|
+
export const MIN_PLAIN_SEC = 0.6;
|
|
33
|
+
|
|
34
|
+
export interface FillPlainOptions {
|
|
35
|
+
outputDurationSec: number;
|
|
36
|
+
/** Output-time starts of each kept span (`spans[].outIn`) — the cuts. */
|
|
37
|
+
clipStarts?: readonly number[];
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/** Clip starts cleaned the same way the zoom planner cleans them: in range,
|
|
41
|
+
* unique, sorted, always including 0 — so both layers agree on where a take
|
|
42
|
+
* begins. */
|
|
43
|
+
function cleanStarts(starts: readonly number[] | undefined, duration: number): number[] {
|
|
44
|
+
const seen = new Set<number>([0]);
|
|
45
|
+
for (const t of starts ?? []) {
|
|
46
|
+
if (Number.isFinite(t) && t > 0 && t < duration) seen.add(t);
|
|
47
|
+
}
|
|
48
|
+
return [...seen].sort((a, b) => a - b);
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Merge graphic cues with derived plain cues covering every gap; returns the
|
|
53
|
+
* union, time-sorted. Input must be the GRAPHIC cues (override-applied,
|
|
54
|
+
* hidden ones already dropped) — feeding an already-filled list back in would
|
|
55
|
+
* treat the previous fill's takes as occupied windows.
|
|
56
|
+
*/
|
|
57
|
+
export function fillPlainCues(
|
|
58
|
+
cues: readonly SceneCue[],
|
|
59
|
+
opts: FillPlainOptions,
|
|
60
|
+
): SceneCue[] {
|
|
61
|
+
const duration = opts.outputDurationSec;
|
|
62
|
+
const sorted = [...cues].sort((a, b) => a.startSec - b.startSec);
|
|
63
|
+
if (duration <= 0) return sorted;
|
|
64
|
+
const starts = cleanStarts(opts.clipStarts, duration);
|
|
65
|
+
|
|
66
|
+
// Gaps: before the first cue, between consecutive ones, after the last.
|
|
67
|
+
const gaps: Array<{ start: number; end: number }> = [];
|
|
68
|
+
let cursor = 0;
|
|
69
|
+
for (const cue of sorted) {
|
|
70
|
+
if (cue.startSec - cursor > 1e-9) gaps.push({ start: cursor, end: cue.startSec });
|
|
71
|
+
cursor = Math.max(cursor, cue.endSec);
|
|
72
|
+
}
|
|
73
|
+
if (duration - cursor > 1e-9) gaps.push({ start: cursor, end: duration });
|
|
74
|
+
|
|
75
|
+
// Split each gap at every cut strictly inside it, then drop slivers.
|
|
76
|
+
const pieces: Array<{ start: number; end: number; clip: number }> = [];
|
|
77
|
+
for (const gap of gaps) {
|
|
78
|
+
const inner = starts.filter((t) => t > gap.start + 1e-9 && t < gap.end - 1e-9);
|
|
79
|
+
const bounds = [gap.start, ...inner, gap.end];
|
|
80
|
+
for (let i = 0; i < bounds.length - 1; i++) {
|
|
81
|
+
const start = bounds[i]!;
|
|
82
|
+
const end = bounds[i + 1]!;
|
|
83
|
+
if (end - start < MIN_PLAIN_SEC) continue;
|
|
84
|
+
// The clip this piece belongs to: the last cut at or before its start.
|
|
85
|
+
let clip = 0;
|
|
86
|
+
for (let c = 0; c < starts.length; c++) {
|
|
87
|
+
if (starts[c]! <= start + 1e-9) clip = c;
|
|
88
|
+
}
|
|
89
|
+
pieces.push({ start, end, clip });
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// Per-clip ordinals: the first piece of a clip keeps the bare name, so a
|
|
94
|
+
// merge (a graphic deleted) lands back on an id that may already carry an
|
|
95
|
+
// override.
|
|
96
|
+
const perClip = new Map<number, number>();
|
|
97
|
+
const plain: SceneCue[] = pieces.map((p) => {
|
|
98
|
+
const k = perClip.get(p.clip) ?? 0;
|
|
99
|
+
perClip.set(p.clip, k + 1);
|
|
100
|
+
return {
|
|
101
|
+
id: k === 0 ? `take-${p.clip}` : `take-${p.clip}-${k}`,
|
|
102
|
+
kind: "plain",
|
|
103
|
+
layout: "full-bleed",
|
|
104
|
+
startSec: p.start,
|
|
105
|
+
endSec: p.end,
|
|
106
|
+
};
|
|
107
|
+
});
|
|
108
|
+
|
|
109
|
+
return [...sorted, ...plain].sort((a, b) => a.startSec - b.startSec);
|
|
110
|
+
}
|