@ossclip/core 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +27 -0
- package/README.md +20 -0
- package/package.json +29 -0
- package/src/analyze.ts +299 -0
- package/src/assemble.ts +124 -0
- package/src/browser.ts +24 -0
- package/src/captions.ts +92 -0
- package/src/clip.ts +306 -0
- package/src/config.ts +66 -0
- package/src/content-rect-detect.ts +162 -0
- package/src/content-rect.ts +324 -0
- package/src/cover.ts +216 -0
- package/src/cta.ts +68 -0
- package/src/cutlist.ts +170 -0
- package/src/exec.ts +36 -0
- package/src/face.ts +519 -0
- package/src/fill.ts +110 -0
- package/src/framing.ts +277 -0
- package/src/grounding.ts +130 -0
- package/src/index.ts +27 -0
- package/src/ingest.ts +83 -0
- package/src/normalize.ts +397 -0
- package/src/overrides.ts +509 -0
- package/src/phonetics.ts +129 -0
- package/src/producer/anthropic.ts +73 -0
- package/src/producer/beats.ts +330 -0
- package/src/producer/claude-cli.ts +150 -0
- package/src/producer/gemini.ts +197 -0
- package/src/producer/index.ts +217 -0
- package/src/producer/mock.ts +101 -0
- package/src/producer/provider.ts +42 -0
- package/src/producer/repair.ts +474 -0
- package/src/producer/scene-props.ts +212 -0
- package/src/producer/tiered.ts +56 -0
- package/src/producer/usage.ts +426 -0
- package/src/report.ts +36 -0
- package/src/scene-registry.ts +246 -0
- package/src/scene-schema.ts +203 -0
- package/src/schema.ts +177 -0
- package/src/source-text.ts +348 -0
- package/src/timemap.ts +115 -0
- package/src/transcribe.ts +67 -0
- package/src/zoom.ts +154 -0
package/src/clip.ts
ADDED
|
@@ -0,0 +1,306 @@
|
|
|
1
|
+
import { z } from "zod/v4";
|
|
2
|
+
import type { Segment, Transcript } from "./schema";
|
|
3
|
+
import { LEAD_KEEP, TAIL_KEEP } from "./cutlist";
|
|
4
|
+
import type { ClipHighlight, Moment } from "./producer/beats";
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* `--clip` highlight selection (R19 §93): choose ONE window of a long take
|
|
8
|
+
* and produce only it. Everything here is deliberately pure and runs BEFORE
|
|
9
|
+
* analyze/cut/captions/scenes — the transcript is sliced to the window and
|
|
10
|
+
* the existing pipeline runs unchanged on the slice (§93.1). Nothing in this
|
|
11
|
+
* module (or because of it) touches captions or the time map; if that ever
|
|
12
|
+
* seems necessary, selection has been put in the wrong place.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
/** The resolved clip window. Word indices are in the index space of the
|
|
16
|
+
* transcript selection ran against (the repaired, PRE-slice transcript);
|
|
17
|
+
* seconds are source time and remain meaningful after slicing. */
|
|
18
|
+
export const ClipWindowSchema = z.object({
|
|
19
|
+
startWord: z.number().int().nonnegative(),
|
|
20
|
+
endWord: z.number().int().nonnegative(),
|
|
21
|
+
startSec: z.number().nonnegative(),
|
|
22
|
+
endSec: z.number().nonnegative(),
|
|
23
|
+
/** The producer's one-line account of why THIS window (§93h). */
|
|
24
|
+
reason: z.string(),
|
|
25
|
+
});
|
|
26
|
+
export type ClipWindow = z.infer<typeof ClipWindowSchema>;
|
|
27
|
+
|
|
28
|
+
/** §93e: a window shorter than half the ask is a selection failure, not a clip. */
|
|
29
|
+
export const CLIP_MIN_FRACTION = 0.5;
|
|
30
|
+
/** Author decision (plan 2026-08-03): sentence snapping within ±20% of target. */
|
|
31
|
+
export const CLIP_SNAP_TOLERANCE = 0.2;
|
|
32
|
+
|
|
33
|
+
/** Word that closes a sentence — ASR punctuation rides on the word text. */
|
|
34
|
+
const SENTENCE_END = /[.!?…]["'")\]»]*$/u;
|
|
35
|
+
|
|
36
|
+
const isSentenceEnd = (t: Transcript, i: number): boolean =>
|
|
37
|
+
SENTENCE_END.test(t.words[i]?.text ?? "");
|
|
38
|
+
const isSentenceStart = (t: Transcript, i: number): boolean =>
|
|
39
|
+
i === 0 || isSentenceEnd(t, i - 1);
|
|
40
|
+
|
|
41
|
+
const durSec = (t: Transcript, start: number, end: number): number =>
|
|
42
|
+
t.words[end]!.end - t.words[start]!.start;
|
|
43
|
+
|
|
44
|
+
export interface ResolvedClip {
|
|
45
|
+
window: ClipWindow;
|
|
46
|
+
/** What resolution changed about the model's raw pick — for the console. */
|
|
47
|
+
notes: string[];
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Validate and snap the model's highlight (§93e + the snapping decision).
|
|
52
|
+
* The indices are untrusted LLM output: clamped, order-checked, snapped to
|
|
53
|
+
* sentence boundaries within ±20% of the target, trimmed if over-long, and
|
|
54
|
+
* REFUSED (thrown, with the reason) when what remains is under half the
|
|
55
|
+
* target — never a silent fallback to the full take, which would quietly
|
|
56
|
+
* "clip" a 20-minute video to 20 minutes.
|
|
57
|
+
*/
|
|
58
|
+
export function resolveClipWindow(
|
|
59
|
+
transcript: Transcript,
|
|
60
|
+
highlight: ClipHighlight | undefined,
|
|
61
|
+
targetSec: number,
|
|
62
|
+
): ResolvedClip {
|
|
63
|
+
const words = transcript.words;
|
|
64
|
+
if (words.length === 0) throw new Error("--clip: transcript has no words to select from");
|
|
65
|
+
if (!highlight) {
|
|
66
|
+
throw new Error(
|
|
67
|
+
"--clip: the producer returned no highlight window — cannot select a clip. " +
|
|
68
|
+
"Re-run, or try a different --llm/--llm-model.",
|
|
69
|
+
);
|
|
70
|
+
}
|
|
71
|
+
const notes: string[] = [];
|
|
72
|
+
const maxIndex = words.length - 1;
|
|
73
|
+
if (highlight.startWord > maxIndex) {
|
|
74
|
+
throw new Error(
|
|
75
|
+
`--clip: highlight starts at word ${highlight.startWord}, beyond the transcript (${maxIndex})`,
|
|
76
|
+
);
|
|
77
|
+
}
|
|
78
|
+
let start = highlight.startWord;
|
|
79
|
+
let end = Math.min(highlight.endWord, maxIndex);
|
|
80
|
+
if (end !== highlight.endWord) notes.push(`endWord clamped ${highlight.endWord} → ${end}`);
|
|
81
|
+
if (end <= start) {
|
|
82
|
+
throw new Error(`--clip: highlight window is empty or inverted (words ${start}–${end})`);
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
const tol = CLIP_SNAP_TOLERANCE * targetSec;
|
|
86
|
+
const cap = targetSec * (1 + CLIP_SNAP_TOLERANCE);
|
|
87
|
+
|
|
88
|
+
// Snap the START to the nearest sentence start within tolerance — a clip
|
|
89
|
+
// that opens mid-sentence reads as broken regardless of the pick's quality.
|
|
90
|
+
{
|
|
91
|
+
let best = -1;
|
|
92
|
+
let bestDist = Infinity;
|
|
93
|
+
for (let i = 0; i <= maxIndex; i++) {
|
|
94
|
+
if (!isSentenceStart(transcript, i)) continue;
|
|
95
|
+
const dist = Math.abs(words[i]!.start - words[start]!.start);
|
|
96
|
+
if (dist <= tol && dist < bestDist) {
|
|
97
|
+
best = i;
|
|
98
|
+
bestDist = dist;
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
if (best !== -1 && best !== start) {
|
|
102
|
+
notes.push(`start snapped to sentence boundary: word ${start} → ${best}`);
|
|
103
|
+
start = best;
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
if (end <= start) end = Math.min(start + 1, maxIndex);
|
|
107
|
+
|
|
108
|
+
// The END: snap to a sentence end. An over-long window is trimmed to the
|
|
109
|
+
// sentence end nearest the target (the hook lives at the start — always
|
|
110
|
+
// trim the tail); an in-tolerance one snaps to the nearest boundary.
|
|
111
|
+
const sentenceEnds: number[] = [];
|
|
112
|
+
for (let i = start; i <= maxIndex; i++) if (isSentenceEnd(transcript, i)) sentenceEnds.push(i);
|
|
113
|
+
if (durSec(transcript, start, end) > cap) {
|
|
114
|
+
let best = -1;
|
|
115
|
+
let bestDist = Infinity;
|
|
116
|
+
for (const e of sentenceEnds) {
|
|
117
|
+
if (e <= start || e > end) continue;
|
|
118
|
+
const d = durSec(transcript, start, e);
|
|
119
|
+
if (d > cap) continue;
|
|
120
|
+
const dist = Math.abs(d - targetSec);
|
|
121
|
+
if (dist < bestDist) {
|
|
122
|
+
best = e;
|
|
123
|
+
bestDist = dist;
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
if (best === -1) {
|
|
127
|
+
// No sentence boundary under the cap — trim at a word boundary instead.
|
|
128
|
+
let e = end;
|
|
129
|
+
while (e > start && durSec(transcript, start, e) > cap) e--;
|
|
130
|
+
best = Math.max(e, start + 1);
|
|
131
|
+
}
|
|
132
|
+
notes.push(
|
|
133
|
+
`window trimmed ${durSec(transcript, start, end).toFixed(1)}s → ` +
|
|
134
|
+
`${durSec(transcript, start, best).toFixed(1)}s (target ${targetSec}s +${(
|
|
135
|
+
CLIP_SNAP_TOLERANCE * 100
|
|
136
|
+
).toFixed(0)}%)`,
|
|
137
|
+
);
|
|
138
|
+
end = best;
|
|
139
|
+
} else if (!isSentenceEnd(transcript, end)) {
|
|
140
|
+
let best = -1;
|
|
141
|
+
let bestDist = Infinity;
|
|
142
|
+
for (const e of sentenceEnds) {
|
|
143
|
+
if (e <= start) continue;
|
|
144
|
+
const dist = Math.abs(words[e]!.end - words[end]!.end);
|
|
145
|
+
if (dist <= tol && dist < bestDist) {
|
|
146
|
+
best = e;
|
|
147
|
+
bestDist = dist;
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
if (best !== -1 && best !== end) {
|
|
151
|
+
notes.push(`end snapped to sentence boundary: word ${end} → ${best}`);
|
|
152
|
+
end = best;
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
const dur = durSec(transcript, start, end);
|
|
157
|
+
if (dur < CLIP_MIN_FRACTION * targetSec) {
|
|
158
|
+
throw new Error(
|
|
159
|
+
`--clip: the selected window is ${dur.toFixed(1)}s — under half the ${targetSec}s target. ` +
|
|
160
|
+
`Refusing to produce it (the take may not contain ${targetSec}s of connected material; ` +
|
|
161
|
+
`try a shorter --clip, or run without it).`,
|
|
162
|
+
);
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
return {
|
|
166
|
+
window: {
|
|
167
|
+
startWord: start,
|
|
168
|
+
endWord: end,
|
|
169
|
+
startSec: words[start]!.start,
|
|
170
|
+
endSec: words[end]!.end,
|
|
171
|
+
reason: highlight.reason,
|
|
172
|
+
},
|
|
173
|
+
notes,
|
|
174
|
+
};
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* A pinned window from `command.json` (§93g): "start:end" word indices.
|
|
179
|
+
* The editor's Render replays the recorded argv, and replay must reproduce
|
|
180
|
+
* the SAME window with zero LLM calls — so the pin is authoritative and is
|
|
181
|
+
* validated but never re-snapped (it was snapped when first resolved).
|
|
182
|
+
*/
|
|
183
|
+
export function parseClipWindowPin(transcript: Transcript, pin: string): ClipWindow {
|
|
184
|
+
const m = /^(\d+):(\d+)$/.exec(pin);
|
|
185
|
+
if (!m) throw new Error(`--clip-window: expected "startWord:endWord", got "${pin}"`);
|
|
186
|
+
const startWord = Number.parseInt(m[1]!, 10);
|
|
187
|
+
const endWord = Number.parseInt(m[2]!, 10);
|
|
188
|
+
const maxIndex = transcript.words.length - 1;
|
|
189
|
+
if (maxIndex < 0) throw new Error("--clip-window: transcript has no words");
|
|
190
|
+
if (startWord > maxIndex || endWord > maxIndex || endWord <= startWord) {
|
|
191
|
+
throw new Error(
|
|
192
|
+
`--clip-window ${pin} does not fit this transcript (${maxIndex + 1} words) — ` +
|
|
193
|
+
`was it recorded against different footage?`,
|
|
194
|
+
);
|
|
195
|
+
}
|
|
196
|
+
return {
|
|
197
|
+
startWord,
|
|
198
|
+
endWord,
|
|
199
|
+
startSec: transcript.words[startWord]!.start,
|
|
200
|
+
endSec: transcript.words[endWord]!.end,
|
|
201
|
+
reason: "pinned window (command.json replay)",
|
|
202
|
+
};
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/** The transcript, cut down to the window. Source-time stamps are untouched —
|
|
206
|
+
* everything downstream still reasons in source time through the cutlist. */
|
|
207
|
+
export function sliceTranscript(transcript: Transcript, window: ClipWindow): Transcript {
|
|
208
|
+
return { ...transcript, words: transcript.words.slice(window.startWord, window.endWord + 1) };
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
/**
|
|
212
|
+
* Re-anchor beat-sheet moments into the sliced index space: moments outside
|
|
213
|
+
* the window drop, partial ones clamp, survivors shift by the window start.
|
|
214
|
+
*/
|
|
215
|
+
export function sliceMoments(moments: readonly Moment[], window: ClipWindow): Moment[] {
|
|
216
|
+
return moments.flatMap((m) => {
|
|
217
|
+
if (m.endWord < window.startWord || m.startWord > window.endWord) return [];
|
|
218
|
+
return [
|
|
219
|
+
{
|
|
220
|
+
...m,
|
|
221
|
+
startWord: Math.max(m.startWord, window.startWord) - window.startWord,
|
|
222
|
+
endWord: Math.min(m.endWord, window.endWord) - window.startWord,
|
|
223
|
+
},
|
|
224
|
+
];
|
|
225
|
+
});
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
/**
|
|
229
|
+
* Slice the RAW transcript by TIME, not by the window's word indices: repairs
|
|
230
|
+
* may merge or split words (`applyRepairs` splices), so raw and repaired
|
|
231
|
+
* index spaces need not line up — but both carry source-time stamps, and the
|
|
232
|
+
* window's seconds are exact. Returns the slice and its offset in the raw
|
|
233
|
+
* index space, for shifting `repairs` alongside.
|
|
234
|
+
*/
|
|
235
|
+
export function sliceRawTranscript(
|
|
236
|
+
raw: Transcript,
|
|
237
|
+
window: ClipWindow,
|
|
238
|
+
): { transcript: Transcript; offset: number } {
|
|
239
|
+
const inWindow = (w: { start: number; end: number }): boolean =>
|
|
240
|
+
w.end > window.startSec && w.start < window.endSec;
|
|
241
|
+
const offset = raw.words.findIndex(inWindow);
|
|
242
|
+
if (offset === -1) return { transcript: { ...raw, words: [] }, offset: 0 };
|
|
243
|
+
const words = raw.words.filter(inWindow);
|
|
244
|
+
return { transcript: { ...raw, words }, offset };
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
/** Keep only repairs that live wholly inside the sliced raw range, shifted
|
|
248
|
+
* into its index space — `production.json` stores raw + repairs as the
|
|
249
|
+
* reproducible pair, and a repair pointing outside the slice breaks that. */
|
|
250
|
+
export function sliceRepairs<T extends { startWord: number; endWord: number }>(
|
|
251
|
+
repairs: readonly T[],
|
|
252
|
+
offset: number,
|
|
253
|
+
wordCount: number,
|
|
254
|
+
): T[] {
|
|
255
|
+
return repairs
|
|
256
|
+
.filter((r) => r.startWord >= offset && r.endWord < offset + wordCount)
|
|
257
|
+
.map((r) => ({ ...r, startWord: r.startWord - offset, endWord: r.endWord - offset }));
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
/**
|
|
261
|
+
* Bound a cutlist to the window: everything outside becomes a single `clip`
|
|
262
|
+
* removal at each end, everything inside is preserved. The result stays a
|
|
263
|
+
* full partition of [0, duration], so the TimeMap invariant
|
|
264
|
+
* (`outputDuration === Σ kept`) holds by construction. The window is padded
|
|
265
|
+
* by the cutlist's own lead/tail keeps so a clip boundary breathes exactly
|
|
266
|
+
* like a take boundary.
|
|
267
|
+
*/
|
|
268
|
+
export function boundCutlistToWindow(
|
|
269
|
+
segments: readonly Segment[],
|
|
270
|
+
window: ClipWindow,
|
|
271
|
+
duration: number,
|
|
272
|
+
): Segment[] {
|
|
273
|
+
const winIn = Math.max(0, Math.min(window.startSec - LEAD_KEEP, duration));
|
|
274
|
+
const winOut = Math.min(duration, Math.max(window.endSec + TAIL_KEEP, winIn));
|
|
275
|
+
const out: Segment[] = [];
|
|
276
|
+
if (winIn > 0) {
|
|
277
|
+
out.push({ srcIn: 0, srcOut: winIn, kind: "remove", reason: "clip", confidence: 1 });
|
|
278
|
+
}
|
|
279
|
+
for (const s of segments) {
|
|
280
|
+
const srcIn = Math.max(s.srcIn, winIn);
|
|
281
|
+
const srcOut = Math.min(s.srcOut, winOut);
|
|
282
|
+
if (srcOut - srcIn <= 1e-9) continue;
|
|
283
|
+
const prev = out[out.length - 1];
|
|
284
|
+
if (prev && prev.kind === s.kind && prev.reason === s.reason) {
|
|
285
|
+
prev.srcOut = srcOut;
|
|
286
|
+
} else {
|
|
287
|
+
out.push({ ...s, srcIn, srcOut });
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
if (winOut < duration) {
|
|
291
|
+
const prev = out[out.length - 1];
|
|
292
|
+
if (prev && prev.kind === "remove" && prev.reason === "clip") {
|
|
293
|
+
prev.srcOut = duration;
|
|
294
|
+
} else {
|
|
295
|
+
out.push({ srcIn: winOut, srcOut: duration, kind: "remove", reason: "clip", confidence: 1 });
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
return out;
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
/** m:ss for the console/report — clip windows live in minutes, not seconds. */
|
|
302
|
+
export function formatClipTime(sec: number): string {
|
|
303
|
+
const m = Math.floor(sec / 60);
|
|
304
|
+
const s = Math.floor(sec % 60);
|
|
305
|
+
return `${m}:${s.toString().padStart(2, "0")}`;
|
|
306
|
+
}
|
package/src/config.ts
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
import { readFileSync } from "node:fs";
|
|
2
|
+
import { homedir } from "node:os";
|
|
3
|
+
import { join } from "node:path";
|
|
4
|
+
|
|
5
|
+
import type { ModelPrice } from "./producer/usage";
|
|
6
|
+
|
|
7
|
+
export interface OssclipConfig {
|
|
8
|
+
ffmpegPath: string;
|
|
9
|
+
ffprobePath: string;
|
|
10
|
+
whisperPath: string;
|
|
11
|
+
modelDir: string;
|
|
12
|
+
model: string;
|
|
13
|
+
/**
|
|
14
|
+
* Model for mechanical LLM calls (repair, scene props) — the editorial beat
|
|
15
|
+
* sheet always uses the main model. "same" disables tiering (FINDINGS §37).
|
|
16
|
+
*/
|
|
17
|
+
fastModel?: string;
|
|
18
|
+
/**
|
|
19
|
+
* Who is in the video — "Ahsan, host of the Code with Ahsan channel".
|
|
20
|
+
* Lets the repair pass recognise a mangled proper noun instead of inventing
|
|
21
|
+
* a plausible one, and stops grounding flagging the speaker's own name.
|
|
22
|
+
*/
|
|
23
|
+
speaker?: string;
|
|
24
|
+
browserExecutable?: string;
|
|
25
|
+
/**
|
|
26
|
+
* USD per million tokens, keyed by model id or family substring — overrides
|
|
27
|
+
* the built-in assumptions in `producer/usage.ts` so a run's cost line
|
|
28
|
+
* reflects the account's actual rates instead of ours.
|
|
29
|
+
*/
|
|
30
|
+
pricing?: Record<string, ModelPrice>;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
const DEFAULTS: OssclipConfig = {
|
|
34
|
+
ffmpegPath: "ffmpeg",
|
|
35
|
+
ffprobePath: "ffprobe",
|
|
36
|
+
whisperPath: "whisper-cli",
|
|
37
|
+
modelDir: join(homedir(), ".ossclip", "models"),
|
|
38
|
+
// small.en over base.en: a large accuracy step for accented English at
|
|
39
|
+
// modest cost — base.en turned "code churn" into a company name that then
|
|
40
|
+
// reached the captions and a hook label (FINDINGS §14b).
|
|
41
|
+
model: "small.en",
|
|
42
|
+
};
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Resolution order per key: env (OSSCLIP_FFMPEG, OSSCLIP_FFPROBE, OSSCLIP_WHISPER,
|
|
46
|
+
* OSSCLIP_MODEL_DIR, OSSCLIP_MODEL, OSSCLIP_BROWSER) → ~/.ossclip/config.json → defaults.
|
|
47
|
+
*/
|
|
48
|
+
export function loadConfig(): OssclipConfig {
|
|
49
|
+
let fileCfg: Partial<OssclipConfig> = {};
|
|
50
|
+
try {
|
|
51
|
+
fileCfg = JSON.parse(readFileSync(join(homedir(), ".ossclip", "config.json"), "utf8"));
|
|
52
|
+
} catch {
|
|
53
|
+
// no config file — fine
|
|
54
|
+
}
|
|
55
|
+
return {
|
|
56
|
+
ffmpegPath: process.env.OSSCLIP_FFMPEG ?? fileCfg.ffmpegPath ?? DEFAULTS.ffmpegPath,
|
|
57
|
+
ffprobePath: process.env.OSSCLIP_FFPROBE ?? fileCfg.ffprobePath ?? DEFAULTS.ffprobePath,
|
|
58
|
+
whisperPath: process.env.OSSCLIP_WHISPER ?? fileCfg.whisperPath ?? DEFAULTS.whisperPath,
|
|
59
|
+
modelDir: process.env.OSSCLIP_MODEL_DIR ?? fileCfg.modelDir ?? DEFAULTS.modelDir,
|
|
60
|
+
model: process.env.OSSCLIP_MODEL ?? fileCfg.model ?? DEFAULTS.model,
|
|
61
|
+
fastModel: process.env.OSSCLIP_FAST_MODEL ?? fileCfg.fastModel,
|
|
62
|
+
speaker: process.env.OSSCLIP_SPEAKER ?? fileCfg.speaker,
|
|
63
|
+
browserExecutable: process.env.OSSCLIP_BROWSER ?? fileCfg.browserExecutable,
|
|
64
|
+
pricing: fileCfg.pricing,
|
|
65
|
+
};
|
|
66
|
+
}
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
import { readFile, writeFile } from "node:fs/promises";
|
|
2
|
+
import { existsSync } from "node:fs";
|
|
3
|
+
import { join } from "node:path";
|
|
4
|
+
import { run } from "./exec";
|
|
5
|
+
import {
|
|
6
|
+
contentRectTimeline,
|
|
7
|
+
parseCropdetect,
|
|
8
|
+
pickTransition,
|
|
9
|
+
type ContentRect,
|
|
10
|
+
type ContentRectSegment,
|
|
11
|
+
} from "./content-rect";
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* The ffmpeg + cache half of letterbox detection. Split from `content-rect.ts`
|
|
15
|
+
* so the pure geometry (timeline, lookup, crop filter) can be imported by the
|
|
16
|
+
* Remotion bundle, which must never pull in node built-ins — the render-time
|
|
17
|
+
* crop of a mixed-framing source needs exactly that geometry (PLAN Task C).
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
export interface DetectContentRectOptions {
|
|
21
|
+
/** Samples per second. Must be dense enough to see the SHORTEST segment. */
|
|
22
|
+
rate?: number;
|
|
23
|
+
/** Hard ceiling on samples so a feature-length source stays cheap. */
|
|
24
|
+
maxSamples?: number;
|
|
25
|
+
cacheDir?: string;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* v2 stored the coarse timeline (PLAN Task C); v3 stores boundaries REFINED
|
|
30
|
+
* to the frame (NORMALIZE plan) — a v2 cache carries ±0.25s boundaries that
|
|
31
|
+
* would be baked into the normalized source, so it is re-measured.
|
|
32
|
+
*/
|
|
33
|
+
const CACHE_VERSION = 3;
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Sampling rate. Task 7 took 12 samples spread over the whole source, which is
|
|
37
|
+
* one every ~5s on a 64s clip and cannot see a 3.5s framing change. cropdetect
|
|
38
|
+
* decodes every frame regardless of the `fps` filter, so a denser rate costs
|
|
39
|
+
* almost nothing beyond parsing.
|
|
40
|
+
*/
|
|
41
|
+
const DEFAULT_RATE = 2;
|
|
42
|
+
const DEFAULT_MAX_SAMPLES = 900;
|
|
43
|
+
|
|
44
|
+
export interface ContentRectDetection {
|
|
45
|
+
/** Framing over source time. Always non-empty; one entry when uniform. */
|
|
46
|
+
timeline: ContentRectSegment[];
|
|
47
|
+
/**
|
|
48
|
+
* The single rect when — and only when — the source has uniform framing.
|
|
49
|
+
* `null` for a mixed source, so a caller that needs one constant has to
|
|
50
|
+
* decide what to do rather than silently getting the first segment's.
|
|
51
|
+
*/
|
|
52
|
+
uniform: ContentRect | null;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Measure the source's framing, cached in the workdir like `face.json`.
|
|
57
|
+
* One decode pass; cropdetect logs at info level, so this runs without
|
|
58
|
+
* `-v error` and reads the filter's stderr lines.
|
|
59
|
+
*/
|
|
60
|
+
export async function detectContentRect(
|
|
61
|
+
tools: { ffmpegPath: string },
|
|
62
|
+
videoPath: string,
|
|
63
|
+
probe: { width: number; height: number; duration: number },
|
|
64
|
+
opts: DetectContentRectOptions = {},
|
|
65
|
+
): Promise<ContentRectDetection> {
|
|
66
|
+
const cachePath = opts.cacheDir ? join(opts.cacheDir, "content-rect.json") : null;
|
|
67
|
+
if (cachePath && existsSync(cachePath)) {
|
|
68
|
+
const cached = JSON.parse(await readFile(cachePath, "utf8")) as {
|
|
69
|
+
version?: number;
|
|
70
|
+
timeline?: ContentRectSegment[];
|
|
71
|
+
};
|
|
72
|
+
if (cached.version === CACHE_VERSION && cached.timeline?.length) {
|
|
73
|
+
return withUniform(cached.timeline);
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
const duration = Math.max(1, probe.duration);
|
|
78
|
+
const rate = Math.min(opts.rate ?? DEFAULT_RATE, (opts.maxSamples ?? DEFAULT_MAX_SAMPLES) / duration);
|
|
79
|
+
const { stderr } = await run(
|
|
80
|
+
tools.ffmpegPath,
|
|
81
|
+
[
|
|
82
|
+
"-i", videoPath,
|
|
83
|
+
"-vf", `fps=${Math.max(0.05, rate).toFixed(4)},cropdetect=limit=24:round=2:reset=1`,
|
|
84
|
+
"-f", "null", "-",
|
|
85
|
+
],
|
|
86
|
+
{ allowNonZero: true },
|
|
87
|
+
);
|
|
88
|
+
const timeline = contentRectTimeline(
|
|
89
|
+
parseCropdetect(stderr),
|
|
90
|
+
probe.width,
|
|
91
|
+
probe.height,
|
|
92
|
+
probe.duration,
|
|
93
|
+
);
|
|
94
|
+
await refineBoundaries(tools, videoPath, timeline, probe);
|
|
95
|
+
if (cachePath) {
|
|
96
|
+
await writeFile(cachePath, JSON.stringify({ version: CACHE_VERSION, timeline }, null, 2));
|
|
97
|
+
}
|
|
98
|
+
return withUniform(timeline);
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/** How far around a coarse boundary the refinement pass looks. Must exceed the
|
|
102
|
+
* coarse sampling half-step (0.25s at 2 Hz) with room for run-absorption
|
|
103
|
+
* slop, and stay small enough that nine boundaries cost under a second of
|
|
104
|
+
* decode each. */
|
|
105
|
+
const REFINE_HALF_WINDOW_SEC = 0.8;
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* Sharpen each boundary of a mixed timeline to the frame, in place.
|
|
109
|
+
*
|
|
110
|
+
* The coarse pass samples at 2 Hz and places boundaries midway between
|
|
111
|
+
* disagreeing samples — good enough to steer a render-time crop, not good
|
|
112
|
+
* enough to BAKE: every frame on the wrong side of a baked boundary gets the
|
|
113
|
+
* wrong window. This decodes ~1.6s around each boundary at native fps
|
|
114
|
+
* (`-ss` before `-i`, so it is a seek, not a scan; cropdetect's `t:` restarts
|
|
115
|
+
* at the seek point) and moves the boundary to the midpoint of the two frames
|
|
116
|
+
* that actually disagree. A window that never straddles the change keeps the
|
|
117
|
+
* coarse estimate — wrong by less than the window, and said so by the caller's
|
|
118
|
+
* log rather than silently.
|
|
119
|
+
*/
|
|
120
|
+
async function refineBoundaries(
|
|
121
|
+
tools: { ffmpegPath: string },
|
|
122
|
+
videoPath: string,
|
|
123
|
+
timeline: ContentRectSegment[],
|
|
124
|
+
probe: { width: number; height: number; duration: number },
|
|
125
|
+
): Promise<void> {
|
|
126
|
+
for (let i = 1; i < timeline.length; i++) {
|
|
127
|
+
const coarse = timeline[i]!.startSec;
|
|
128
|
+
const from = Math.max(0, coarse - REFINE_HALF_WINDOW_SEC);
|
|
129
|
+
const { stderr } = await run(
|
|
130
|
+
tools.ffmpegPath,
|
|
131
|
+
[
|
|
132
|
+
"-ss", from.toFixed(3),
|
|
133
|
+
"-t", (REFINE_HALF_WINDOW_SEC * 2).toFixed(3),
|
|
134
|
+
"-i", videoPath,
|
|
135
|
+
"-vf", "cropdetect=limit=24:round=2:reset=1",
|
|
136
|
+
"-f", "null", "-",
|
|
137
|
+
],
|
|
138
|
+
{ allowNonZero: true },
|
|
139
|
+
);
|
|
140
|
+
const samples = parseCropdetect(stderr).map((s) => ({ ...s, tSec: from + s.tSec }));
|
|
141
|
+
const exact = pickTransition(
|
|
142
|
+
samples,
|
|
143
|
+
timeline[i - 1]!.rect,
|
|
144
|
+
timeline[i]!.rect,
|
|
145
|
+
probe.width,
|
|
146
|
+
probe.height,
|
|
147
|
+
);
|
|
148
|
+
if (exact !== null) {
|
|
149
|
+
timeline[i - 1]!.endSec = exact;
|
|
150
|
+
timeline[i]!.startSec = exact;
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
function withUniform(timeline: ContentRectSegment[]): ContentRectDetection {
|
|
156
|
+
return { timeline, uniform: timeline.length === 1 ? timeline[0]!.rect : null };
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/** Total source seconds the framing is NOT the full frame — for reporting. */
|
|
160
|
+
export function letterboxedSeconds(timeline: readonly ContentRectSegment[]): number {
|
|
161
|
+
return timeline.reduce((s, seg) => (seg.rect.full ? s : s + (seg.endSec - seg.startSec)), 0);
|
|
162
|
+
}
|