@ossclip/core 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/framing.ts ADDED
@@ -0,0 +1,277 @@
1
+ import type { Transcript } from "./schema";
2
+ import { SCENE_REGISTRY } from "./scene-registry";
3
+ import type { Layout } from "./scene-schema";
4
+ import type { Moment } from "./producer/beats";
5
+ import { headFracInSlot } from "./normalize";
6
+
7
+ /**
8
+ * Producer framing awareness (PLAN 2026-07-28, Tasks A and B).
9
+ *
10
+ * The beat-sheet prompt used to contain zero geometry — intent, duration,
11
+ * component menu, transcript. So the producer happily put `video-top` (a wide
12
+ * band) on moments where the speaker fills 44% of a portrait canvas, and the
13
+ * band cover-cropped the head to 206% of its height. Three rounds of tuning a
14
+ * global crop constant could not fix that, because it is not a cropping
15
+ * problem: the pixels a wide band wants do not exist in a portrait close-up.
16
+ * The fix is editorial — don't put a wide band on a close-up moment — which
17
+ * makes it the PRODUCER's decision, and this module is how the producer gets
18
+ * the evidence (Task A) and how its choice is checked (Task B).
19
+ *
20
+ * Two design rules, from the plan:
21
+ * - The model never does pixel math. The brief is qualitative — CLOSE /
22
+ * MEDIUM / WIDE plus which layouts that rules out. The arithmetic stays
23
+ * in `headFracInSlot`, where it is tested.
24
+ * - The repair pass ships regardless of how well the prompt behaves. A
25
+ * prompt is a request; `repairMomentLayouts` is the constraint (the §35
26
+ * lesson).
27
+ */
28
+
29
+ /** One framing window in SOURCE time, from the normalization plan. */
30
+ export interface FramingWindow {
31
+ startSec: number;
32
+ endSec: number;
33
+ /** Face height as a fraction of the (normalized) canvas during this window. */
34
+ faceFracOfCanvas: number;
35
+ }
36
+
37
+ /**
38
+ * A layout's video-slot shape. Injected by the CLI from `layoutSlots` —
39
+ * core stays React- and scenes-free, and the feasibility question needs only
40
+ * the slot's ASPECT and whether the video is the subject there at all.
41
+ */
42
+ export interface LayoutFraming {
43
+ layout: Layout;
44
+ /** Slot width over height, in output pixels. */
45
+ slotAspect: number;
46
+ /**
47
+ * False for slots where the head-fits rule does not apply: a pip bubble is
48
+ * MEANT to be a tight head shot, and a `graphic-only` slot never draws.
49
+ */
50
+ primary: boolean;
51
+ }
52
+
53
+ /** Everything a framing judgement needs, assembled once by the CLI. */
54
+ export interface FramingContext {
55
+ windows: FramingWindow[];
56
+ /** The normalized canvas's width over height. */
57
+ canvasAspect: number;
58
+ layouts: LayoutFraming[];
59
+ /** The idle zoom's peak — the head must fit at the tightest moment. */
60
+ zoom: number;
61
+ }
62
+
63
+ /** The moment's span in SOURCE seconds, off the transcript's own stamps. */
64
+ export function momentSourceWindow(
65
+ transcript: Transcript,
66
+ startWord: number,
67
+ endWord: number,
68
+ ): { startSec: number; endSec: number } | null {
69
+ const first = transcript.words[startWord];
70
+ const last = transcript.words[endWord];
71
+ if (!first || !last) return null;
72
+ return { startSec: first.start, endSec: last.end };
73
+ }
74
+
75
+ /** The tightest framing inside [startSec, endSec] — a span is judged by its
76
+ * worst moment, not its average. */
77
+ export function worstFaceFrac(
78
+ windows: readonly FramingWindow[],
79
+ startSec: number,
80
+ endSec: number,
81
+ ): number {
82
+ let frac = 0;
83
+ for (const w of windows) {
84
+ if (w.startSec < endSec && w.endSec > startSec) frac = Math.max(frac, w.faceFracOfCanvas);
85
+ }
86
+ return frac;
87
+ }
88
+
89
+ /** Can this layout's slot hold the whole head at this framing? Non-primary
90
+ * slots are always feasible — the rule does not apply to them. */
91
+ export function layoutFeasible(ctx: FramingContext, layout: Layout, faceFrac: number): boolean {
92
+ const entry = ctx.layouts.find((l) => l.layout === layout);
93
+ if (!entry || !entry.primary || faceFrac <= 0) return true;
94
+ return headFracInSlot(faceFrac, ctx.canvasAspect, entry.slotAspect, ctx.zoom) <= 1;
95
+ }
96
+
97
+ /** The primary layouts this framing rules out, worst offender first. */
98
+ export function infeasibleLayouts(ctx: FramingContext, faceFrac: number): Layout[] {
99
+ return ctx.layouts
100
+ .filter((l) => !layoutFeasible(ctx, l.layout, faceFrac))
101
+ .sort(
102
+ (a, b) =>
103
+ headFracInSlot(faceFrac, ctx.canvasAspect, b.slotAspect, ctx.zoom) -
104
+ headFracInSlot(faceFrac, ctx.canvasAspect, a.slotAspect, ctx.zoom),
105
+ )
106
+ .map((l) => l.layout);
107
+ }
108
+
109
+ /** Qualitative label for the brief. CLOSE is defined by CONSEQUENCE — some
110
+ * layout is unavailable — not by an arbitrary fraction threshold. */
111
+ function shotLabel(ctx: FramingContext, faceFrac: number): "CLOSE" | "MEDIUM" | "WIDE" {
112
+ if (infeasibleLayouts(ctx, faceFrac).length > 0) return "CLOSE";
113
+ return faceFrac >= 0.3 ? "MEDIUM" : "WIDE";
114
+ }
115
+
116
+ /**
117
+ * The framing brief for the beat-sheet prompt (Task A).
118
+ *
119
+ * One line per framing stretch, in WORD INDICES — the producer reasons in
120
+ * word spans and never sees a second. Adjacent windows whose constraint is
121
+ * identical merge, so a ten-segment source with two real framings reads as a
122
+ * handful of lines, not a table. Windows with no face measurement carry no
123
+ * signal and produce no line; a source with no measured window at all returns
124
+ * "" and the prompt is unchanged.
125
+ */
126
+ export function buildFramingBrief(ctx: FramingContext, transcript: Transcript): string {
127
+ if (transcript.words.length === 0) return "";
128
+
129
+ // Word range per window: a word belongs to the window containing its middle.
130
+ const mid = (i: number): number => {
131
+ const w = transcript.words[i]!;
132
+ return (w.start + w.end) / 2;
133
+ };
134
+ type Line = { fromWord: number; toWord: number; faceFrac: number; avoid: Layout[] };
135
+ const lines: Line[] = [];
136
+ for (const win of ctx.windows) {
137
+ if (win.faceFracOfCanvas <= 0) continue;
138
+ let from = -1;
139
+ let to = -1;
140
+ for (let i = 0; i < transcript.words.length; i++) {
141
+ if (mid(i) >= win.startSec && mid(i) < win.endSec) {
142
+ if (from === -1) from = i;
143
+ to = i;
144
+ }
145
+ }
146
+ if (from === -1) continue;
147
+ const avoid = infeasibleLayouts(ctx, win.faceFracOfCanvas);
148
+ const prev = lines[lines.length - 1];
149
+ if (prev && prev.toWord === from - 1 && sameAvoid(prev.avoid, avoid)) {
150
+ // Same constraint, contiguous words: one line. The face fraction kept is
151
+ // the WORST, consistent with how spans are judged everywhere else.
152
+ prev.toWord = to;
153
+ prev.faceFrac = Math.max(prev.faceFrac, win.faceFracOfCanvas);
154
+ } else {
155
+ lines.push({ fromWord: from, toWord: to, faceFrac: win.faceFracOfCanvas, avoid });
156
+ }
157
+ }
158
+ if (lines.length === 0) return "";
159
+
160
+ const body = lines
161
+ .map((l) => {
162
+ const label = shotLabel(ctx, l.faceFrac);
163
+ const detail = `the face fills ~${Math.round(l.faceFrac * 100)}% of the frame height`;
164
+ return l.avoid.length > 0
165
+ ? `- words ${l.fromWord}-${l.toWord}: ${label} shot (${detail}) — layouts ` +
166
+ `${l.avoid.join(", ")} would crop the head here and are UNAVAILABLE`
167
+ : `- words ${l.fromWord}-${l.toWord}: ${label} shot (${detail}) — any layout works`;
168
+ })
169
+ .join("\n");
170
+ return (
171
+ "Camera framing by word range (measured from the footage — hard constraints, not suggestions):\n" +
172
+ body
173
+ );
174
+ }
175
+
176
+ function sameAvoid(a: readonly Layout[], b: readonly Layout[]): boolean {
177
+ return a.length === b.length && a.every((l, i) => l === b[i]);
178
+ }
179
+
180
+ export interface LayoutRepair {
181
+ /** Index into the moments array. */
182
+ moment: number;
183
+ issue: string;
184
+ }
185
+
186
+ /**
187
+ * The Task B safety net: rewrite any moment whose layout cannot physically
188
+ * hold the head at that moment's framing.
189
+ *
190
+ * Runs AFTER `normalizeBeatSheet` (word spans are already valid) and before
191
+ * scene generation. The producer's explicit choice is kept when feasible;
192
+ * an infeasible one — or an infeasible registry default when the producer
193
+ * said nothing — is swapped for the first feasible layout among the
194
+ * component's default and alternates. When NOTHING is feasible the least-bad
195
+ * candidate is taken and the issue says so, because a scene that trims a
196
+ * little is still better than silently rendering the worst option.
197
+ */
198
+ export function repairMomentLayouts(
199
+ moments: readonly Moment[],
200
+ transcript: Transcript,
201
+ ctx: FramingContext,
202
+ ): { moments: Moment[]; issues: LayoutRepair[] } {
203
+ const issues: LayoutRepair[] = [];
204
+ const out = moments.map((m, idx) => {
205
+ if (m.sceneKind === "none") return m;
206
+ const meta = SCENE_REGISTRY[m.sceneKind];
207
+ const window = momentSourceWindow(transcript, m.startWord, m.endWord);
208
+ if (!meta || !window) return m;
209
+ const faceFrac = worstFaceFrac(ctx.windows, window.startSec, window.endSec);
210
+ if (faceFrac <= 0) return m;
211
+
212
+ const requested = m.layout ?? meta.defaultLayout;
213
+ if (layoutFeasible(ctx, requested, faceFrac)) return m;
214
+
215
+ const candidates = [...new Set<Layout>([meta.defaultLayout, ...meta.altLayouts])];
216
+ const feasible = candidates.filter((l) => layoutFeasible(ctx, l, faceFrac));
217
+ if (feasible.length > 0) {
218
+ issues.push({
219
+ moment: idx,
220
+ issue: `layout ${requested} would crop the head on this close shot; using ${feasible[0]}`,
221
+ });
222
+ return { ...m, layout: feasible[0]! };
223
+ }
224
+ // Nothing fits: take the candidate that trims least, and say so.
225
+ const leastBad = candidates.reduce((a, b) => {
226
+ const fr = (l: Layout): number => {
227
+ const entry = ctx.layouts.find((e) => e.layout === l);
228
+ return entry && entry.primary
229
+ ? headFracInSlot(faceFrac, ctx.canvasAspect, entry.slotAspect, ctx.zoom)
230
+ : 0;
231
+ };
232
+ return fr(b) < fr(a) ? b : a;
233
+ });
234
+ issues.push({
235
+ moment: idx,
236
+ issue:
237
+ `no ${m.sceneKind} layout fully fits the head on this close shot; ` +
238
+ `${leastBad} trims least`,
239
+ });
240
+ return { ...m, layout: leastBad };
241
+ });
242
+ return { moments: out, issues };
243
+ }
244
+
245
+ /**
246
+ * Layouts that make sense in a LANDSCAPE frame (R15).
247
+ *
248
+ * `video-top`, `pip-bubble` and `graphic-only` are vertical-format ideas: they
249
+ * exist to split a tall frame between a face and a card, or to demote the
250
+ * speaker to an inset because a 9:16 frame cannot show both at a readable
251
+ * size. A 16:9 export has room the vertical frame never had, and §54 gives it
252
+ * layouts that USE that room instead of blurring it away: a broadcast
253
+ * `lower-third` (picture whole, card in the bottom band) and `split-left`/
254
+ * `split-right` (speaker fills one half, graphic the other — two layouts
255
+ * because the side matters: eyeline and the room's contents differ left vs
256
+ * right, and the producer or the editor should be able to pick).
257
+ */
258
+ export const LANDSCAPE_LAYOUTS: readonly Layout[] = [
259
+ "full-bleed",
260
+ "blurred-behind",
261
+ "lower-third",
262
+ "split-left",
263
+ "split-right",
264
+ ];
265
+
266
+ /**
267
+ * Nearest landscape-appropriate layout. Identity for the five that qualify;
268
+ * the vertical splits map to their landscape analog — `video-top` and
269
+ * `pip-bubble` both mean "face and graphic share the frame", which in 16:9 is
270
+ * a side-by-side split, not a blur — and `graphic-only` (the graphic IS the
271
+ * subject) keeps `blurred-behind`, the layout that lets a card own the frame
272
+ * without discarding the picture.
273
+ */
274
+ export function landscapeLayout(layout: Layout): Layout {
275
+ if (LANDSCAPE_LAYOUTS.includes(layout)) return layout;
276
+ return layout === "graphic-only" ? "blurred-behind" : "split-left";
277
+ }
@@ -0,0 +1,130 @@
1
+ import type { Scene } from "./scene-schema";
2
+ import type { Transcript } from "./schema";
3
+ import { soundsSimilar } from "./phonetics";
4
+
5
+ /**
6
+ * Copy-grounding post-check (FINDINGS §14a): label-ish props must reuse
7
+ * nouns the take actually contains. The producer prompt demands grounding in
8
+ * the moment's slice; this check is deliberately looser — it flags a token
9
+ * only when it appears NOWHERE in the transcript — so what it does flag is a
10
+ * high-confidence hallucination ("REVENUE" on a code-churn stat), visible in
11
+ * the report without watching the video.
12
+ *
13
+ * Run this against the REPAIRED transcript, never the raw one. §17 was this
14
+ * check fighting the mishearing mitigation: the producer correctly wrote
15
+ * "code churn" for a take transcribed as "coach and", and the check reported
16
+ * the repair as an invention. Repairing once, up front, removes the conflict
17
+ * at its source — which is why no phonetic tolerance lives here. Tolerance
18
+ * was tried and rejected: matching copy against the take by sound absolves
19
+ * real hallucinations too ("CODECHUN REVENUE" reads as a repair of "code
20
+ * churn"), and a second mitigation pulling against the first is exactly what
21
+ * produced §17.
22
+ */
23
+
24
+ export interface GroundingIssue {
25
+ sceneId: string;
26
+ component: string;
27
+ field: string;
28
+ token: string;
29
+ }
30
+
31
+ /**
32
+ * Fields that carry factual labels, per component. Conversational/stylized
33
+ * copy (ChatMock messages, TerminalMock lines) is invented by design and is
34
+ * not checked.
35
+ */
36
+ const CHECKED_FIELDS: Record<string, string[]> = {
37
+ TitleCard: ["eyebrow", "title", "sub"],
38
+ StatCard: ["label", "caption"],
39
+ RuleCard: ["kicker", "text", "struck"],
40
+ StrikethroughReveal: ["lines"],
41
+ FlowDiagram: ["nodes"],
42
+ ScreenshotFrame: ["label"],
43
+ };
44
+
45
+ /**
46
+ * Function words carry no factual claim, so flagging them is pure noise —
47
+ * this check is only worth anything if it is precise (FINDINGS §30, where
48
+ * `caption "but"` was reported as ungrounded copy). Kept to grammar and
49
+ * generic connectives: domain nouns stay checkable, because inventing one is
50
+ * exactly the failure this exists to catch.
51
+ */
52
+ const STOPWORDS = new Set([
53
+ "the", "a", "an", "of", "to", "in", "on", "at", "for", "and", "or", "vs",
54
+ "no", "not", "non", "per", "than", "then", "with", "without", "into",
55
+ "over", "under", "more", "less", "most", "least", "your", "our", "my",
56
+ "his", "her", "their", "its", "it", "is", "are", "was", "be", "this",
57
+ "that", "these", "those", "you", "we", "they", "one", "two", "all",
58
+ "every", "each", "when", "how", "why", "what", "now", "today", "rule",
59
+ "step", "do", "dont", "done",
60
+ // §30: conjunctions, auxiliaries, prepositions and degree words.
61
+ "but", "so", "yet", "if", "else", "because", "while", "since", "until",
62
+ "from", "about", "after", "before", "during", "between", "through",
63
+ "up", "down", "off", "out", "away", "back", "here", "there", "again",
64
+ "just", "very", "really", "quite", "even", "still", "only", "also", "too",
65
+ "am", "were", "been", "being", "has", "have", "had", "can", "cant",
66
+ "will", "wont", "would", "should", "could", "may", "might", "must",
67
+ "get", "gets", "got", "let", "lets", "make", "makes", "made",
68
+ "some", "any", "many", "much", "few", "both", "own", "same", "other",
69
+ "another", "such", "which", "who", "whom", "whose", "where",
70
+ "always", "never", "often", "sometimes", "usually", "already", "soon",
71
+ "actually", "basically", "literally", "probably", "maybe", "perhaps",
72
+ ]);
73
+
74
+ function tokenize(s: string): string[] {
75
+ return s
76
+ .toLowerCase()
77
+ .split(/[^a-z0-9]+/)
78
+ .filter(Boolean);
79
+ }
80
+
81
+ /** Numbers, units and short glue don't need transcript support. */
82
+ function needsSupport(token: string): boolean {
83
+ if (token.length < 3) return false;
84
+ if (/\d/.test(token)) return false;
85
+ return !STOPWORDS.has(token);
86
+ }
87
+
88
+ function stringsOf(value: unknown): string[] {
89
+ if (typeof value === "string") return [value];
90
+ if (Array.isArray(value)) return value.flatMap(stringsOf);
91
+ if (value && typeof value === "object") return Object.values(value).flatMap(stringsOf);
92
+ return [];
93
+ }
94
+
95
+ export function checkGrounding(
96
+ scenes: readonly Scene[],
97
+ transcript: Transcript,
98
+ /**
99
+ * Who the speaker is (`--speaker`). Their own name and brand are legitimate
100
+ * on screen even when the recognizer mangled every utterance of them, so the
101
+ * hint counts as spoken vocabulary — otherwise the check fights the repair
102
+ * pass, which is the §17 mistake in a new place.
103
+ */
104
+ speaker?: string,
105
+ ): GroundingIssue[] {
106
+ const spoken = new Set([
107
+ ...transcript.words.flatMap((w) => tokenize(w.text)),
108
+ ...(speaker ? tokenize(speaker) : []),
109
+ ]);
110
+ const supported = (token: string): boolean =>
111
+ spoken.has(token) ||
112
+ spoken.has(`${token}s`) ||
113
+ (token.endsWith("s") && spoken.has(token.slice(0, -1)));
114
+
115
+ const issues: GroundingIssue[] = [];
116
+ for (const scene of scenes) {
117
+ const fields = CHECKED_FIELDS[scene.component] ?? [];
118
+ const merged = { ...scene.props, ...scene.overrides };
119
+ for (const field of fields) {
120
+ for (const text of stringsOf(merged[field])) {
121
+ for (const token of tokenize(text)) {
122
+ if (needsSupport(token) && !supported(token)) {
123
+ issues.push({ sceneId: scene.id, component: scene.component, field, token });
124
+ }
125
+ }
126
+ }
127
+ }
128
+ }
129
+ return issues;
130
+ }
package/src/index.ts ADDED
@@ -0,0 +1,27 @@
1
+ export * from "./schema";
2
+ export * from "./scene-schema";
3
+ export * from "./overrides";
4
+ export * from "./scene-registry";
5
+ export * from "./assemble";
6
+ export * from "./fill";
7
+ export * from "./producer/index";
8
+ export * from "./timemap";
9
+ export * from "./ingest";
10
+ export * from "./transcribe";
11
+ export * from "./analyze";
12
+ export * from "./cutlist";
13
+ export * from "./clip";
14
+ export * from "./captions";
15
+ export * from "./zoom";
16
+ export * from "./grounding";
17
+ export * from "./cta";
18
+ export * from "./content-rect";
19
+ export * from "./content-rect-detect";
20
+ export * from "./normalize";
21
+ export * from "./framing";
22
+ export * from "./face";
23
+ export * from "./cover";
24
+ export * from "./source-text";
25
+ export * from "./report";
26
+ export * from "./config";
27
+ export { run } from "./exec";
package/src/ingest.ts ADDED
@@ -0,0 +1,83 @@
1
+ import { run } from "./exec";
2
+ import type { Probe } from "./schema";
3
+
4
+ export interface IngestTools {
5
+ ffmpegPath: string;
6
+ ffprobePath: string;
7
+ }
8
+
9
+ export async function probe(tools: IngestTools, path: string): Promise<Probe> {
10
+ const { stdout } = await run(tools.ffprobePath, [
11
+ "-v", "error",
12
+ "-print_format", "json",
13
+ "-show_streams",
14
+ "-show_format",
15
+ path,
16
+ ]);
17
+ const info = JSON.parse(stdout) as {
18
+ streams?: Array<{
19
+ codec_type?: string;
20
+ width?: number;
21
+ height?: number;
22
+ avg_frame_rate?: string;
23
+ r_frame_rate?: string;
24
+ }>;
25
+ format?: { duration?: string };
26
+ };
27
+ const video = info.streams?.find((s) => s.codec_type === "video");
28
+ const audio = info.streams?.find((s) => s.codec_type === "audio");
29
+ if (!video) throw new Error(`no video stream in ${path}`);
30
+ const rate = video.avg_frame_rate && video.avg_frame_rate !== "0/0" ? video.avg_frame_rate : video.r_frame_rate;
31
+ const [num, den] = (rate ?? "30/1").split("/").map(Number);
32
+ const duration = Number(info.format?.duration);
33
+ if (!Number.isFinite(duration) || duration <= 0) throw new Error(`could not determine duration of ${path}`);
34
+ return {
35
+ duration,
36
+ width: video.width ?? 0,
37
+ height: video.height ?? 0,
38
+ fps: den ? (num ?? 30) / den : 30,
39
+ hasAudio: Boolean(audio),
40
+ };
41
+ }
42
+
43
+ /** Extract 16 kHz mono PCM audio for ASR + silence analysis. */
44
+ export async function extractAudio(tools: IngestTools, src: string, outWav: string): Promise<void> {
45
+ await run(tools.ffmpegPath, [
46
+ "-y", "-i", src,
47
+ "-vn", "-ar", "16000", "-ac", "1", "-c:a", "pcm_s16le",
48
+ outWav,
49
+ ]);
50
+ }
51
+
52
+ /**
53
+ * Re-encode with dense keyframes so EDL playback (<OffthreadVideo> with many
54
+ * small trims) seeks fast. Optional — most sources play fine untouched —
55
+ * EXCEPT when the source is letterboxed: then this pass also trims the baked
56
+ * bars (`crop`), so everything downstream sees the picture, not picture+bars
57
+ * (PLAN Task 7), and the pass stops being optional.
58
+ */
59
+ export async function makeMezzanine(
60
+ tools: IngestTools,
61
+ src: string,
62
+ out: string,
63
+ opts: { cropVf?: string } = {},
64
+ ): Promise<void> {
65
+ await run(tools.ffmpegPath, [
66
+ "-y", "-i", src,
67
+ ...(opts.cropVf ? ["-vf", opts.cropVf] : []),
68
+ "-c:v", "libx264", "-preset", "veryfast", "-crf", "18", "-g", "30", "-pix_fmt", "yuv420p",
69
+ "-c:a", "aac", "-b:a", "192k",
70
+ out,
71
+ ]);
72
+ }
73
+
74
+ /** EBU R128 loudness normalization post-pass; video stream is copied untouched. */
75
+ export async function loudnorm(tools: IngestTools, src: string, out: string): Promise<void> {
76
+ await run(tools.ffmpegPath, [
77
+ "-y", "-i", src,
78
+ "-c:v", "copy",
79
+ "-af", "loudnorm=I=-16:TP=-1.5:LRA=11",
80
+ "-c:a", "aac", "-b:a", "192k",
81
+ out,
82
+ ]);
83
+ }