@ossclip/core 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +27 -0
- package/README.md +20 -0
- package/package.json +29 -0
- package/src/analyze.ts +299 -0
- package/src/assemble.ts +124 -0
- package/src/browser.ts +24 -0
- package/src/captions.ts +92 -0
- package/src/clip.ts +306 -0
- package/src/config.ts +66 -0
- package/src/content-rect-detect.ts +162 -0
- package/src/content-rect.ts +324 -0
- package/src/cover.ts +216 -0
- package/src/cta.ts +68 -0
- package/src/cutlist.ts +170 -0
- package/src/exec.ts +36 -0
- package/src/face.ts +519 -0
- package/src/fill.ts +110 -0
- package/src/framing.ts +277 -0
- package/src/grounding.ts +130 -0
- package/src/index.ts +27 -0
- package/src/ingest.ts +83 -0
- package/src/normalize.ts +397 -0
- package/src/overrides.ts +509 -0
- package/src/phonetics.ts +129 -0
- package/src/producer/anthropic.ts +73 -0
- package/src/producer/beats.ts +330 -0
- package/src/producer/claude-cli.ts +150 -0
- package/src/producer/gemini.ts +197 -0
- package/src/producer/index.ts +217 -0
- package/src/producer/mock.ts +101 -0
- package/src/producer/provider.ts +42 -0
- package/src/producer/repair.ts +474 -0
- package/src/producer/scene-props.ts +212 -0
- package/src/producer/tiered.ts +56 -0
- package/src/producer/usage.ts +426 -0
- package/src/report.ts +36 -0
- package/src/scene-registry.ts +246 -0
- package/src/scene-schema.ts +203 -0
- package/src/schema.ts +177 -0
- package/src/source-text.ts +348 -0
- package/src/timemap.ts +115 -0
- package/src/transcribe.ts +67 -0
- package/src/zoom.ts +154 -0
package/src/framing.ts
ADDED
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
import type { Transcript } from "./schema";
|
|
2
|
+
import { SCENE_REGISTRY } from "./scene-registry";
|
|
3
|
+
import type { Layout } from "./scene-schema";
|
|
4
|
+
import type { Moment } from "./producer/beats";
|
|
5
|
+
import { headFracInSlot } from "./normalize";
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Producer framing awareness (PLAN 2026-07-28, Tasks A and B).
|
|
9
|
+
*
|
|
10
|
+
* The beat-sheet prompt used to contain zero geometry — intent, duration,
|
|
11
|
+
* component menu, transcript. So the producer happily put `video-top` (a wide
|
|
12
|
+
* band) on moments where the speaker fills 44% of a portrait canvas, and the
|
|
13
|
+
* band cover-cropped the head to 206% of its height. Three rounds of tuning a
|
|
14
|
+
* global crop constant could not fix that, because it is not a cropping
|
|
15
|
+
* problem: the pixels a wide band wants do not exist in a portrait close-up.
|
|
16
|
+
* The fix is editorial — don't put a wide band on a close-up moment — which
|
|
17
|
+
* makes it the PRODUCER's decision, and this module is how the producer gets
|
|
18
|
+
* the evidence (Task A) and how its choice is checked (Task B).
|
|
19
|
+
*
|
|
20
|
+
* Two design rules, from the plan:
|
|
21
|
+
* - The model never does pixel math. The brief is qualitative — CLOSE /
|
|
22
|
+
* MEDIUM / WIDE plus which layouts that rules out. The arithmetic stays
|
|
23
|
+
* in `headFracInSlot`, where it is tested.
|
|
24
|
+
* - The repair pass ships regardless of how well the prompt behaves. A
|
|
25
|
+
* prompt is a request; `repairMomentLayouts` is the constraint (the §35
|
|
26
|
+
* lesson).
|
|
27
|
+
*/
|
|
28
|
+
|
|
29
|
+
/** One framing window in SOURCE time, from the normalization plan. */
|
|
30
|
+
export interface FramingWindow {
|
|
31
|
+
startSec: number;
|
|
32
|
+
endSec: number;
|
|
33
|
+
/** Face height as a fraction of the (normalized) canvas during this window. */
|
|
34
|
+
faceFracOfCanvas: number;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* A layout's video-slot shape. Injected by the CLI from `layoutSlots` —
|
|
39
|
+
* core stays React- and scenes-free, and the feasibility question needs only
|
|
40
|
+
* the slot's ASPECT and whether the video is the subject there at all.
|
|
41
|
+
*/
|
|
42
|
+
export interface LayoutFraming {
|
|
43
|
+
layout: Layout;
|
|
44
|
+
/** Slot width over height, in output pixels. */
|
|
45
|
+
slotAspect: number;
|
|
46
|
+
/**
|
|
47
|
+
* False for slots where the head-fits rule does not apply: a pip bubble is
|
|
48
|
+
* MEANT to be a tight head shot, and a `graphic-only` slot never draws.
|
|
49
|
+
*/
|
|
50
|
+
primary: boolean;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/** Everything a framing judgement needs, assembled once by the CLI. */
|
|
54
|
+
export interface FramingContext {
|
|
55
|
+
windows: FramingWindow[];
|
|
56
|
+
/** The normalized canvas's width over height. */
|
|
57
|
+
canvasAspect: number;
|
|
58
|
+
layouts: LayoutFraming[];
|
|
59
|
+
/** The idle zoom's peak — the head must fit at the tightest moment. */
|
|
60
|
+
zoom: number;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/** The moment's span in SOURCE seconds, off the transcript's own stamps. */
|
|
64
|
+
export function momentSourceWindow(
|
|
65
|
+
transcript: Transcript,
|
|
66
|
+
startWord: number,
|
|
67
|
+
endWord: number,
|
|
68
|
+
): { startSec: number; endSec: number } | null {
|
|
69
|
+
const first = transcript.words[startWord];
|
|
70
|
+
const last = transcript.words[endWord];
|
|
71
|
+
if (!first || !last) return null;
|
|
72
|
+
return { startSec: first.start, endSec: last.end };
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** The tightest framing inside [startSec, endSec] — a span is judged by its
|
|
76
|
+
* worst moment, not its average. */
|
|
77
|
+
export function worstFaceFrac(
|
|
78
|
+
windows: readonly FramingWindow[],
|
|
79
|
+
startSec: number,
|
|
80
|
+
endSec: number,
|
|
81
|
+
): number {
|
|
82
|
+
let frac = 0;
|
|
83
|
+
for (const w of windows) {
|
|
84
|
+
if (w.startSec < endSec && w.endSec > startSec) frac = Math.max(frac, w.faceFracOfCanvas);
|
|
85
|
+
}
|
|
86
|
+
return frac;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/** Can this layout's slot hold the whole head at this framing? Non-primary
|
|
90
|
+
* slots are always feasible — the rule does not apply to them. */
|
|
91
|
+
export function layoutFeasible(ctx: FramingContext, layout: Layout, faceFrac: number): boolean {
|
|
92
|
+
const entry = ctx.layouts.find((l) => l.layout === layout);
|
|
93
|
+
if (!entry || !entry.primary || faceFrac <= 0) return true;
|
|
94
|
+
return headFracInSlot(faceFrac, ctx.canvasAspect, entry.slotAspect, ctx.zoom) <= 1;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
/** The primary layouts this framing rules out, worst offender first. */
|
|
98
|
+
export function infeasibleLayouts(ctx: FramingContext, faceFrac: number): Layout[] {
|
|
99
|
+
return ctx.layouts
|
|
100
|
+
.filter((l) => !layoutFeasible(ctx, l.layout, faceFrac))
|
|
101
|
+
.sort(
|
|
102
|
+
(a, b) =>
|
|
103
|
+
headFracInSlot(faceFrac, ctx.canvasAspect, b.slotAspect, ctx.zoom) -
|
|
104
|
+
headFracInSlot(faceFrac, ctx.canvasAspect, a.slotAspect, ctx.zoom),
|
|
105
|
+
)
|
|
106
|
+
.map((l) => l.layout);
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/** Qualitative label for the brief. CLOSE is defined by CONSEQUENCE — some
|
|
110
|
+
* layout is unavailable — not by an arbitrary fraction threshold. */
|
|
111
|
+
function shotLabel(ctx: FramingContext, faceFrac: number): "CLOSE" | "MEDIUM" | "WIDE" {
|
|
112
|
+
if (infeasibleLayouts(ctx, faceFrac).length > 0) return "CLOSE";
|
|
113
|
+
return faceFrac >= 0.3 ? "MEDIUM" : "WIDE";
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/**
|
|
117
|
+
* The framing brief for the beat-sheet prompt (Task A).
|
|
118
|
+
*
|
|
119
|
+
* One line per framing stretch, in WORD INDICES — the producer reasons in
|
|
120
|
+
* word spans and never sees a second. Adjacent windows whose constraint is
|
|
121
|
+
* identical merge, so a ten-segment source with two real framings reads as a
|
|
122
|
+
* handful of lines, not a table. Windows with no face measurement carry no
|
|
123
|
+
* signal and produce no line; a source with no measured window at all returns
|
|
124
|
+
* "" and the prompt is unchanged.
|
|
125
|
+
*/
|
|
126
|
+
export function buildFramingBrief(ctx: FramingContext, transcript: Transcript): string {
|
|
127
|
+
if (transcript.words.length === 0) return "";
|
|
128
|
+
|
|
129
|
+
// Word range per window: a word belongs to the window containing its middle.
|
|
130
|
+
const mid = (i: number): number => {
|
|
131
|
+
const w = transcript.words[i]!;
|
|
132
|
+
return (w.start + w.end) / 2;
|
|
133
|
+
};
|
|
134
|
+
type Line = { fromWord: number; toWord: number; faceFrac: number; avoid: Layout[] };
|
|
135
|
+
const lines: Line[] = [];
|
|
136
|
+
for (const win of ctx.windows) {
|
|
137
|
+
if (win.faceFracOfCanvas <= 0) continue;
|
|
138
|
+
let from = -1;
|
|
139
|
+
let to = -1;
|
|
140
|
+
for (let i = 0; i < transcript.words.length; i++) {
|
|
141
|
+
if (mid(i) >= win.startSec && mid(i) < win.endSec) {
|
|
142
|
+
if (from === -1) from = i;
|
|
143
|
+
to = i;
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
if (from === -1) continue;
|
|
147
|
+
const avoid = infeasibleLayouts(ctx, win.faceFracOfCanvas);
|
|
148
|
+
const prev = lines[lines.length - 1];
|
|
149
|
+
if (prev && prev.toWord === from - 1 && sameAvoid(prev.avoid, avoid)) {
|
|
150
|
+
// Same constraint, contiguous words: one line. The face fraction kept is
|
|
151
|
+
// the WORST, consistent with how spans are judged everywhere else.
|
|
152
|
+
prev.toWord = to;
|
|
153
|
+
prev.faceFrac = Math.max(prev.faceFrac, win.faceFracOfCanvas);
|
|
154
|
+
} else {
|
|
155
|
+
lines.push({ fromWord: from, toWord: to, faceFrac: win.faceFracOfCanvas, avoid });
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
if (lines.length === 0) return "";
|
|
159
|
+
|
|
160
|
+
const body = lines
|
|
161
|
+
.map((l) => {
|
|
162
|
+
const label = shotLabel(ctx, l.faceFrac);
|
|
163
|
+
const detail = `the face fills ~${Math.round(l.faceFrac * 100)}% of the frame height`;
|
|
164
|
+
return l.avoid.length > 0
|
|
165
|
+
? `- words ${l.fromWord}-${l.toWord}: ${label} shot (${detail}) — layouts ` +
|
|
166
|
+
`${l.avoid.join(", ")} would crop the head here and are UNAVAILABLE`
|
|
167
|
+
: `- words ${l.fromWord}-${l.toWord}: ${label} shot (${detail}) — any layout works`;
|
|
168
|
+
})
|
|
169
|
+
.join("\n");
|
|
170
|
+
return (
|
|
171
|
+
"Camera framing by word range (measured from the footage — hard constraints, not suggestions):\n" +
|
|
172
|
+
body
|
|
173
|
+
);
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
function sameAvoid(a: readonly Layout[], b: readonly Layout[]): boolean {
|
|
177
|
+
return a.length === b.length && a.every((l, i) => l === b[i]);
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
export interface LayoutRepair {
|
|
181
|
+
/** Index into the moments array. */
|
|
182
|
+
moment: number;
|
|
183
|
+
issue: string;
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
/**
|
|
187
|
+
* The Task B safety net: rewrite any moment whose layout cannot physically
|
|
188
|
+
* hold the head at that moment's framing.
|
|
189
|
+
*
|
|
190
|
+
* Runs AFTER `normalizeBeatSheet` (word spans are already valid) and before
|
|
191
|
+
* scene generation. The producer's explicit choice is kept when feasible;
|
|
192
|
+
* an infeasible one — or an infeasible registry default when the producer
|
|
193
|
+
* said nothing — is swapped for the first feasible layout among the
|
|
194
|
+
* component's default and alternates. When NOTHING is feasible the least-bad
|
|
195
|
+
* candidate is taken and the issue says so, because a scene that trims a
|
|
196
|
+
* little is still better than silently rendering the worst option.
|
|
197
|
+
*/
|
|
198
|
+
export function repairMomentLayouts(
|
|
199
|
+
moments: readonly Moment[],
|
|
200
|
+
transcript: Transcript,
|
|
201
|
+
ctx: FramingContext,
|
|
202
|
+
): { moments: Moment[]; issues: LayoutRepair[] } {
|
|
203
|
+
const issues: LayoutRepair[] = [];
|
|
204
|
+
const out = moments.map((m, idx) => {
|
|
205
|
+
if (m.sceneKind === "none") return m;
|
|
206
|
+
const meta = SCENE_REGISTRY[m.sceneKind];
|
|
207
|
+
const window = momentSourceWindow(transcript, m.startWord, m.endWord);
|
|
208
|
+
if (!meta || !window) return m;
|
|
209
|
+
const faceFrac = worstFaceFrac(ctx.windows, window.startSec, window.endSec);
|
|
210
|
+
if (faceFrac <= 0) return m;
|
|
211
|
+
|
|
212
|
+
const requested = m.layout ?? meta.defaultLayout;
|
|
213
|
+
if (layoutFeasible(ctx, requested, faceFrac)) return m;
|
|
214
|
+
|
|
215
|
+
const candidates = [...new Set<Layout>([meta.defaultLayout, ...meta.altLayouts])];
|
|
216
|
+
const feasible = candidates.filter((l) => layoutFeasible(ctx, l, faceFrac));
|
|
217
|
+
if (feasible.length > 0) {
|
|
218
|
+
issues.push({
|
|
219
|
+
moment: idx,
|
|
220
|
+
issue: `layout ${requested} would crop the head on this close shot; using ${feasible[0]}`,
|
|
221
|
+
});
|
|
222
|
+
return { ...m, layout: feasible[0]! };
|
|
223
|
+
}
|
|
224
|
+
// Nothing fits: take the candidate that trims least, and say so.
|
|
225
|
+
const leastBad = candidates.reduce((a, b) => {
|
|
226
|
+
const fr = (l: Layout): number => {
|
|
227
|
+
const entry = ctx.layouts.find((e) => e.layout === l);
|
|
228
|
+
return entry && entry.primary
|
|
229
|
+
? headFracInSlot(faceFrac, ctx.canvasAspect, entry.slotAspect, ctx.zoom)
|
|
230
|
+
: 0;
|
|
231
|
+
};
|
|
232
|
+
return fr(b) < fr(a) ? b : a;
|
|
233
|
+
});
|
|
234
|
+
issues.push({
|
|
235
|
+
moment: idx,
|
|
236
|
+
issue:
|
|
237
|
+
`no ${m.sceneKind} layout fully fits the head on this close shot; ` +
|
|
238
|
+
`${leastBad} trims least`,
|
|
239
|
+
});
|
|
240
|
+
return { ...m, layout: leastBad };
|
|
241
|
+
});
|
|
242
|
+
return { moments: out, issues };
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
/**
|
|
246
|
+
* Layouts that make sense in a LANDSCAPE frame (R15).
|
|
247
|
+
*
|
|
248
|
+
* `video-top`, `pip-bubble` and `graphic-only` are vertical-format ideas: they
|
|
249
|
+
* exist to split a tall frame between a face and a card, or to demote the
|
|
250
|
+
* speaker to an inset because a 9:16 frame cannot show both at a readable
|
|
251
|
+
* size. A 16:9 export has room the vertical frame never had, and §54 gives it
|
|
252
|
+
* layouts that USE that room instead of blurring it away: a broadcast
|
|
253
|
+
* `lower-third` (picture whole, card in the bottom band) and `split-left`/
|
|
254
|
+
* `split-right` (speaker fills one half, graphic the other — two layouts
|
|
255
|
+
* because the side matters: eyeline and the room's contents differ left vs
|
|
256
|
+
* right, and the producer or the editor should be able to pick).
|
|
257
|
+
*/
|
|
258
|
+
export const LANDSCAPE_LAYOUTS: readonly Layout[] = [
|
|
259
|
+
"full-bleed",
|
|
260
|
+
"blurred-behind",
|
|
261
|
+
"lower-third",
|
|
262
|
+
"split-left",
|
|
263
|
+
"split-right",
|
|
264
|
+
];
|
|
265
|
+
|
|
266
|
+
/**
|
|
267
|
+
* Nearest landscape-appropriate layout. Identity for the five that qualify;
|
|
268
|
+
* the vertical splits map to their landscape analog — `video-top` and
|
|
269
|
+
* `pip-bubble` both mean "face and graphic share the frame", which in 16:9 is
|
|
270
|
+
* a side-by-side split, not a blur — and `graphic-only` (the graphic IS the
|
|
271
|
+
* subject) keeps `blurred-behind`, the layout that lets a card own the frame
|
|
272
|
+
* without discarding the picture.
|
|
273
|
+
*/
|
|
274
|
+
export function landscapeLayout(layout: Layout): Layout {
|
|
275
|
+
if (LANDSCAPE_LAYOUTS.includes(layout)) return layout;
|
|
276
|
+
return layout === "graphic-only" ? "blurred-behind" : "split-left";
|
|
277
|
+
}
|
package/src/grounding.ts
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
import type { Scene } from "./scene-schema";
|
|
2
|
+
import type { Transcript } from "./schema";
|
|
3
|
+
import { soundsSimilar } from "./phonetics";
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Copy-grounding post-check (FINDINGS §14a): label-ish props must reuse
|
|
7
|
+
* nouns the take actually contains. The producer prompt demands grounding in
|
|
8
|
+
* the moment's slice; this check is deliberately looser — it flags a token
|
|
9
|
+
* only when it appears NOWHERE in the transcript — so what it does flag is a
|
|
10
|
+
* high-confidence hallucination ("REVENUE" on a code-churn stat), visible in
|
|
11
|
+
* the report without watching the video.
|
|
12
|
+
*
|
|
13
|
+
* Run this against the REPAIRED transcript, never the raw one. §17 was this
|
|
14
|
+
* check fighting the mishearing mitigation: the producer correctly wrote
|
|
15
|
+
* "code churn" for a take transcribed as "coach and", and the check reported
|
|
16
|
+
* the repair as an invention. Repairing once, up front, removes the conflict
|
|
17
|
+
* at its source — which is why no phonetic tolerance lives here. Tolerance
|
|
18
|
+
* was tried and rejected: matching copy against the take by sound absolves
|
|
19
|
+
* real hallucinations too ("CODECHUN REVENUE" reads as a repair of "code
|
|
20
|
+
* churn"), and a second mitigation pulling against the first is exactly what
|
|
21
|
+
* produced §17.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
export interface GroundingIssue {
|
|
25
|
+
sceneId: string;
|
|
26
|
+
component: string;
|
|
27
|
+
field: string;
|
|
28
|
+
token: string;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* Fields that carry factual labels, per component. Conversational/stylized
|
|
33
|
+
* copy (ChatMock messages, TerminalMock lines) is invented by design and is
|
|
34
|
+
* not checked.
|
|
35
|
+
*/
|
|
36
|
+
const CHECKED_FIELDS: Record<string, string[]> = {
|
|
37
|
+
TitleCard: ["eyebrow", "title", "sub"],
|
|
38
|
+
StatCard: ["label", "caption"],
|
|
39
|
+
RuleCard: ["kicker", "text", "struck"],
|
|
40
|
+
StrikethroughReveal: ["lines"],
|
|
41
|
+
FlowDiagram: ["nodes"],
|
|
42
|
+
ScreenshotFrame: ["label"],
|
|
43
|
+
};
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Function words carry no factual claim, so flagging them is pure noise —
|
|
47
|
+
* this check is only worth anything if it is precise (FINDINGS §30, where
|
|
48
|
+
* `caption "but"` was reported as ungrounded copy). Kept to grammar and
|
|
49
|
+
* generic connectives: domain nouns stay checkable, because inventing one is
|
|
50
|
+
* exactly the failure this exists to catch.
|
|
51
|
+
*/
|
|
52
|
+
const STOPWORDS = new Set([
|
|
53
|
+
"the", "a", "an", "of", "to", "in", "on", "at", "for", "and", "or", "vs",
|
|
54
|
+
"no", "not", "non", "per", "than", "then", "with", "without", "into",
|
|
55
|
+
"over", "under", "more", "less", "most", "least", "your", "our", "my",
|
|
56
|
+
"his", "her", "their", "its", "it", "is", "are", "was", "be", "this",
|
|
57
|
+
"that", "these", "those", "you", "we", "they", "one", "two", "all",
|
|
58
|
+
"every", "each", "when", "how", "why", "what", "now", "today", "rule",
|
|
59
|
+
"step", "do", "dont", "done",
|
|
60
|
+
// §30: conjunctions, auxiliaries, prepositions and degree words.
|
|
61
|
+
"but", "so", "yet", "if", "else", "because", "while", "since", "until",
|
|
62
|
+
"from", "about", "after", "before", "during", "between", "through",
|
|
63
|
+
"up", "down", "off", "out", "away", "back", "here", "there", "again",
|
|
64
|
+
"just", "very", "really", "quite", "even", "still", "only", "also", "too",
|
|
65
|
+
"am", "were", "been", "being", "has", "have", "had", "can", "cant",
|
|
66
|
+
"will", "wont", "would", "should", "could", "may", "might", "must",
|
|
67
|
+
"get", "gets", "got", "let", "lets", "make", "makes", "made",
|
|
68
|
+
"some", "any", "many", "much", "few", "both", "own", "same", "other",
|
|
69
|
+
"another", "such", "which", "who", "whom", "whose", "where",
|
|
70
|
+
"always", "never", "often", "sometimes", "usually", "already", "soon",
|
|
71
|
+
"actually", "basically", "literally", "probably", "maybe", "perhaps",
|
|
72
|
+
]);
|
|
73
|
+
|
|
74
|
+
function tokenize(s: string): string[] {
|
|
75
|
+
return s
|
|
76
|
+
.toLowerCase()
|
|
77
|
+
.split(/[^a-z0-9]+/)
|
|
78
|
+
.filter(Boolean);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** Numbers, units and short glue don't need transcript support. */
|
|
82
|
+
function needsSupport(token: string): boolean {
|
|
83
|
+
if (token.length < 3) return false;
|
|
84
|
+
if (/\d/.test(token)) return false;
|
|
85
|
+
return !STOPWORDS.has(token);
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
function stringsOf(value: unknown): string[] {
|
|
89
|
+
if (typeof value === "string") return [value];
|
|
90
|
+
if (Array.isArray(value)) return value.flatMap(stringsOf);
|
|
91
|
+
if (value && typeof value === "object") return Object.values(value).flatMap(stringsOf);
|
|
92
|
+
return [];
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
export function checkGrounding(
|
|
96
|
+
scenes: readonly Scene[],
|
|
97
|
+
transcript: Transcript,
|
|
98
|
+
/**
|
|
99
|
+
* Who the speaker is (`--speaker`). Their own name and brand are legitimate
|
|
100
|
+
* on screen even when the recognizer mangled every utterance of them, so the
|
|
101
|
+
* hint counts as spoken vocabulary — otherwise the check fights the repair
|
|
102
|
+
* pass, which is the §17 mistake in a new place.
|
|
103
|
+
*/
|
|
104
|
+
speaker?: string,
|
|
105
|
+
): GroundingIssue[] {
|
|
106
|
+
const spoken = new Set([
|
|
107
|
+
...transcript.words.flatMap((w) => tokenize(w.text)),
|
|
108
|
+
...(speaker ? tokenize(speaker) : []),
|
|
109
|
+
]);
|
|
110
|
+
const supported = (token: string): boolean =>
|
|
111
|
+
spoken.has(token) ||
|
|
112
|
+
spoken.has(`${token}s`) ||
|
|
113
|
+
(token.endsWith("s") && spoken.has(token.slice(0, -1)));
|
|
114
|
+
|
|
115
|
+
const issues: GroundingIssue[] = [];
|
|
116
|
+
for (const scene of scenes) {
|
|
117
|
+
const fields = CHECKED_FIELDS[scene.component] ?? [];
|
|
118
|
+
const merged = { ...scene.props, ...scene.overrides };
|
|
119
|
+
for (const field of fields) {
|
|
120
|
+
for (const text of stringsOf(merged[field])) {
|
|
121
|
+
for (const token of tokenize(text)) {
|
|
122
|
+
if (needsSupport(token) && !supported(token)) {
|
|
123
|
+
issues.push({ sceneId: scene.id, component: scene.component, field, token });
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
return issues;
|
|
130
|
+
}
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
export * from "./schema";
|
|
2
|
+
export * from "./scene-schema";
|
|
3
|
+
export * from "./overrides";
|
|
4
|
+
export * from "./scene-registry";
|
|
5
|
+
export * from "./assemble";
|
|
6
|
+
export * from "./fill";
|
|
7
|
+
export * from "./producer/index";
|
|
8
|
+
export * from "./timemap";
|
|
9
|
+
export * from "./ingest";
|
|
10
|
+
export * from "./transcribe";
|
|
11
|
+
export * from "./analyze";
|
|
12
|
+
export * from "./cutlist";
|
|
13
|
+
export * from "./clip";
|
|
14
|
+
export * from "./captions";
|
|
15
|
+
export * from "./zoom";
|
|
16
|
+
export * from "./grounding";
|
|
17
|
+
export * from "./cta";
|
|
18
|
+
export * from "./content-rect";
|
|
19
|
+
export * from "./content-rect-detect";
|
|
20
|
+
export * from "./normalize";
|
|
21
|
+
export * from "./framing";
|
|
22
|
+
export * from "./face";
|
|
23
|
+
export * from "./cover";
|
|
24
|
+
export * from "./source-text";
|
|
25
|
+
export * from "./report";
|
|
26
|
+
export * from "./config";
|
|
27
|
+
export { run } from "./exec";
|
package/src/ingest.ts
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
import { run } from "./exec";
|
|
2
|
+
import type { Probe } from "./schema";
|
|
3
|
+
|
|
4
|
+
export interface IngestTools {
|
|
5
|
+
ffmpegPath: string;
|
|
6
|
+
ffprobePath: string;
|
|
7
|
+
}
|
|
8
|
+
|
|
9
|
+
export async function probe(tools: IngestTools, path: string): Promise<Probe> {
|
|
10
|
+
const { stdout } = await run(tools.ffprobePath, [
|
|
11
|
+
"-v", "error",
|
|
12
|
+
"-print_format", "json",
|
|
13
|
+
"-show_streams",
|
|
14
|
+
"-show_format",
|
|
15
|
+
path,
|
|
16
|
+
]);
|
|
17
|
+
const info = JSON.parse(stdout) as {
|
|
18
|
+
streams?: Array<{
|
|
19
|
+
codec_type?: string;
|
|
20
|
+
width?: number;
|
|
21
|
+
height?: number;
|
|
22
|
+
avg_frame_rate?: string;
|
|
23
|
+
r_frame_rate?: string;
|
|
24
|
+
}>;
|
|
25
|
+
format?: { duration?: string };
|
|
26
|
+
};
|
|
27
|
+
const video = info.streams?.find((s) => s.codec_type === "video");
|
|
28
|
+
const audio = info.streams?.find((s) => s.codec_type === "audio");
|
|
29
|
+
if (!video) throw new Error(`no video stream in ${path}`);
|
|
30
|
+
const rate = video.avg_frame_rate && video.avg_frame_rate !== "0/0" ? video.avg_frame_rate : video.r_frame_rate;
|
|
31
|
+
const [num, den] = (rate ?? "30/1").split("/").map(Number);
|
|
32
|
+
const duration = Number(info.format?.duration);
|
|
33
|
+
if (!Number.isFinite(duration) || duration <= 0) throw new Error(`could not determine duration of ${path}`);
|
|
34
|
+
return {
|
|
35
|
+
duration,
|
|
36
|
+
width: video.width ?? 0,
|
|
37
|
+
height: video.height ?? 0,
|
|
38
|
+
fps: den ? (num ?? 30) / den : 30,
|
|
39
|
+
hasAudio: Boolean(audio),
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/** Extract 16 kHz mono PCM audio for ASR + silence analysis. */
|
|
44
|
+
export async function extractAudio(tools: IngestTools, src: string, outWav: string): Promise<void> {
|
|
45
|
+
await run(tools.ffmpegPath, [
|
|
46
|
+
"-y", "-i", src,
|
|
47
|
+
"-vn", "-ar", "16000", "-ac", "1", "-c:a", "pcm_s16le",
|
|
48
|
+
outWav,
|
|
49
|
+
]);
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Re-encode with dense keyframes so EDL playback (<OffthreadVideo> with many
|
|
54
|
+
* small trims) seeks fast. Optional — most sources play fine untouched —
|
|
55
|
+
* EXCEPT when the source is letterboxed: then this pass also trims the baked
|
|
56
|
+
* bars (`crop`), so everything downstream sees the picture, not picture+bars
|
|
57
|
+
* (PLAN Task 7), and the pass stops being optional.
|
|
58
|
+
*/
|
|
59
|
+
export async function makeMezzanine(
|
|
60
|
+
tools: IngestTools,
|
|
61
|
+
src: string,
|
|
62
|
+
out: string,
|
|
63
|
+
opts: { cropVf?: string } = {},
|
|
64
|
+
): Promise<void> {
|
|
65
|
+
await run(tools.ffmpegPath, [
|
|
66
|
+
"-y", "-i", src,
|
|
67
|
+
...(opts.cropVf ? ["-vf", opts.cropVf] : []),
|
|
68
|
+
"-c:v", "libx264", "-preset", "veryfast", "-crf", "18", "-g", "30", "-pix_fmt", "yuv420p",
|
|
69
|
+
"-c:a", "aac", "-b:a", "192k",
|
|
70
|
+
out,
|
|
71
|
+
]);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** EBU R128 loudness normalization post-pass; video stream is copied untouched. */
|
|
75
|
+
export async function loudnorm(tools: IngestTools, src: string, out: string): Promise<void> {
|
|
76
|
+
await run(tools.ffmpegPath, [
|
|
77
|
+
"-y", "-i", src,
|
|
78
|
+
"-c:v", "copy",
|
|
79
|
+
"-af", "loudnorm=I=-16:TP=-1.5:LRA=11",
|
|
80
|
+
"-c:a", "aac", "-b:a", "192k",
|
|
81
|
+
out,
|
|
82
|
+
]);
|
|
83
|
+
}
|