@ossclip/core 0.1.4 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/blooper.ts +102 -0
- package/src/clip.ts +3 -2
- package/src/config.ts +37 -1
- package/src/content-rect.ts +13 -0
- package/src/cutlist.ts +68 -9
- package/src/grounding.ts +11 -1
- package/src/index.ts +1 -0
- package/src/ingest.ts +38 -2
- package/src/normalize.ts +32 -9
- package/src/producer/beats.ts +211 -15
- package/src/producer/index.ts +20 -2
- package/src/schema.ts +12 -0
package/package.json
CHANGED
package/src/blooper.ts
ADDED
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
import { normalizeToken } from "./analyze";
|
|
2
|
+
import { isSentenceStart } from "./clip";
|
|
3
|
+
import type { Transcript } from "./schema";
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Blooper removal by SPOKEN MARKER (R27 §122).
|
|
7
|
+
*
|
|
8
|
+
* The ask is "drop the flubbed takes, not just the silences", and the general
|
|
9
|
+
* form of that is semantic: a model reading the transcript deciding which
|
|
10
|
+
* attempt was bad. The authoring roadmap flags exactly why that is expensive —
|
|
11
|
+
* `buildCutlist` is today a pure function of (raw transcript, analysis,
|
|
12
|
+
* duration, level), and a model in that path ends the guarantee that the same
|
|
13
|
+
* input and `--cleanup` always produce the same edit.
|
|
14
|
+
*
|
|
15
|
+
* A marker the speaker says OUT LOUD is the deterministic subset. It needs no
|
|
16
|
+
* judgement: the word is in the transcript or it is not. So this ships the
|
|
17
|
+
* useful half of the feature and leaves the guarantee intact — the semantic
|
|
18
|
+
* detector remains unbuilt, deliberately.
|
|
19
|
+
*
|
|
20
|
+
* The pattern, from the take that motivated it:
|
|
21
|
+
*
|
|
22
|
+
* "That could be one of the cases where you can say" flubbed attempt
|
|
23
|
+
* "blooper." marker
|
|
24
|
+
* "That could be one of" flubbed again
|
|
25
|
+
* "blooper." marker
|
|
26
|
+
* "That could be the exit condition." the good take
|
|
27
|
+
*
|
|
28
|
+
* The marker TERMINATES a bad attempt, so removal runs backwards from it to
|
|
29
|
+
* the start of the sentence it spoiled, and consecutive marked attempts
|
|
30
|
+
* collapse into one cut.
|
|
31
|
+
*/
|
|
32
|
+
|
|
33
|
+
/** A span of transcript to drop, in word indices (inclusive) and source seconds. */
|
|
34
|
+
export interface BloopSpan {
|
|
35
|
+
startWord: number;
|
|
36
|
+
endWord: number;
|
|
37
|
+
startSec: number;
|
|
38
|
+
endSec: number;
|
|
39
|
+
/** How many marker words this span swallowed — 2+ means repeated attempts. */
|
|
40
|
+
markers: number;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Spans the speaker marked as bloopers.
|
|
45
|
+
*
|
|
46
|
+
* `marker` is matched with `normalizeToken`, the same normalizer the filler
|
|
47
|
+
* detector uses, so case and trailing punctuation do not matter — ASR writes
|
|
48
|
+
* the word as "blooper." with the period riding on it.
|
|
49
|
+
*
|
|
50
|
+
* Returns spans in transcript order, non-overlapping.
|
|
51
|
+
*/
|
|
52
|
+
export function findBloopSpans(transcript: Transcript, marker: string): BloopSpan[] {
|
|
53
|
+
const want = normalizeToken(marker);
|
|
54
|
+
if (!want) return [];
|
|
55
|
+
const words = transcript.words;
|
|
56
|
+
const isMarker = (i: number): boolean => normalizeToken(words[i]?.text ?? "") === want;
|
|
57
|
+
|
|
58
|
+
const spans: BloopSpan[] = [];
|
|
59
|
+
for (let i = 0; i < words.length; i++) {
|
|
60
|
+
if (!isMarker(i)) continue;
|
|
61
|
+
|
|
62
|
+
// Walk back over the attempt this marker spoiled, to the start of its
|
|
63
|
+
// sentence. The marker's own text usually ENDS a sentence ("blooper."), so
|
|
64
|
+
// the scan starts at the word before it.
|
|
65
|
+
let start = i;
|
|
66
|
+
while (start > 0 && !isSentenceStart(transcript, start)) start--;
|
|
67
|
+
|
|
68
|
+
// Consecutive attempts: if everything between the previous span's end and
|
|
69
|
+
// this attempt's start is already being dropped, merge rather than leave a
|
|
70
|
+
// one-word island of a sentence nobody finished.
|
|
71
|
+
const prev = spans[spans.length - 1];
|
|
72
|
+
let markers = 1;
|
|
73
|
+
if (prev && start <= prev.endWord + 1) {
|
|
74
|
+
spans.pop();
|
|
75
|
+
start = prev.startWord;
|
|
76
|
+
markers = prev.markers + 1;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
spans.push({
|
|
80
|
+
startWord: start,
|
|
81
|
+
endWord: i,
|
|
82
|
+
startSec: words[start]!.start,
|
|
83
|
+
endSec: words[i]!.end,
|
|
84
|
+
markers,
|
|
85
|
+
});
|
|
86
|
+
}
|
|
87
|
+
return spans;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* A one-line account per span, for `report.txt`. The cut report justifies every
|
|
92
|
+
* other removal; a cut this aggressive — whole sentences, not dead air — owes
|
|
93
|
+
* the user the words it took.
|
|
94
|
+
*/
|
|
95
|
+
export function formatBloopSpan(transcript: Transcript, span: BloopSpan): string {
|
|
96
|
+
const said = transcript.words
|
|
97
|
+
.slice(span.startWord, span.endWord + 1)
|
|
98
|
+
.map((w) => w.text)
|
|
99
|
+
.join(" ");
|
|
100
|
+
const attempts = span.markers > 1 ? ` (${span.markers} attempts)` : "";
|
|
101
|
+
return `"${said}"${attempts}`;
|
|
102
|
+
}
|
package/src/clip.ts
CHANGED
|
@@ -33,9 +33,10 @@ export const CLIP_SNAP_TOLERANCE = 0.2;
|
|
|
33
33
|
/** Word that closes a sentence — ASR punctuation rides on the word text. */
|
|
34
34
|
const SENTENCE_END = /[.!?…]["'")\]»]*$/u;
|
|
35
35
|
|
|
36
|
-
|
|
36
|
+
/** Exported since R27 §122: the blooper cut walks back to a sentence start. */
|
|
37
|
+
export const isSentenceEnd = (t: Transcript, i: number): boolean =>
|
|
37
38
|
SENTENCE_END.test(t.words[i]?.text ?? "");
|
|
38
|
-
const isSentenceStart = (t: Transcript, i: number): boolean =>
|
|
39
|
+
export const isSentenceStart = (t: Transcript, i: number): boolean =>
|
|
39
40
|
i === 0 || isSentenceEnd(t, i - 1);
|
|
40
41
|
|
|
41
42
|
const durSec = (t: Transcript, start: number, end: number): number =>
|
package/src/config.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { readFileSync } from "node:fs";
|
|
1
|
+
import { mkdirSync, readFileSync, writeFileSync } from "node:fs";
|
|
2
2
|
import { homedir } from "node:os";
|
|
3
3
|
import { join } from "node:path";
|
|
4
4
|
|
|
@@ -41,6 +41,42 @@ const DEFAULTS: OssclipConfig = {
|
|
|
41
41
|
model: "small.en",
|
|
42
42
|
};
|
|
43
43
|
|
|
44
|
+
/** Everything ossclip owns on disk lives under here: config.json, models/, bin/, .env. */
|
|
45
|
+
export const CONFIG_DIR = join(homedir(), ".ossclip");
|
|
46
|
+
|
|
47
|
+
export function configFilePath(baseDir: string = CONFIG_DIR): string {
|
|
48
|
+
return join(baseDir, "config.json");
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Merge a patch into `~/.ossclip/config.json`, creating it if needed.
|
|
53
|
+
*
|
|
54
|
+
* Read-merge-write over the RAW file, not a loaded OssclipConfig: users
|
|
55
|
+
* hand-edit this file, and keys the patch doesn't touch (`pricing`,
|
|
56
|
+
* `speaker`, comments-by-convention like `_note`) must survive a
|
|
57
|
+
* `ossclip setup` run untouched.
|
|
58
|
+
*/
|
|
59
|
+
export function saveConfigPatch(patch: Partial<OssclipConfig>, baseDir: string = CONFIG_DIR): string {
|
|
60
|
+
const path = configFilePath(baseDir);
|
|
61
|
+
let existing: Record<string, unknown> = {};
|
|
62
|
+
try {
|
|
63
|
+
existing = JSON.parse(readFileSync(path, "utf8")) as Record<string, unknown>;
|
|
64
|
+
} catch {
|
|
65
|
+
// absent or unparseable — a fresh object; setup never destroys a broken
|
|
66
|
+
// file silently, so keep a .bak when the file existed but didn't parse
|
|
67
|
+
try {
|
|
68
|
+
const raw = readFileSync(path, "utf8");
|
|
69
|
+
writeFileSync(`${path}.bak`, raw);
|
|
70
|
+
} catch {
|
|
71
|
+
// truly absent — nothing to back up
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
mkdirSync(baseDir, { recursive: true });
|
|
75
|
+
const merged = { ...existing, ...patch };
|
|
76
|
+
writeFileSync(path, `${JSON.stringify(merged, null, 2)}\n`);
|
|
77
|
+
return path;
|
|
78
|
+
}
|
|
79
|
+
|
|
44
80
|
/**
|
|
45
81
|
* Resolution order per key: env (OSSCLIP_FFMPEG, OSSCLIP_FFPROBE, OSSCLIP_WHISPER,
|
|
46
82
|
* OSSCLIP_MODEL_DIR, OSSCLIP_MODEL, OSSCLIP_BROWSER) → ~/.ossclip/config.json → defaults.
|
package/src/content-rect.ts
CHANGED
|
@@ -96,6 +96,19 @@ export function stableContentRect(
|
|
|
96
96
|
const whole: ContentRect = { x: 0, y: 0, w: width, h: height, full: true };
|
|
97
97
|
if (rects.length === 0 || width <= 0 || height <= 0) return whole;
|
|
98
98
|
|
|
99
|
+
// A measurement that does not FIT the frame was taken in a different
|
|
100
|
+
// coordinate space than the one we are reconciling it against, so nothing it
|
|
101
|
+
// says can be trusted (R27 §119). This is how a portrait take read as
|
|
102
|
+
// landscape used to produce a "letterbox": cropdetect measured the rotated
|
|
103
|
+
// 2160x3840 frame, the caller believed 3840x2160, and clamping the union of
|
|
104
|
+
// the two orientations yielded a 2160x2160 square that was never on screen.
|
|
105
|
+
// Refuse, exactly as MIN_CONTENT_FRAC refuses an implausibly small rect —
|
|
106
|
+
// cropping on bad evidence is far worse than leaving bars alone.
|
|
107
|
+
const fits = rects.every(
|
|
108
|
+
(r) => r.x >= 0 && r.y >= 0 && r.x + r.w <= width && r.y + r.h <= height,
|
|
109
|
+
);
|
|
110
|
+
if (!fits) return whole;
|
|
111
|
+
|
|
99
112
|
let left = Infinity;
|
|
100
113
|
let top = Infinity;
|
|
101
114
|
let right = -Infinity;
|
package/src/cutlist.ts
CHANGED
|
@@ -6,12 +6,23 @@ interface LevelPolicy {
|
|
|
6
6
|
/** Resulting gap length after tightening. */
|
|
7
7
|
tightenTo: number;
|
|
8
8
|
removeFillers: boolean;
|
|
9
|
+
/**
|
|
10
|
+
* Run-up kept before the first word and after the last (R27 §127).
|
|
11
|
+
*
|
|
12
|
+
* Level-dependent, because "how hard should I cut" plainly covers the ends
|
|
13
|
+
* too, and a fixed 0.25/0.35 was the one thing `--cleanup aggressive` could
|
|
14
|
+
* not tighten. On a short that LOOPS, a third of a second of the speaker
|
|
15
|
+
* sitting there after the last word is a visible dead beat every time the
|
|
16
|
+
* video repeats — the reason this surfaced on a real render.
|
|
17
|
+
*/
|
|
18
|
+
leadKeep: number;
|
|
19
|
+
tailKeep: number;
|
|
9
20
|
}
|
|
10
21
|
|
|
11
22
|
const POLICIES: Record<Exclude<CleanupLevel, "exact">, LevelPolicy> = {
|
|
12
|
-
light: { pauseMin: 1.2, tightenTo: 0.3, removeFillers: false },
|
|
13
|
-
standard: { pauseMin: 0.7, tightenTo: 0.22, removeFillers: true },
|
|
14
|
-
aggressive: { pauseMin: 0.5, tightenTo: 0.18, removeFillers: true },
|
|
23
|
+
light: { pauseMin: 1.2, tightenTo: 0.3, removeFillers: false, leadKeep: 0.35, tailKeep: 0.45 },
|
|
24
|
+
standard: { pauseMin: 0.7, tightenTo: 0.22, removeFillers: true, leadKeep: 0.25, tailKeep: 0.35 },
|
|
25
|
+
aggressive: { pauseMin: 0.5, tightenTo: 0.18, removeFillers: true, leadKeep: 0.12, tailKeep: 0.15 },
|
|
15
26
|
};
|
|
16
27
|
|
|
17
28
|
/** Dead air kept before the first word / after the last word. Exported for
|
|
@@ -49,10 +60,26 @@ export interface BuildCutlistArgs {
|
|
|
49
60
|
analysis: Analysis;
|
|
50
61
|
duration: number;
|
|
51
62
|
level: CleanupLevel;
|
|
63
|
+
/**
|
|
64
|
+
* Spans the speaker marked as bloopers out loud (R27 §122), from
|
|
65
|
+
* `findBloopSpans`. Passed in rather than detected here so this stays a pure
|
|
66
|
+
* function of its arguments — and so `--blooper-marker` is the only thing
|
|
67
|
+
* that can put a `retake` cut in the timeline.
|
|
68
|
+
*/
|
|
69
|
+
bloops?: readonly { startWord: number; endWord: number; startSec: number; endSec: number }[];
|
|
52
70
|
}
|
|
53
71
|
|
|
54
|
-
export function buildCutlist({
|
|
72
|
+
export function buildCutlist({
|
|
73
|
+
transcript,
|
|
74
|
+
analysis,
|
|
75
|
+
duration,
|
|
76
|
+
level,
|
|
77
|
+
bloops,
|
|
78
|
+
}: BuildCutlistArgs): Segment[] {
|
|
55
79
|
const keepAll: Segment[] = [{ srcIn: 0, srcOut: duration, kind: "keep" }];
|
|
80
|
+
// `exact` means exact: it is the escape hatch for "touch nothing", and a
|
|
81
|
+
// blooper cut is still a cut. --blooper-marker with --cleanup exact is a
|
|
82
|
+
// contradiction, and the flag the user typed second does not get to win.
|
|
56
83
|
if (level === "exact") return keepAll;
|
|
57
84
|
const policy = POLICIES[level];
|
|
58
85
|
const words = transcript.words;
|
|
@@ -62,17 +89,49 @@ export function buildCutlist({ transcript, analysis, duration, level }: BuildCut
|
|
|
62
89
|
|
|
63
90
|
const removals: Removal[] = [];
|
|
64
91
|
|
|
92
|
+
// Marked bloopers, injected BEFORE the sort/merge below so they inherit the
|
|
93
|
+
// whole existing machine: merging with the silence that brackets the flub,
|
|
94
|
+
// MIN_KEEP sliver folding, and the partition emit. Source is "acoustic"
|
|
95
|
+
// because the boundaries are word stamps we chose deliberately — the
|
|
96
|
+
// protected-word pass must not push them back off the words they exist to
|
|
97
|
+
// remove.
|
|
98
|
+
for (const b of bloops ?? []) {
|
|
99
|
+
removals.push({
|
|
100
|
+
start: b.startSec,
|
|
101
|
+
end: b.endSec,
|
|
102
|
+
reason: "retake",
|
|
103
|
+
confidence: 1,
|
|
104
|
+
source: "acoustic",
|
|
105
|
+
});
|
|
106
|
+
}
|
|
107
|
+
|
|
65
108
|
for (const pause of analysis.cuttable) {
|
|
66
|
-
|
|
67
|
-
|
|
109
|
+
// Lead and tail are decided by the SILENCE's position in the file, not by
|
|
110
|
+
// comparing it to a word stamp (R27 §127). Whisper's `-ml 1` stamps stretch
|
|
111
|
+
// to fill gaps: on a real take the first word was stamped 0.00–0.53 over
|
|
112
|
+
// silence that plainly starts at 0.00, so `pause.end <= first.start` was
|
|
113
|
+
// false and the opening dead air fell through to the interior rule — where
|
|
114
|
+
// it was under `pauseMin` and survived. The tail failed the same way, by a
|
|
115
|
+
// 0.07s overlap, leaving the speaker on screen looking down after the last
|
|
116
|
+
// word. Dead air touching either end of the file IS lead/tail, whatever the
|
|
117
|
+
// recognizer claims about where words begin.
|
|
118
|
+
const isLead = pause.start <= 1e-6;
|
|
119
|
+
const isTail = pause.end >= duration - 1e-6;
|
|
68
120
|
if (isLead) {
|
|
69
|
-
//
|
|
70
|
-
|
|
121
|
+
// Keep LEAD_KEEP of run-up before speech starts (hook starts fast).
|
|
122
|
+
// Measured back from the END of the silence — where speech actually
|
|
123
|
+
// begins — rather than from a word stamp that may cover the silence.
|
|
124
|
+
const speechStarts = first !== undefined ? Math.max(pause.end, first.start) : pause.end;
|
|
125
|
+
const end = Math.min(pause.end, speechStarts - policy.leadKeep);
|
|
71
126
|
if (end - pause.start >= MIN_REMOVAL) {
|
|
72
127
|
removals.push({ start: pause.start, end, reason: "silence", confidence: 0.95, source: "acoustic" });
|
|
73
128
|
}
|
|
74
129
|
} else if (isTail) {
|
|
75
|
-
|
|
130
|
+
// Same, mirrored: the take ends when the speech does, so keep TAIL_KEEP
|
|
131
|
+
// past the last word and drop everything after — including the pause the
|
|
132
|
+
// recognizer's final stamp bled into.
|
|
133
|
+
const speechEnds = last !== undefined ? Math.min(pause.start, last.end) : pause.start;
|
|
134
|
+
const start = Math.max(pause.start, speechEnds + policy.tailKeep);
|
|
76
135
|
if (pause.end - start >= MIN_REMOVAL) {
|
|
77
136
|
removals.push({ start, end: pause.end, reason: "silence", confidence: 0.95, source: "acoustic" });
|
|
78
137
|
}
|
package/src/grounding.ts
CHANGED
|
@@ -88,7 +88,17 @@ function needsSupport(token: string): boolean {
|
|
|
88
88
|
function stringsOf(value: unknown): string[] {
|
|
89
89
|
if (typeof value === "string") return [value];
|
|
90
90
|
if (Array.isArray(value)) return value.flatMap(stringsOf);
|
|
91
|
-
if (value && typeof value === "object")
|
|
91
|
+
if (value && typeof value === "object") {
|
|
92
|
+
// A structured line carries its copy in `text`; its siblings are RENDERING
|
|
93
|
+
// DIRECTIVES, not words anyone reads (R27 §126). Walking every value made
|
|
94
|
+
// the check judge them as on-screen copy, so a StrikethroughReveal line
|
|
95
|
+
// `{text, struck, mark: "cross"}` was reported as inventing "cross" — and
|
|
96
|
+
// the take can never contain those tokens, so the warning was unfixable
|
|
97
|
+
// by construction. Two of the four warnings on a real render were this.
|
|
98
|
+
const text = (value as Record<string, unknown>).text;
|
|
99
|
+
if (typeof text === "string") return [text];
|
|
100
|
+
return Object.values(value).flatMap(stringsOf);
|
|
101
|
+
}
|
|
92
102
|
return [];
|
|
93
103
|
}
|
|
94
104
|
|
package/src/index.ts
CHANGED
package/src/ingest.ts
CHANGED
|
@@ -6,6 +6,27 @@ export interface IngestTools {
|
|
|
6
6
|
ffprobePath: string;
|
|
7
7
|
}
|
|
8
8
|
|
|
9
|
+
/**
|
|
10
|
+
* The stream's rotation, normalized to 0/90/180/270 (R27 §119).
|
|
11
|
+
*
|
|
12
|
+
* Two spellings, because containers disagree: a Display Matrix side-datum
|
|
13
|
+
* (modern ffprobe, and the only one a concatenated MP4 keeps) or the legacy
|
|
14
|
+
* `rotate` tag. ffprobe reports the matrix angle signed — -90 and 270 are the
|
|
15
|
+
* same quarter turn — so everything is folded into [0, 360).
|
|
16
|
+
*/
|
|
17
|
+
export function normalizeRotation(raw: number | string | undefined): number {
|
|
18
|
+
const n = typeof raw === "string" ? Number(raw) : raw;
|
|
19
|
+
if (n === undefined || !Number.isFinite(n)) return 0;
|
|
20
|
+
const deg = ((Math.round(n) % 360) + 360) % 360;
|
|
21
|
+
// Anything that is not a quarter turn cannot swap an axis; treat as upright.
|
|
22
|
+
return deg % 90 === 0 ? deg : 0;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/** A quarter turn exchanges the axes, so the DISPLAYED frame is w/h swapped. */
|
|
26
|
+
export function rotationSwapsAxes(rotation: number): boolean {
|
|
27
|
+
return rotation === 90 || rotation === 270;
|
|
28
|
+
}
|
|
29
|
+
|
|
9
30
|
export async function probe(tools: IngestTools, path: string): Promise<Probe> {
|
|
10
31
|
const { stdout } = await run(tools.ffprobePath, [
|
|
11
32
|
"-v", "error",
|
|
@@ -21,6 +42,8 @@ export async function probe(tools: IngestTools, path: string): Promise<Probe> {
|
|
|
21
42
|
height?: number;
|
|
22
43
|
avg_frame_rate?: string;
|
|
23
44
|
r_frame_rate?: string;
|
|
45
|
+
side_data_list?: Array<{ rotation?: number }>;
|
|
46
|
+
tags?: { rotate?: string };
|
|
24
47
|
}>;
|
|
25
48
|
format?: { duration?: string };
|
|
26
49
|
};
|
|
@@ -31,12 +54,25 @@ export async function probe(tools: IngestTools, path: string): Promise<Probe> {
|
|
|
31
54
|
const [num, den] = (rate ?? "30/1").split("/").map(Number);
|
|
32
55
|
const duration = Number(info.format?.duration);
|
|
33
56
|
if (!Number.isFinite(duration) || duration <= 0) throw new Error(`could not determine duration of ${path}`);
|
|
57
|
+
// `side_data_list` carries several kinds of datum (ambient viewing
|
|
58
|
+
// environment, content light level); only one of them has a rotation.
|
|
59
|
+
const matrix = video.side_data_list?.find((s) => typeof s.rotation === "number");
|
|
60
|
+
const rotation = normalizeRotation(matrix?.rotation ?? video.tags?.rotate);
|
|
61
|
+
const rawW = video.width ?? 0;
|
|
62
|
+
const rawH = video.height ?? 0;
|
|
63
|
+
// Report what is DISPLAYED. ffmpeg auto-rotates in the filter chain, so every
|
|
64
|
+
// measurement taken through it (cropdetect, face, the mezzanine) is already
|
|
65
|
+
// in this space; returning the raw stream size made the pipeline reconcile
|
|
66
|
+
// two orientations into a bogus square and "detect" a letterbox on a
|
|
67
|
+
// full-frame portrait take (R27 §119).
|
|
68
|
+
const swap = rotationSwapsAxes(rotation);
|
|
34
69
|
return {
|
|
35
70
|
duration,
|
|
36
|
-
width:
|
|
37
|
-
height:
|
|
71
|
+
width: swap ? rawH : rawW,
|
|
72
|
+
height: swap ? rawW : rawH,
|
|
38
73
|
fps: den ? (num ?? 30) / den : 30,
|
|
39
74
|
hasAudio: Boolean(audio),
|
|
75
|
+
...(rotation !== 0 ? { rotation } : {}),
|
|
40
76
|
};
|
|
41
77
|
}
|
|
42
78
|
|
package/src/normalize.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { rename, rm } from "node:fs/promises";
|
|
1
2
|
import { run } from "./exec";
|
|
2
3
|
import type { ContentRectSegment } from "./content-rect";
|
|
3
4
|
import type { WindowFace } from "./face";
|
|
@@ -368,7 +369,14 @@ export function normalizationFilterGraph(plan: NormalizePlan): string {
|
|
|
368
369
|
return (
|
|
369
370
|
`[0:v]trim=start=${s.startSec.toFixed(3)}:end=${s.endSec.toFixed(3)},` +
|
|
370
371
|
`setpts=PTS-STARTPTS,crop=${w.w}:${w.h}:${w.x}:${w.y},` +
|
|
371
|
-
|
|
372
|
+
// setsar=1 is load-bearing, not tidiness (R27 §125). Every segment is
|
|
373
|
+
// scaled to the SAME canvas, but from a DIFFERENT crop, and ffmpeg
|
|
374
|
+
// derives a sample aspect from that ratio: a 946x1682 crop yields SAR
|
|
375
|
+
// 1683:1682 and a 932x1660 crop 1377:1376. `concat` requires identical
|
|
376
|
+
// SAR across inputs and aborts the whole bake when they disagree, so a
|
|
377
|
+
// take whose framing varies — exactly the take normalization exists
|
|
378
|
+
// for — failed to render at all.
|
|
379
|
+
`scale=${plan.canvas.width}:${plan.canvas.height},setsar=1[v${i}]`
|
|
372
380
|
);
|
|
373
381
|
});
|
|
374
382
|
const inputs = plan.segments.map((_, i) => `[v${i}]`).join("");
|
|
@@ -386,12 +394,27 @@ export async function bakeNormalizedSource(
|
|
|
386
394
|
plan: NormalizePlan,
|
|
387
395
|
outPath: string,
|
|
388
396
|
): Promise<void> {
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
+
// Encode to a sibling temp path and rename only on success (R27 §125).
|
|
398
|
+
// ffmpeg writes the container header as it goes, so a bake that dies
|
|
399
|
+
// mid-graph leaves a file with no `moov` atom — and the cache upstream keys
|
|
400
|
+
// on EXISTENCE, so that corpse is then reused as a valid normalized source
|
|
401
|
+
// on every later run. The failure surfaces as "moov atom not found" from a
|
|
402
|
+
// step that never ran, and deleting the workdir is the only way out. Rename
|
|
403
|
+
// is atomic on a POSIX filesystem, so the cache can only ever see a file
|
|
404
|
+
// ffmpeg finished writing.
|
|
405
|
+
const partial = `${outPath}.partial.mp4`;
|
|
406
|
+
try {
|
|
407
|
+
await run(tools.ffmpegPath, [
|
|
408
|
+
"-y", "-i", input,
|
|
409
|
+
"-filter_complex", normalizationFilterGraph(plan),
|
|
410
|
+
"-map", "[v]", "-map", "0:a?",
|
|
411
|
+
"-c:v", "libx264", "-preset", "veryfast", "-crf", "18", "-g", "30", "-pix_fmt", "yuv420p",
|
|
412
|
+
"-c:a", "aac", "-b:a", "192k",
|
|
413
|
+
partial,
|
|
414
|
+
]);
|
|
415
|
+
await rename(partial, outPath);
|
|
416
|
+
} catch (err) {
|
|
417
|
+
await rm(partial, { force: true });
|
|
418
|
+
throw err;
|
|
419
|
+
}
|
|
397
420
|
}
|
package/src/producer/beats.ts
CHANGED
|
@@ -6,13 +6,39 @@ import { MAX_SCENE_SEC } from "../assemble";
|
|
|
6
6
|
import { COVER_MAX_WORDS, coverHeadline } from "../cover";
|
|
7
7
|
import type { LlmProvider } from "./provider";
|
|
8
8
|
|
|
9
|
+
/**
|
|
10
|
+
* Free text from the model, capped rather than refused (R27 §123).
|
|
11
|
+
*
|
|
12
|
+
* The standing doctrine (§112) is that LLM output is untrusted input,
|
|
13
|
+
* "validated where the pipeline can still degrade instead of at the point
|
|
14
|
+
* where it can only die". A bare `.max(n)` is the second kind: on two of three
|
|
15
|
+
* real runs the editorial call came back with a 61-character `onScreenCopy`
|
|
16
|
+
* and the whole produce died at the Zod boundary — transcription, analysis and
|
|
17
|
+
* the cut all discarded over one character of a headline.
|
|
18
|
+
*
|
|
19
|
+
* `preprocess` keeps `maxLength: n` in the JSON schema the provider is handed,
|
|
20
|
+
* so the model is still ASKED for the limit; it just no longer costs a run
|
|
21
|
+
* when the model misses by a word. Truncation prefers the last word boundary,
|
|
22
|
+
* and adds no ellipsis — the prompt explicitly forbids one on cover text.
|
|
23
|
+
*/
|
|
24
|
+
export function cappedText(max: number): z.ZodType<string> {
|
|
25
|
+
return z.preprocess((v) => {
|
|
26
|
+
if (typeof v !== "string" || v.length <= max) return v;
|
|
27
|
+
const cut = v.slice(0, max);
|
|
28
|
+
const lastSpace = cut.lastIndexOf(" ");
|
|
29
|
+
// Only honour a word boundary that keeps most of the budget; a single very
|
|
30
|
+
// long word would otherwise collapse to nothing.
|
|
31
|
+
return (lastSpace > max * 0.6 ? cut.slice(0, lastSpace) : cut).trimEnd();
|
|
32
|
+
}, z.string().max(max)) as z.ZodType<string>;
|
|
33
|
+
}
|
|
34
|
+
|
|
9
35
|
/** Call 1 — the editorial call (PHASE1 §4): moments, copy, component picks. */
|
|
10
36
|
export const MomentSchema = z.object({
|
|
11
37
|
startWord: z.number().int().nonnegative(),
|
|
12
38
|
endWord: z.number().int().nonnegative(),
|
|
13
|
-
purpose:
|
|
39
|
+
purpose: cappedText(100),
|
|
14
40
|
/** Short on-screen copy for this beat — the fallback TitleCard title. */
|
|
15
|
-
onScreenCopy:
|
|
41
|
+
onScreenCopy: cappedText(60),
|
|
16
42
|
/** "none" = plain talking head with captions; otherwise a library component. */
|
|
17
43
|
sceneKind: z.union([SceneComponentIdSchema, z.literal("none")]),
|
|
18
44
|
/**
|
|
@@ -24,25 +50,30 @@ export const MomentSchema = z.object({
|
|
|
24
50
|
layout: LayoutSchema.optional().describe(
|
|
25
51
|
"stage layout for this scene; omit for the component default. NEVER a layout the framing brief marks UNAVAILABLE for these words",
|
|
26
52
|
),
|
|
27
|
-
rationale:
|
|
53
|
+
rationale: cappedText(120).optional(),
|
|
28
54
|
});
|
|
29
55
|
export type Moment = z.infer<typeof MomentSchema>;
|
|
30
56
|
|
|
31
57
|
export const BeatSheetSchema = z.object({
|
|
32
|
-
hook:
|
|
58
|
+
hook: cappedText(120),
|
|
33
59
|
/**
|
|
34
60
|
* Banner text for the cover image (FINDINGS §31). Written here rather than
|
|
35
61
|
* by a second LLM call, because the producer is already choosing the hook —
|
|
36
62
|
* this is the same editorial judgement, shortened for a thumbnail.
|
|
37
63
|
*/
|
|
38
|
-
coverText:
|
|
39
|
-
.string()
|
|
40
|
-
.max(60)
|
|
64
|
+
coverText: cappedText(60)
|
|
41
65
|
.optional()
|
|
42
66
|
.describe(
|
|
43
67
|
`cover banner: at most ${COVER_MAX_WORDS} words, the hook compressed to a thumbnail headline`,
|
|
44
68
|
),
|
|
45
|
-
|
|
69
|
+
/**
|
|
70
|
+
* Raised from 12 to 24 (§118): with the alternation policy above, a cap of
|
|
71
|
+
* 12 moments is a ceiling of ~6 graphics however long the take is. A 64s
|
|
72
|
+
* take enumerating five features needs seven graphic beats — hook, five
|
|
73
|
+
* features, payoff — and therefore ~14 moments to alternate between them.
|
|
74
|
+
* The cap was binding before the coverage budget ever was.
|
|
75
|
+
*/
|
|
76
|
+
moments: z.array(MomentSchema).min(1).max(24),
|
|
46
77
|
});
|
|
47
78
|
export type BeatSheet = z.infer<typeof BeatSheetSchema>;
|
|
48
79
|
|
|
@@ -77,7 +108,8 @@ Virality grammar — follow these as hard policies:
|
|
|
77
108
|
- Use contrast/negation beats (StrikethroughReveal, RuleCard with struck alternatives) when the speaker rejects an idea.
|
|
78
109
|
- End with a payoff or takeaway moment.
|
|
79
110
|
- A moment spans the FULL stretch of speech about its beat — typically 5-15 seconds — in transcript order, non-overlapping. The graphic stays on screen for the ENTIRE moment, so the word range must cover everything the graphic refers to: a stat card leaves when the speaker moves on, not before.
|
|
80
|
-
-
|
|
111
|
+
- COUNT: the user prompt states how many graphic moments this take should get. That number is a TARGET, not a maximum — hit it. Planning under it is the most common failure: a take that makes five distinct points and gets two graphics has been under-produced, whatever the coverage percentage says.
|
|
112
|
+
- COVERAGE: graphics should be on screen for roughly 40-50% of the runtime. A graphic spends its whole moment against that budget, so when the target implies many graphics, make each moment SHORTER rather than dropping moments — more, tighter graphics beats fewer, longer ones. Spread them evenly: never leave a stretch longer than ~10 seconds with no graphic.
|
|
81
113
|
- VARIETY: never the same component twice in a row, and prefer a component you have NOT used yet in this video — reuse a treatment only when the beat genuinely calls for it. A repeat reads as a template.
|
|
82
114
|
- Keep the face LARGE: prefer StatCard/RuleCard/ScreenshotFrame (they sit under a big face) over TitleCard (face becomes a small bubble); use FlowDiagram/TerminalMock sparingly — they remove the face entirely and only earn that when the graphic IS the point.
|
|
83
115
|
- The transcript is ASR output and may contain mishearings: an unfamiliar proper noun is more likely a mistranscription of a common phrase than a real entity — write on-screen copy with the common-sense reading, never a suspected mishearing.
|
|
@@ -117,11 +149,23 @@ export function buildBeatsUserPrompt(
|
|
|
117
149
|
const menu = Object.entries(SCENE_REGISTRY)
|
|
118
150
|
.map(([id, meta]) => `- ${id}: ${meta.whenToUse}`)
|
|
119
151
|
.join("\n");
|
|
152
|
+
// §118: state the graphic COUNT explicitly. Everything else in this prompt
|
|
153
|
+
// describes what a good graphic is; nothing said how many to plan, and the
|
|
154
|
+
// coverage budget downstream only ever removes.
|
|
155
|
+
const enumerated = countEnumeratedBeats(transcript);
|
|
156
|
+
const target = graphicsTarget(clip?.targetSec ?? duration, enumerated);
|
|
157
|
+
const targetLine =
|
|
158
|
+
`Graphic moments to plan: ${target}` +
|
|
159
|
+
(enumerated > 0
|
|
160
|
+
? ` — the speaker enumerates ${enumerated} points out loud, so each one earns its own graphic, plus a hook and a payoff.\n`
|
|
161
|
+
: ` (about one per ${SEC_PER_GRAPHIC}s of runtime). Plan this many unless the take genuinely cannot carry them.\n`);
|
|
120
162
|
return (
|
|
121
163
|
`Intent: ${intent ?? "make this clear, punchy and viral-worthy"}\n` +
|
|
122
164
|
(clip
|
|
123
|
-
? `Target clip length: ~${clip.targetSec.toFixed(0)}s (see CLIP SELECTION below)\n
|
|
124
|
-
: `Output duration after the cut: ${duration.toFixed(1)}s\n
|
|
165
|
+
? `Target clip length: ~${clip.targetSec.toFixed(0)}s (see CLIP SELECTION below)\n`
|
|
166
|
+
: `Output duration after the cut: ${duration.toFixed(1)}s\n`) +
|
|
167
|
+
targetLine +
|
|
168
|
+
"\n" +
|
|
125
169
|
// Landscape layout guidance (R21 §101): without it the first real 16:9
|
|
126
170
|
// run put nearly every graphic in a lower third. A deterministic variety
|
|
127
171
|
// pass downstream is the guarantee; this is the steer.
|
|
@@ -146,6 +190,87 @@ export interface BeatsValidationIssue {
|
|
|
146
190
|
issue: string;
|
|
147
191
|
}
|
|
148
192
|
|
|
193
|
+
/**
|
|
194
|
+
* How many graphics a take of this length should be asked for (§118).
|
|
195
|
+
*
|
|
196
|
+
* The failure this exists for: nothing ever told the producer how many
|
|
197
|
+
* graphics to plan. `GRAPHICS_COVERAGE_TARGET` reads like a target and is
|
|
198
|
+
* only a ceiling — the demote loop below runs when the model plans too MANY
|
|
199
|
+
* and does nothing at all when it plans too few. On one 64s take the model
|
|
200
|
+
* planned three graphics against a budget that allowed roughly twenty-nine
|
|
201
|
+
* seconds of them; the loop never executed once, so no existing mechanism
|
|
202
|
+
* had an opinion.
|
|
203
|
+
*
|
|
204
|
+
* One graphic per ~9s of runtime, which is the density the prompt's own
|
|
205
|
+
* "never leave a stretch longer than ~10 seconds with no graphic" rule
|
|
206
|
+
* implies, floored at the §29 short-take count.
|
|
207
|
+
*/
|
|
208
|
+
export const SEC_PER_GRAPHIC = 9;
|
|
209
|
+
|
|
210
|
+
/**
|
|
211
|
+
* Ordinal cues a speaker uses to enumerate. A take that counts its own
|
|
212
|
+
* points out loud is telling us how many graphics it wants, and that signal
|
|
213
|
+
* is free, deterministic, and better than any runtime heuristic.
|
|
214
|
+
*/
|
|
215
|
+
const ORDINAL_WORDS = [
|
|
216
|
+
"one", "two", "three", "four", "five", "six", "seven", "eight", "nine", "ten",
|
|
217
|
+
];
|
|
218
|
+
const ORDINAL_ADJECTIVES = [
|
|
219
|
+
"first", "second", "third", "fourth", "fifth",
|
|
220
|
+
"sixth", "seventh", "eighth", "ninth", "tenth",
|
|
221
|
+
];
|
|
222
|
+
|
|
223
|
+
/**
|
|
224
|
+
* How many distinct enumerated beats the speaker announces — "number one …
|
|
225
|
+
* number two", "first … second", "step 3". Counts DISTINCT ordinals so a
|
|
226
|
+
* speaker who says "number two" twice doesn't inflate the target, and
|
|
227
|
+
* requires at least two so a passing "first of all" isn't read as a list.
|
|
228
|
+
*/
|
|
229
|
+
export function countEnumeratedBeats(transcript: Transcript): number {
|
|
230
|
+
const words = transcript.words.map((w) =>
|
|
231
|
+
w.text.toLowerCase().replace(/[^a-z0-9]/g, ""),
|
|
232
|
+
);
|
|
233
|
+
const seen = new Set<number>();
|
|
234
|
+
for (let i = 0; i < words.length; i++) {
|
|
235
|
+
const w = words[i]!;
|
|
236
|
+
const adj = ORDINAL_ADJECTIVES.indexOf(w);
|
|
237
|
+
if (adj !== -1) {
|
|
238
|
+
seen.add(adj + 1);
|
|
239
|
+
continue;
|
|
240
|
+
}
|
|
241
|
+
// "number one" / "step 2" / "point three" — the ordinal must FOLLOW a
|
|
242
|
+
// counting noun, or every stray "one" in the take counts as a beat.
|
|
243
|
+
if (w !== "number" && w !== "step" && w !== "point" && w !== "tip") continue;
|
|
244
|
+
const next = words[i + 1];
|
|
245
|
+
if (!next) continue;
|
|
246
|
+
const spelled = ORDINAL_WORDS.indexOf(next);
|
|
247
|
+
if (spelled !== -1) {
|
|
248
|
+
seen.add(spelled + 1);
|
|
249
|
+
continue;
|
|
250
|
+
}
|
|
251
|
+
const digit = Number.parseInt(next, 10);
|
|
252
|
+
if (Number.isInteger(digit) && digit >= 1 && digit <= 10) seen.add(digit);
|
|
253
|
+
}
|
|
254
|
+
return seen.size >= 2 ? seen.size : 0;
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
/**
|
|
258
|
+
* The number of graphics to ASK for — whichever of structure and runtime is
|
|
259
|
+
* larger. An enumerated take earns its own count plus a hook and a payoff
|
|
260
|
+
* (the virality grammar demands both anyway), but a long take that happens
|
|
261
|
+
* to enumerate three points still has everything else in it, so runtime
|
|
262
|
+
* density is a floor rather than a loser.
|
|
263
|
+
*/
|
|
264
|
+
export function graphicsTarget(runtimeSec: number, enumerated: number): number {
|
|
265
|
+
const byRuntime = Math.max(
|
|
266
|
+
SHORT_TAKE_MIN_GRAPHICS,
|
|
267
|
+
Math.round(runtimeSec / SEC_PER_GRAPHIC),
|
|
268
|
+
);
|
|
269
|
+
const byStructure = enumerated > 0 ? enumerated + 2 : 0;
|
|
270
|
+
// Never more than the moment schema can carry once alternation is counted.
|
|
271
|
+
return Math.min(Math.max(byRuntime, byStructure), 12);
|
|
272
|
+
}
|
|
273
|
+
|
|
149
274
|
/** Fraction of the runtime that should show a graphic (FINDINGS §7). */
|
|
150
275
|
export const GRAPHICS_COVERAGE_TARGET = 0.45;
|
|
151
276
|
/**
|
|
@@ -157,6 +282,28 @@ export const GRAPHICS_COVERAGE_TARGET = 0.45;
|
|
|
157
282
|
export const SHORT_TAKE_SEC = 45;
|
|
158
283
|
export const SHORT_TAKE_MIN_GRAPHICS = 4;
|
|
159
284
|
|
|
285
|
+
/** Why the target was what it was, when the take enumerated itself. */
|
|
286
|
+
function enumeratedNote(transcript: Transcript): string | null {
|
|
287
|
+
const n = countEnumeratedBeats(transcript);
|
|
288
|
+
return n > 0 ? ` — the take enumerates ${n} points` : null;
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
/**
|
|
292
|
+
* The one-line graphics accounting (§118b): delivered vs asked, and why the
|
|
293
|
+
* ask was what it was. One formatter for the console issue and `report.txt`,
|
|
294
|
+
* so the two can never say different things about the same run.
|
|
295
|
+
*/
|
|
296
|
+
export function formatGraphicsAccounting(
|
|
297
|
+
delivered: number,
|
|
298
|
+
asked: number,
|
|
299
|
+
transcript: Transcript,
|
|
300
|
+
): string {
|
|
301
|
+
return (
|
|
302
|
+
`graphics: ${delivered} of ${asked} planned` +
|
|
303
|
+
(enumeratedNote(transcript) ?? ` (target is ~1 per ${SEC_PER_GRAPHIC}s)`)
|
|
304
|
+
);
|
|
305
|
+
}
|
|
306
|
+
|
|
160
307
|
/** A moment's approximate seconds of speech, from the transcript word stamps. */
|
|
161
308
|
function momentDuration(m: Moment, transcript: Transcript): number {
|
|
162
309
|
const first = transcript.words[m.startWord];
|
|
@@ -170,10 +317,21 @@ function momentMidpoint(m: Moment, transcript: Transcript): number {
|
|
|
170
317
|
return first && last ? (first.start + last.end) / 2 : 0;
|
|
171
318
|
}
|
|
172
319
|
|
|
173
|
-
/**
|
|
320
|
+
/**
|
|
321
|
+
* Semantic validation beyond the schema; repairs what it can, reports the rest.
|
|
322
|
+
*
|
|
323
|
+
* `askedGraphics` is the count the PROMPT stated (§118b): pass it so the
|
|
324
|
+
* shortfall check measures against what was actually asked for — on a clip
|
|
325
|
+
* run the internal fallback would measure against the full take's runtime,
|
|
326
|
+
* not the clip target the prompt named. `null` skips the check entirely (the
|
|
327
|
+
* pre-slice pass of a clip run, whose sheet is renormalized after slicing —
|
|
328
|
+
* two passes reporting the same shortfall would say it twice). Omitted, the
|
|
329
|
+
* ask is derived from the transcript's own span.
|
|
330
|
+
*/
|
|
174
331
|
export function normalizeBeatSheet(
|
|
175
332
|
sheet: BeatSheet,
|
|
176
333
|
transcript: Transcript,
|
|
334
|
+
askedGraphics?: number | null,
|
|
177
335
|
): { sheet: BeatSheet; issues: BeatsValidationIssue[] } {
|
|
178
336
|
const wordCount = transcript.words.length;
|
|
179
337
|
const issues: BeatsValidationIssue[] = [];
|
|
@@ -225,6 +383,15 @@ export function normalizeBeatSheet(
|
|
|
225
383
|
|
|
226
384
|
// On a short take the count floor outranks the percentage — never demote
|
|
227
385
|
// below it, whatever the coverage budget says (§29).
|
|
386
|
+
//
|
|
387
|
+
// §118 decided NOT to extend this floor above 45s, and the reason matters:
|
|
388
|
+
// a floor that outranks the ceiling at every length would fight §114's
|
|
389
|
+
// full-span pricing — more graphics × whole moments blows past 45%, the
|
|
390
|
+
// loop below starts removing what the floor just required, and the two
|
|
391
|
+
// rules oscillate. It would also be treating the wrong failure. When the
|
|
392
|
+
// producer UNDER-plans, this loop never runs at all, so no floor here
|
|
393
|
+
// could have helped; the fix is the target in the prompt. What this layer
|
|
394
|
+
// owes the user instead is to SAY so — see the shortfall issue below.
|
|
228
395
|
const minGraphics = runtime < SHORT_TAKE_SEC ? SHORT_TAKE_MIN_GRAPHICS : 0;
|
|
229
396
|
|
|
230
397
|
for (;;) {
|
|
@@ -293,6 +460,22 @@ export function normalizeBeatSheet(
|
|
|
293
460
|
issues.push({ moment: -1, issue: `coverText shortened to "${coverText}"` });
|
|
294
461
|
}
|
|
295
462
|
|
|
463
|
+
// §118b: a run that under-delivers must say so. The producer was asked for
|
|
464
|
+
// a specific number of graphics; if fewer survive, that is a fact about
|
|
465
|
+
// this render the report should carry, exactly as every cut is justified.
|
|
466
|
+
// Silence is what let three graphics on a five-point take look normal.
|
|
467
|
+
const delivered = surviving().length;
|
|
468
|
+
const asked =
|
|
469
|
+
askedGraphics === undefined
|
|
470
|
+
? graphicsTarget(runtime, countEnumeratedBeats(transcript))
|
|
471
|
+
: askedGraphics;
|
|
472
|
+
if (asked !== null && delivered < asked) {
|
|
473
|
+
issues.push({
|
|
474
|
+
moment: -1,
|
|
475
|
+
issue: formatGraphicsAccounting(delivered, asked, transcript),
|
|
476
|
+
});
|
|
477
|
+
}
|
|
478
|
+
|
|
296
479
|
return { sheet: { hook: sheet.hook, coverText, moments }, issues };
|
|
297
480
|
}
|
|
298
481
|
|
|
@@ -305,10 +488,22 @@ export async function generateBeatSheet(
|
|
|
305
488
|
framingBrief?: string,
|
|
306
489
|
clip?: { targetSec: number },
|
|
307
490
|
aspect?: "9:16" | "16:9",
|
|
308
|
-
): Promise<{
|
|
491
|
+
): Promise<{
|
|
492
|
+
sheet: BeatSheet;
|
|
493
|
+
issues: BeatsValidationIssue[];
|
|
494
|
+
/** The graphic count the prompt asked for (§118b) — what "asked" means everywhere downstream. */
|
|
495
|
+
asked: number;
|
|
496
|
+
highlight?: ClipHighlight;
|
|
497
|
+
}> {
|
|
309
498
|
const user =
|
|
310
499
|
(speaker ? `The speaker: ${speaker}\n\n` : "") +
|
|
311
500
|
buildBeatsUserPrompt(transcript, duration, intent, framingBrief, clip, aspect);
|
|
501
|
+
// The same number `buildBeatsUserPrompt` states — computed from the same
|
|
502
|
+
// inputs by the same pure functions, so the check and the ask agree.
|
|
503
|
+
const asked = graphicsTarget(
|
|
504
|
+
clip?.targetSec ?? duration,
|
|
505
|
+
countEnumeratedBeats(transcript),
|
|
506
|
+
);
|
|
312
507
|
if (clip) {
|
|
313
508
|
// Same editorial call, extended schema (R19 §93d) — the highlight and the
|
|
314
509
|
// beat sheet come from ONE judgement, so they cannot disagree.
|
|
@@ -318,7 +513,8 @@ export async function generateBeatSheet(
|
|
|
318
513
|
schema: ClipBeatSheetSchema,
|
|
319
514
|
schemaName: "clip_beat_sheet",
|
|
320
515
|
});
|
|
321
|
-
|
|
516
|
+
// `null`: the post-slice renormalization owns the shortfall check.
|
|
517
|
+
return { ...normalizeBeatSheet(raw, transcript, null), asked, highlight: raw.highlight };
|
|
322
518
|
}
|
|
323
519
|
const raw = await provider.complete({
|
|
324
520
|
system: PRODUCER_SYSTEM,
|
|
@@ -326,5 +522,5 @@ export async function generateBeatSheet(
|
|
|
326
522
|
schema: BeatSheetSchema,
|
|
327
523
|
schemaName: "beat_sheet",
|
|
328
524
|
});
|
|
329
|
-
return normalizeBeatSheet(raw, transcript);
|
|
525
|
+
return { ...normalizeBeatSheet(raw, transcript, asked), asked };
|
|
330
526
|
}
|
package/src/producer/index.ts
CHANGED
|
@@ -106,6 +106,13 @@ export function defaultProviderName(env: NodeJS.ProcessEnv = process.env): Provi
|
|
|
106
106
|
export interface ProduceScenesResult {
|
|
107
107
|
beatSheet: BeatSheet;
|
|
108
108
|
beatIssues: BeatsValidationIssue[];
|
|
109
|
+
/**
|
|
110
|
+
* The graphics accounting (§118b): how many the prompt asked for and how
|
|
111
|
+
* many survived planning. `delivered` equals the scene count — layout
|
|
112
|
+
* repair never demotes, and a failed props call falls back to a TitleCard
|
|
113
|
+
* rather than dropping the scene.
|
|
114
|
+
*/
|
|
115
|
+
graphics: { asked: number; delivered: number };
|
|
109
116
|
scenes: Scene[];
|
|
110
117
|
failures: ScenePropsFailure[];
|
|
111
118
|
/**
|
|
@@ -154,7 +161,7 @@ export async function produceScenes(
|
|
|
154
161
|
const framingBrief = args.framing
|
|
155
162
|
? buildFramingBrief(args.framing, args.transcript)
|
|
156
163
|
: undefined;
|
|
157
|
-
const { sheet, issues, highlight } = await generateBeatSheet(
|
|
164
|
+
const { sheet, issues, asked, highlight } = await generateBeatSheet(
|
|
158
165
|
provider,
|
|
159
166
|
args.transcript,
|
|
160
167
|
args.outputDuration,
|
|
@@ -187,6 +194,9 @@ export async function produceScenes(
|
|
|
187
194
|
const renorm = normalizeBeatSheet(
|
|
188
195
|
{ hook: sheet.hook, coverText: sheet.coverText, moments: anchored },
|
|
189
196
|
transcript,
|
|
197
|
+
// The ask the prompt stated — NOT re-derived from the slice, which
|
|
198
|
+
// would compare the model against a number it was never given (§118b).
|
|
199
|
+
asked,
|
|
190
200
|
);
|
|
191
201
|
workingSheet = renorm.sheet;
|
|
192
202
|
issues.push(...renorm.issues);
|
|
@@ -213,5 +223,13 @@ export async function produceScenes(
|
|
|
213
223
|
const { scenes, failures } = await generateScenes(provider, moments, transcript, {
|
|
214
224
|
framing: args.framing,
|
|
215
225
|
});
|
|
216
|
-
|
|
226
|
+
const delivered = moments.filter((m) => m.sceneKind !== "none").length;
|
|
227
|
+
return {
|
|
228
|
+
beatSheet: { ...workingSheet, moments },
|
|
229
|
+
beatIssues: issues,
|
|
230
|
+
graphics: { asked, delivered },
|
|
231
|
+
scenes,
|
|
232
|
+
failures,
|
|
233
|
+
clip,
|
|
234
|
+
};
|
|
217
235
|
}
|
package/src/schema.ts
CHANGED
|
@@ -71,10 +71,22 @@ export type Analysis = z.infer<typeof AnalysisSchema>;
|
|
|
71
71
|
|
|
72
72
|
export const ProbeSchema = z.object({
|
|
73
73
|
duration: z.number().positive(),
|
|
74
|
+
/**
|
|
75
|
+
* DISPLAYED dimensions, after the rotation matrix (R27 §119) — not the raw
|
|
76
|
+
* stream's. A phone/camera writes a portrait take as a landscape stream plus
|
|
77
|
+
* a 90° display matrix, and ffmpeg's filter chain auto-rotates, so the raw
|
|
78
|
+
* numbers disagree with every measurement taken through ffmpeg.
|
|
79
|
+
*/
|
|
74
80
|
width: z.number().int().positive(),
|
|
75
81
|
height: z.number().int().positive(),
|
|
76
82
|
fps: z.number().positive(),
|
|
77
83
|
hasAudio: z.boolean(),
|
|
84
|
+
/**
|
|
85
|
+
* The stream's rotation in degrees (0/90/180/270), recorded so a workdir says
|
|
86
|
+
* why its geometry is what it is. Optional: pre-§119 `production.json` files
|
|
87
|
+
* predate it and must still parse.
|
|
88
|
+
*/
|
|
89
|
+
rotation: z.number().int().optional(),
|
|
78
90
|
});
|
|
79
91
|
export type Probe = z.infer<typeof ProbeSchema>;
|
|
80
92
|
|