@ossclip/core 0.1.4 → 0.1.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ossclip/core",
3
- "version": "0.1.4",
3
+ "version": "0.1.6",
4
4
  "description": "ossclip's framework-free pipeline: schema, transcription, analysis, cutlist, captions, framing, and the LLM producer",
5
5
  "type": "module",
6
6
  "license": "MIT",
package/src/blooper.ts ADDED
@@ -0,0 +1,102 @@
1
+ import { normalizeToken } from "./analyze";
2
+ import { isSentenceStart } from "./clip";
3
+ import type { Transcript } from "./schema";
4
+
5
+ /**
6
+ * Blooper removal by SPOKEN MARKER (R27 §122).
7
+ *
8
+ * The ask is "drop the flubbed takes, not just the silences", and the general
9
+ * form of that is semantic: a model reading the transcript deciding which
10
+ * attempt was bad. The authoring roadmap flags exactly why that is expensive —
11
+ * `buildCutlist` is today a pure function of (raw transcript, analysis,
12
+ * duration, level), and a model in that path ends the guarantee that the same
13
+ * input and `--cleanup` always produce the same edit.
14
+ *
15
+ * A marker the speaker says OUT LOUD is the deterministic subset. It needs no
16
+ * judgement: the word is in the transcript or it is not. So this ships the
17
+ * useful half of the feature and leaves the guarantee intact — the semantic
18
+ * detector remains unbuilt, deliberately.
19
+ *
20
+ * The pattern, from the take that motivated it:
21
+ *
22
+ * "That could be one of the cases where you can say" flubbed attempt
23
+ * "blooper." marker
24
+ * "That could be one of" flubbed again
25
+ * "blooper." marker
26
+ * "That could be the exit condition." the good take
27
+ *
28
+ * The marker TERMINATES a bad attempt, so removal runs backwards from it to
29
+ * the start of the sentence it spoiled, and consecutive marked attempts
30
+ * collapse into one cut.
31
+ */
32
+
33
+ /** A span of transcript to drop, in word indices (inclusive) and source seconds. */
34
+ export interface BloopSpan {
35
+ startWord: number;
36
+ endWord: number;
37
+ startSec: number;
38
+ endSec: number;
39
+ /** How many marker words this span swallowed — 2+ means repeated attempts. */
40
+ markers: number;
41
+ }
42
+
43
+ /**
44
+ * Spans the speaker marked as bloopers.
45
+ *
46
+ * `marker` is matched with `normalizeToken`, the same normalizer the filler
47
+ * detector uses, so case and trailing punctuation do not matter — ASR writes
48
+ * the word as "blooper." with the period riding on it.
49
+ *
50
+ * Returns spans in transcript order, non-overlapping.
51
+ */
52
+ export function findBloopSpans(transcript: Transcript, marker: string): BloopSpan[] {
53
+ const want = normalizeToken(marker);
54
+ if (!want) return [];
55
+ const words = transcript.words;
56
+ const isMarker = (i: number): boolean => normalizeToken(words[i]?.text ?? "") === want;
57
+
58
+ const spans: BloopSpan[] = [];
59
+ for (let i = 0; i < words.length; i++) {
60
+ if (!isMarker(i)) continue;
61
+
62
+ // Walk back over the attempt this marker spoiled, to the start of its
63
+ // sentence. The marker's own text usually ENDS a sentence ("blooper."), so
64
+ // the scan starts at the word before it.
65
+ let start = i;
66
+ while (start > 0 && !isSentenceStart(transcript, start)) start--;
67
+
68
+ // Consecutive attempts: if everything between the previous span's end and
69
+ // this attempt's start is already being dropped, merge rather than leave a
70
+ // one-word island of a sentence nobody finished.
71
+ const prev = spans[spans.length - 1];
72
+ let markers = 1;
73
+ if (prev && start <= prev.endWord + 1) {
74
+ spans.pop();
75
+ start = prev.startWord;
76
+ markers = prev.markers + 1;
77
+ }
78
+
79
+ spans.push({
80
+ startWord: start,
81
+ endWord: i,
82
+ startSec: words[start]!.start,
83
+ endSec: words[i]!.end,
84
+ markers,
85
+ });
86
+ }
87
+ return spans;
88
+ }
89
+
90
+ /**
91
+ * A one-line account per span, for `report.txt`. The cut report justifies every
92
+ * other removal; a cut this aggressive — whole sentences, not dead air — owes
93
+ * the user the words it took.
94
+ */
95
+ export function formatBloopSpan(transcript: Transcript, span: BloopSpan): string {
96
+ const said = transcript.words
97
+ .slice(span.startWord, span.endWord + 1)
98
+ .map((w) => w.text)
99
+ .join(" ");
100
+ const attempts = span.markers > 1 ? ` (${span.markers} attempts)` : "";
101
+ return `"${said}"${attempts}`;
102
+ }
package/src/clip.ts CHANGED
@@ -33,9 +33,10 @@ export const CLIP_SNAP_TOLERANCE = 0.2;
33
33
  /** Word that closes a sentence — ASR punctuation rides on the word text. */
34
34
  const SENTENCE_END = /[.!?…]["'")\]»]*$/u;
35
35
 
36
- const isSentenceEnd = (t: Transcript, i: number): boolean =>
36
+ /** Exported since R27 §122: the blooper cut walks back to a sentence start. */
37
+ export const isSentenceEnd = (t: Transcript, i: number): boolean =>
37
38
  SENTENCE_END.test(t.words[i]?.text ?? "");
38
- const isSentenceStart = (t: Transcript, i: number): boolean =>
39
+ export const isSentenceStart = (t: Transcript, i: number): boolean =>
39
40
  i === 0 || isSentenceEnd(t, i - 1);
40
41
 
41
42
  const durSec = (t: Transcript, start: number, end: number): number =>
package/src/config.ts CHANGED
@@ -1,9 +1,12 @@
1
- import { readFileSync } from "node:fs";
1
+ import { mkdirSync, readFileSync, writeFileSync } from "node:fs";
2
2
  import { homedir } from "node:os";
3
3
  import { join } from "node:path";
4
4
 
5
5
  import type { ModelPrice } from "./producer/usage";
6
6
 
7
+ /** Whether a finished `produce` offers to open the editor. */
8
+ export type OpenEditorPref = "ask" | "always" | "never";
9
+
7
10
  export interface OssclipConfig {
8
11
  ffmpegPath: string;
9
12
  ffprobePath: string;
@@ -21,6 +24,12 @@ export interface OssclipConfig {
21
24
  * a plausible one, and stops grounding flagging the speaker's own name.
22
25
  */
23
26
  speaker?: string;
27
+ /**
28
+ * What a finished produce run does about the editor: ask (default), always
29
+ * open, or never mention it. Written by the post-produce prompt when the
30
+ * user picks one of its "stop asking" answers.
31
+ */
32
+ openEditorAfterProduce?: OpenEditorPref;
24
33
  browserExecutable?: string;
25
34
  /**
26
35
  * USD per million tokens, keyed by model id or family substring — overrides
@@ -41,6 +50,42 @@ const DEFAULTS: OssclipConfig = {
41
50
  model: "small.en",
42
51
  };
43
52
 
53
+ /** Everything ossclip owns on disk lives under here: config.json, models/, bin/, .env. */
54
+ export const CONFIG_DIR = join(homedir(), ".ossclip");
55
+
56
+ export function configFilePath(baseDir: string = CONFIG_DIR): string {
57
+ return join(baseDir, "config.json");
58
+ }
59
+
60
+ /**
61
+ * Merge a patch into `~/.ossclip/config.json`, creating it if needed.
62
+ *
63
+ * Read-merge-write over the RAW file, not a loaded OssclipConfig: users
64
+ * hand-edit this file, and keys the patch doesn't touch (`pricing`,
65
+ * `speaker`, comments-by-convention like `_note`) must survive a
66
+ * `ossclip setup` run untouched.
67
+ */
68
+ export function saveConfigPatch(patch: Partial<OssclipConfig>, baseDir: string = CONFIG_DIR): string {
69
+ const path = configFilePath(baseDir);
70
+ let existing: Record<string, unknown> = {};
71
+ try {
72
+ existing = JSON.parse(readFileSync(path, "utf8")) as Record<string, unknown>;
73
+ } catch {
74
+ // absent or unparseable — a fresh object; setup never destroys a broken
75
+ // file silently, so keep a .bak when the file existed but didn't parse
76
+ try {
77
+ const raw = readFileSync(path, "utf8");
78
+ writeFileSync(`${path}.bak`, raw);
79
+ } catch {
80
+ // truly absent — nothing to back up
81
+ }
82
+ }
83
+ mkdirSync(baseDir, { recursive: true });
84
+ const merged = { ...existing, ...patch };
85
+ writeFileSync(path, `${JSON.stringify(merged, null, 2)}\n`);
86
+ return path;
87
+ }
88
+
44
89
  /**
45
90
  * Resolution order per key: env (OSSCLIP_FFMPEG, OSSCLIP_FFPROBE, OSSCLIP_WHISPER,
46
91
  * OSSCLIP_MODEL_DIR, OSSCLIP_MODEL, OSSCLIP_BROWSER) → ~/.ossclip/config.json → defaults.
@@ -60,6 +105,8 @@ export function loadConfig(): OssclipConfig {
60
105
  model: process.env.OSSCLIP_MODEL ?? fileCfg.model ?? DEFAULTS.model,
61
106
  fastModel: process.env.OSSCLIP_FAST_MODEL ?? fileCfg.fastModel,
62
107
  speaker: process.env.OSSCLIP_SPEAKER ?? fileCfg.speaker,
108
+ openEditorAfterProduce: (process.env.OSSCLIP_OPEN_EDITOR ??
109
+ fileCfg.openEditorAfterProduce) as OpenEditorPref | undefined,
63
110
  browserExecutable: process.env.OSSCLIP_BROWSER ?? fileCfg.browserExecutable,
64
111
  pricing: fileCfg.pricing,
65
112
  };
@@ -96,6 +96,19 @@ export function stableContentRect(
96
96
  const whole: ContentRect = { x: 0, y: 0, w: width, h: height, full: true };
97
97
  if (rects.length === 0 || width <= 0 || height <= 0) return whole;
98
98
 
99
+ // A measurement that does not FIT the frame was taken in a different
100
+ // coordinate space than the one we are reconciling it against, so nothing it
101
+ // says can be trusted (R27 §119). This is how a portrait take read as
102
+ // landscape used to produce a "letterbox": cropdetect measured the rotated
103
+ // 2160x3840 frame, the caller believed 3840x2160, and clamping the union of
104
+ // the two orientations yielded a 2160x2160 square that was never on screen.
105
+ // Refuse, exactly as MIN_CONTENT_FRAC refuses an implausibly small rect —
106
+ // cropping on bad evidence is far worse than leaving bars alone.
107
+ const fits = rects.every(
108
+ (r) => r.x >= 0 && r.y >= 0 && r.x + r.w <= width && r.y + r.h <= height,
109
+ );
110
+ if (!fits) return whole;
111
+
99
112
  let left = Infinity;
100
113
  let top = Infinity;
101
114
  let right = -Infinity;
package/src/cutlist.ts CHANGED
@@ -6,12 +6,23 @@ interface LevelPolicy {
6
6
  /** Resulting gap length after tightening. */
7
7
  tightenTo: number;
8
8
  removeFillers: boolean;
9
+ /**
10
+ * Run-up kept before the first word and after the last (R27 §127).
11
+ *
12
+ * Level-dependent, because "how hard should I cut" plainly covers the ends
13
+ * too, and a fixed 0.25/0.35 was the one thing `--cleanup aggressive` could
14
+ * not tighten. On a short that LOOPS, a third of a second of the speaker
15
+ * sitting there after the last word is a visible dead beat every time the
16
+ * video repeats — the reason this surfaced on a real render.
17
+ */
18
+ leadKeep: number;
19
+ tailKeep: number;
9
20
  }
10
21
 
11
22
  const POLICIES: Record<Exclude<CleanupLevel, "exact">, LevelPolicy> = {
12
- light: { pauseMin: 1.2, tightenTo: 0.3, removeFillers: false },
13
- standard: { pauseMin: 0.7, tightenTo: 0.22, removeFillers: true },
14
- aggressive: { pauseMin: 0.5, tightenTo: 0.18, removeFillers: true },
23
+ light: { pauseMin: 1.2, tightenTo: 0.3, removeFillers: false, leadKeep: 0.35, tailKeep: 0.45 },
24
+ standard: { pauseMin: 0.7, tightenTo: 0.22, removeFillers: true, leadKeep: 0.25, tailKeep: 0.35 },
25
+ aggressive: { pauseMin: 0.5, tightenTo: 0.18, removeFillers: true, leadKeep: 0.12, tailKeep: 0.15 },
15
26
  };
16
27
 
17
28
  /** Dead air kept before the first word / after the last word. Exported for
@@ -49,10 +60,26 @@ export interface BuildCutlistArgs {
49
60
  analysis: Analysis;
50
61
  duration: number;
51
62
  level: CleanupLevel;
63
+ /**
64
+ * Spans the speaker marked as bloopers out loud (R27 §122), from
65
+ * `findBloopSpans`. Passed in rather than detected here so this stays a pure
66
+ * function of its arguments — and so `--blooper-marker` is the only thing
67
+ * that can put a `retake` cut in the timeline.
68
+ */
69
+ bloops?: readonly { startWord: number; endWord: number; startSec: number; endSec: number }[];
52
70
  }
53
71
 
54
- export function buildCutlist({ transcript, analysis, duration, level }: BuildCutlistArgs): Segment[] {
72
+ export function buildCutlist({
73
+ transcript,
74
+ analysis,
75
+ duration,
76
+ level,
77
+ bloops,
78
+ }: BuildCutlistArgs): Segment[] {
55
79
  const keepAll: Segment[] = [{ srcIn: 0, srcOut: duration, kind: "keep" }];
80
+ // `exact` means exact: it is the escape hatch for "touch nothing", and a
81
+ // blooper cut is still a cut. --blooper-marker with --cleanup exact is a
82
+ // contradiction, and the flag the user typed second does not get to win.
56
83
  if (level === "exact") return keepAll;
57
84
  const policy = POLICIES[level];
58
85
  const words = transcript.words;
@@ -62,17 +89,49 @@ export function buildCutlist({ transcript, analysis, duration, level }: BuildCut
62
89
 
63
90
  const removals: Removal[] = [];
64
91
 
92
+ // Marked bloopers, injected BEFORE the sort/merge below so they inherit the
93
+ // whole existing machine: merging with the silence that brackets the flub,
94
+ // MIN_KEEP sliver folding, and the partition emit. Source is "acoustic"
95
+ // because the boundaries are word stamps we chose deliberately — the
96
+ // protected-word pass must not push them back off the words they exist to
97
+ // remove.
98
+ for (const b of bloops ?? []) {
99
+ removals.push({
100
+ start: b.startSec,
101
+ end: b.endSec,
102
+ reason: "retake",
103
+ confidence: 1,
104
+ source: "acoustic",
105
+ });
106
+ }
107
+
65
108
  for (const pause of analysis.cuttable) {
66
- const isLead = first !== undefined && pause.end <= first.start + 1e-6;
67
- const isTail = last !== undefined && pause.start >= last.end - 1e-6;
109
+ // Lead and tail are decided by the SILENCE's position in the file, not by
110
+ // comparing it to a word stamp (R27 §127). Whisper's `-ml 1` stamps stretch
111
+ // to fill gaps: on a real take the first word was stamped 0.00–0.53 over
112
+ // silence that plainly starts at 0.00, so `pause.end <= first.start` was
113
+ // false and the opening dead air fell through to the interior rule — where
114
+ // it was under `pauseMin` and survived. The tail failed the same way, by a
115
+ // 0.07s overlap, leaving the speaker on screen looking down after the last
116
+ // word. Dead air touching either end of the file IS lead/tail, whatever the
117
+ // recognizer claims about where words begin.
118
+ const isLead = pause.start <= 1e-6;
119
+ const isTail = pause.end >= duration - 1e-6;
68
120
  if (isLead) {
69
- // Trim dead air before the first word down to LEAD_KEEP (hook starts fast).
70
- const end = Math.min(pause.end, (first?.start ?? pause.end) - LEAD_KEEP);
121
+ // Keep LEAD_KEEP of run-up before speech starts (hook starts fast).
122
+ // Measured back from the END of the silence — where speech actually
123
+ // begins — rather than from a word stamp that may cover the silence.
124
+ const speechStarts = first !== undefined ? Math.max(pause.end, first.start) : pause.end;
125
+ const end = Math.min(pause.end, speechStarts - policy.leadKeep);
71
126
  if (end - pause.start >= MIN_REMOVAL) {
72
127
  removals.push({ start: pause.start, end, reason: "silence", confidence: 0.95, source: "acoustic" });
73
128
  }
74
129
  } else if (isTail) {
75
- const start = Math.max(pause.start, (last?.end ?? pause.start) + TAIL_KEEP);
130
+ // Same, mirrored: the take ends when the speech does, so keep TAIL_KEEP
131
+ // past the last word and drop everything after — including the pause the
132
+ // recognizer's final stamp bled into.
133
+ const speechEnds = last !== undefined ? Math.min(pause.start, last.end) : pause.start;
134
+ const start = Math.max(pause.start, speechEnds + policy.tailKeep);
76
135
  if (pause.end - start >= MIN_REMOVAL) {
77
136
  removals.push({ start, end: pause.end, reason: "silence", confidence: 0.95, source: "acoustic" });
78
137
  }
package/src/grounding.ts CHANGED
@@ -88,7 +88,17 @@ function needsSupport(token: string): boolean {
88
88
  function stringsOf(value: unknown): string[] {
89
89
  if (typeof value === "string") return [value];
90
90
  if (Array.isArray(value)) return value.flatMap(stringsOf);
91
- if (value && typeof value === "object") return Object.values(value).flatMap(stringsOf);
91
+ if (value && typeof value === "object") {
92
+ // A structured line carries its copy in `text`; its siblings are RENDERING
93
+ // DIRECTIVES, not words anyone reads (R27 §126). Walking every value made
94
+ // the check judge them as on-screen copy, so a StrikethroughReveal line
95
+ // `{text, struck, mark: "cross"}` was reported as inventing "cross" — and
96
+ // the take can never contain those tokens, so the warning was unfixable
97
+ // by construction. Two of the four warnings on a real render were this.
98
+ const text = (value as Record<string, unknown>).text;
99
+ if (typeof text === "string") return [text];
100
+ return Object.values(value).flatMap(stringsOf);
101
+ }
92
102
  return [];
93
103
  }
94
104
 
package/src/index.ts CHANGED
@@ -11,6 +11,7 @@ export * from "./transcribe";
11
11
  export * from "./analyze";
12
12
  export * from "./cutlist";
13
13
  export * from "./clip";
14
+ export * from "./blooper";
14
15
  export * from "./captions";
15
16
  export * from "./zoom";
16
17
  export * from "./grounding";
package/src/ingest.ts CHANGED
@@ -6,6 +6,27 @@ export interface IngestTools {
6
6
  ffprobePath: string;
7
7
  }
8
8
 
9
+ /**
10
+ * The stream's rotation, normalized to 0/90/180/270 (R27 §119).
11
+ *
12
+ * Two spellings, because containers disagree: a Display Matrix side-datum
13
+ * (modern ffprobe, and the only one a concatenated MP4 keeps) or the legacy
14
+ * `rotate` tag. ffprobe reports the matrix angle signed — -90 and 270 are the
15
+ * same quarter turn — so everything is folded into [0, 360).
16
+ */
17
+ export function normalizeRotation(raw: number | string | undefined): number {
18
+ const n = typeof raw === "string" ? Number(raw) : raw;
19
+ if (n === undefined || !Number.isFinite(n)) return 0;
20
+ const deg = ((Math.round(n) % 360) + 360) % 360;
21
+ // Anything that is not a quarter turn cannot swap an axis; treat as upright.
22
+ return deg % 90 === 0 ? deg : 0;
23
+ }
24
+
25
+ /** A quarter turn exchanges the axes, so the DISPLAYED frame is w/h swapped. */
26
+ export function rotationSwapsAxes(rotation: number): boolean {
27
+ return rotation === 90 || rotation === 270;
28
+ }
29
+
9
30
  export async function probe(tools: IngestTools, path: string): Promise<Probe> {
10
31
  const { stdout } = await run(tools.ffprobePath, [
11
32
  "-v", "error",
@@ -21,6 +42,8 @@ export async function probe(tools: IngestTools, path: string): Promise<Probe> {
21
42
  height?: number;
22
43
  avg_frame_rate?: string;
23
44
  r_frame_rate?: string;
45
+ side_data_list?: Array<{ rotation?: number }>;
46
+ tags?: { rotate?: string };
24
47
  }>;
25
48
  format?: { duration?: string };
26
49
  };
@@ -31,12 +54,25 @@ export async function probe(tools: IngestTools, path: string): Promise<Probe> {
31
54
  const [num, den] = (rate ?? "30/1").split("/").map(Number);
32
55
  const duration = Number(info.format?.duration);
33
56
  if (!Number.isFinite(duration) || duration <= 0) throw new Error(`could not determine duration of ${path}`);
57
+ // `side_data_list` carries several kinds of datum (ambient viewing
58
+ // environment, content light level); only one of them has a rotation.
59
+ const matrix = video.side_data_list?.find((s) => typeof s.rotation === "number");
60
+ const rotation = normalizeRotation(matrix?.rotation ?? video.tags?.rotate);
61
+ const rawW = video.width ?? 0;
62
+ const rawH = video.height ?? 0;
63
+ // Report what is DISPLAYED. ffmpeg auto-rotates in the filter chain, so every
64
+ // measurement taken through it (cropdetect, face, the mezzanine) is already
65
+ // in this space; returning the raw stream size made the pipeline reconcile
66
+ // two orientations into a bogus square and "detect" a letterbox on a
67
+ // full-frame portrait take (R27 §119).
68
+ const swap = rotationSwapsAxes(rotation);
34
69
  return {
35
70
  duration,
36
- width: video.width ?? 0,
37
- height: video.height ?? 0,
71
+ width: swap ? rawH : rawW,
72
+ height: swap ? rawW : rawH,
38
73
  fps: den ? (num ?? 30) / den : 30,
39
74
  hasAudio: Boolean(audio),
75
+ ...(rotation !== 0 ? { rotation } : {}),
40
76
  };
41
77
  }
42
78
 
package/src/normalize.ts CHANGED
@@ -1,3 +1,4 @@
1
+ import { rename, rm } from "node:fs/promises";
1
2
  import { run } from "./exec";
2
3
  import type { ContentRectSegment } from "./content-rect";
3
4
  import type { WindowFace } from "./face";
@@ -368,7 +369,14 @@ export function normalizationFilterGraph(plan: NormalizePlan): string {
368
369
  return (
369
370
  `[0:v]trim=start=${s.startSec.toFixed(3)}:end=${s.endSec.toFixed(3)},` +
370
371
  `setpts=PTS-STARTPTS,crop=${w.w}:${w.h}:${w.x}:${w.y},` +
371
- `scale=${plan.canvas.width}:${plan.canvas.height}[v${i}]`
372
+ // setsar=1 is load-bearing, not tidiness (R27 §125). Every segment is
373
+ // scaled to the SAME canvas, but from a DIFFERENT crop, and ffmpeg
374
+ // derives a sample aspect from that ratio: a 946x1682 crop yields SAR
375
+ // 1683:1682 and a 932x1660 crop 1377:1376. `concat` requires identical
376
+ // SAR across inputs and aborts the whole bake when they disagree, so a
377
+ // take whose framing varies — exactly the take normalization exists
378
+ // for — failed to render at all.
379
+ `scale=${plan.canvas.width}:${plan.canvas.height},setsar=1[v${i}]`
372
380
  );
373
381
  });
374
382
  const inputs = plan.segments.map((_, i) => `[v${i}]`).join("");
@@ -386,12 +394,27 @@ export async function bakeNormalizedSource(
386
394
  plan: NormalizePlan,
387
395
  outPath: string,
388
396
  ): Promise<void> {
389
- await run(tools.ffmpegPath, [
390
- "-y", "-i", input,
391
- "-filter_complex", normalizationFilterGraph(plan),
392
- "-map", "[v]", "-map", "0:a?",
393
- "-c:v", "libx264", "-preset", "veryfast", "-crf", "18", "-g", "30", "-pix_fmt", "yuv420p",
394
- "-c:a", "aac", "-b:a", "192k",
395
- outPath,
396
- ]);
397
+ // Encode to a sibling temp path and rename only on success (R27 §125).
398
+ // ffmpeg writes the container header as it goes, so a bake that dies
399
+ // mid-graph leaves a file with no `moov` atom — and the cache upstream keys
400
+ // on EXISTENCE, so that corpse is then reused as a valid normalized source
401
+ // on every later run. The failure surfaces as "moov atom not found" from a
402
+ // step that never ran, and deleting the workdir is the only way out. Rename
403
+ // is atomic on a POSIX filesystem, so the cache can only ever see a file
404
+ // ffmpeg finished writing.
405
+ const partial = `${outPath}.partial.mp4`;
406
+ try {
407
+ await run(tools.ffmpegPath, [
408
+ "-y", "-i", input,
409
+ "-filter_complex", normalizationFilterGraph(plan),
410
+ "-map", "[v]", "-map", "0:a?",
411
+ "-c:v", "libx264", "-preset", "veryfast", "-crf", "18", "-g", "30", "-pix_fmt", "yuv420p",
412
+ "-c:a", "aac", "-b:a", "192k",
413
+ partial,
414
+ ]);
415
+ await rename(partial, outPath);
416
+ } catch (err) {
417
+ await rm(partial, { force: true });
418
+ throw err;
419
+ }
397
420
  }
@@ -6,13 +6,39 @@ import { MAX_SCENE_SEC } from "../assemble";
6
6
  import { COVER_MAX_WORDS, coverHeadline } from "../cover";
7
7
  import type { LlmProvider } from "./provider";
8
8
 
9
+ /**
10
+ * Free text from the model, capped rather than refused (R27 §123).
11
+ *
12
+ * The standing doctrine (§112) is that LLM output is untrusted input,
13
+ * "validated where the pipeline can still degrade instead of at the point
14
+ * where it can only die". A bare `.max(n)` is the second kind: on two of three
15
+ * real runs the editorial call came back with a 61-character `onScreenCopy`
16
+ * and the whole produce died at the Zod boundary — transcription, analysis and
17
+ * the cut all discarded over one character of a headline.
18
+ *
19
+ * `preprocess` keeps `maxLength: n` in the JSON schema the provider is handed,
20
+ * so the model is still ASKED for the limit; it just no longer costs a run
21
+ * when the model misses by a word. Truncation prefers the last word boundary,
22
+ * and adds no ellipsis — the prompt explicitly forbids one on cover text.
23
+ */
24
+ export function cappedText(max: number): z.ZodType<string> {
25
+ return z.preprocess((v) => {
26
+ if (typeof v !== "string" || v.length <= max) return v;
27
+ const cut = v.slice(0, max);
28
+ const lastSpace = cut.lastIndexOf(" ");
29
+ // Only honour a word boundary that keeps most of the budget; a single very
30
+ // long word would otherwise collapse to nothing.
31
+ return (lastSpace > max * 0.6 ? cut.slice(0, lastSpace) : cut).trimEnd();
32
+ }, z.string().max(max)) as z.ZodType<string>;
33
+ }
34
+
9
35
  /** Call 1 — the editorial call (PHASE1 §4): moments, copy, component picks. */
10
36
  export const MomentSchema = z.object({
11
37
  startWord: z.number().int().nonnegative(),
12
38
  endWord: z.number().int().nonnegative(),
13
- purpose: z.string().max(100),
39
+ purpose: cappedText(100),
14
40
  /** Short on-screen copy for this beat — the fallback TitleCard title. */
15
- onScreenCopy: z.string().max(60),
41
+ onScreenCopy: cappedText(60),
16
42
  /** "none" = plain talking head with captions; otherwise a library component. */
17
43
  sceneKind: z.union([SceneComponentIdSchema, z.literal("none")]),
18
44
  /**
@@ -24,25 +50,30 @@ export const MomentSchema = z.object({
24
50
  layout: LayoutSchema.optional().describe(
25
51
  "stage layout for this scene; omit for the component default. NEVER a layout the framing brief marks UNAVAILABLE for these words",
26
52
  ),
27
- rationale: z.string().max(120).optional(),
53
+ rationale: cappedText(120).optional(),
28
54
  });
29
55
  export type Moment = z.infer<typeof MomentSchema>;
30
56
 
31
57
  export const BeatSheetSchema = z.object({
32
- hook: z.string().max(120),
58
+ hook: cappedText(120),
33
59
  /**
34
60
  * Banner text for the cover image (FINDINGS §31). Written here rather than
35
61
  * by a second LLM call, because the producer is already choosing the hook —
36
62
  * this is the same editorial judgement, shortened for a thumbnail.
37
63
  */
38
- coverText: z
39
- .string()
40
- .max(60)
64
+ coverText: cappedText(60)
41
65
  .optional()
42
66
  .describe(
43
67
  `cover banner: at most ${COVER_MAX_WORDS} words, the hook compressed to a thumbnail headline`,
44
68
  ),
45
- moments: z.array(MomentSchema).min(1).max(12),
69
+ /**
70
+ * Raised from 12 to 24 (§118): with the alternation policy above, a cap of
71
+ * 12 moments is a ceiling of ~6 graphics however long the take is. A 64s
72
+ * take enumerating five features needs seven graphic beats — hook, five
73
+ * features, payoff — and therefore ~14 moments to alternate between them.
74
+ * The cap was binding before the coverage budget ever was.
75
+ */
76
+ moments: z.array(MomentSchema).min(1).max(24),
46
77
  });
47
78
  export type BeatSheet = z.infer<typeof BeatSheetSchema>;
48
79
 
@@ -77,7 +108,8 @@ Virality grammar — follow these as hard policies:
77
108
  - Use contrast/negation beats (StrikethroughReveal, RuleCard with struck alternatives) when the speaker rejects an idea.
78
109
  - End with a payoff or takeaway moment.
79
110
  - A moment spans the FULL stretch of speech about its beat — typically 5-15 seconds — in transcript order, non-overlapping. The graphic stays on screen for the ENTIRE moment, so the word range must cover everything the graphic refers to: a stat card leaves when the speaker moves on, not before.
80
- - COVERAGE: graphics should be on screen for roughly 40-50% of the runtime. A graphic spends its whole moment against that budget, so be selective — give graphics to the moments where one genuinely earns the frame, and spread them evenly: never leave a stretch longer than ~10 seconds with no graphic.
111
+ - COUNT: the user prompt states how many graphic moments this take should get. That number is a TARGET, not a maximum — hit it. Planning under it is the most common failure: a take that makes five distinct points and gets two graphics has been under-produced, whatever the coverage percentage says.
112
+ - COVERAGE: graphics should be on screen for roughly 40-50% of the runtime. A graphic spends its whole moment against that budget, so when the target implies many graphics, make each moment SHORTER rather than dropping moments — more, tighter graphics beats fewer, longer ones. Spread them evenly: never leave a stretch longer than ~10 seconds with no graphic.
81
113
  - VARIETY: never the same component twice in a row, and prefer a component you have NOT used yet in this video — reuse a treatment only when the beat genuinely calls for it. A repeat reads as a template.
82
114
  - Keep the face LARGE: prefer StatCard/RuleCard/ScreenshotFrame (they sit under a big face) over TitleCard (face becomes a small bubble); use FlowDiagram/TerminalMock sparingly — they remove the face entirely and only earn that when the graphic IS the point.
83
115
  - The transcript is ASR output and may contain mishearings: an unfamiliar proper noun is more likely a mistranscription of a common phrase than a real entity — write on-screen copy with the common-sense reading, never a suspected mishearing.
@@ -117,11 +149,23 @@ export function buildBeatsUserPrompt(
117
149
  const menu = Object.entries(SCENE_REGISTRY)
118
150
  .map(([id, meta]) => `- ${id}: ${meta.whenToUse}`)
119
151
  .join("\n");
152
+ // §118: state the graphic COUNT explicitly. Everything else in this prompt
153
+ // describes what a good graphic is; nothing said how many to plan, and the
154
+ // coverage budget downstream only ever removes.
155
+ const enumerated = countEnumeratedBeats(transcript);
156
+ const target = graphicsTarget(clip?.targetSec ?? duration, enumerated);
157
+ const targetLine =
158
+ `Graphic moments to plan: ${target}` +
159
+ (enumerated > 0
160
+ ? ` — the speaker enumerates ${enumerated} points out loud, so each one earns its own graphic, plus a hook and a payoff.\n`
161
+ : ` (about one per ${SEC_PER_GRAPHIC}s of runtime). Plan this many unless the take genuinely cannot carry them.\n`);
120
162
  return (
121
163
  `Intent: ${intent ?? "make this clear, punchy and viral-worthy"}\n` +
122
164
  (clip
123
- ? `Target clip length: ~${clip.targetSec.toFixed(0)}s (see CLIP SELECTION below)\n\n`
124
- : `Output duration after the cut: ${duration.toFixed(1)}s\n\n`) +
165
+ ? `Target clip length: ~${clip.targetSec.toFixed(0)}s (see CLIP SELECTION below)\n`
166
+ : `Output duration after the cut: ${duration.toFixed(1)}s\n`) +
167
+ targetLine +
168
+ "\n" +
125
169
  // Landscape layout guidance (R21 §101): without it the first real 16:9
126
170
  // run put nearly every graphic in a lower third. A deterministic variety
127
171
  // pass downstream is the guarantee; this is the steer.
@@ -146,6 +190,87 @@ export interface BeatsValidationIssue {
146
190
  issue: string;
147
191
  }
148
192
 
193
+ /**
194
+ * How many graphics a take of this length should be asked for (§118).
195
+ *
196
+ * The failure this exists for: nothing ever told the producer how many
197
+ * graphics to plan. `GRAPHICS_COVERAGE_TARGET` reads like a target and is
198
+ * only a ceiling — the demote loop below runs when the model plans too MANY
199
+ * and does nothing at all when it plans too few. On one 64s take the model
200
+ * planned three graphics against a budget that allowed roughly twenty-nine
201
+ * seconds of them; the loop never executed once, so no existing mechanism
202
+ * had an opinion.
203
+ *
204
+ * One graphic per ~9s of runtime, which is the density the prompt's own
205
+ * "never leave a stretch longer than ~10 seconds with no graphic" rule
206
+ * implies, floored at the §29 short-take count.
207
+ */
208
+ export const SEC_PER_GRAPHIC = 9;
209
+
210
+ /**
211
+ * Ordinal cues a speaker uses to enumerate. A take that counts its own
212
+ * points out loud is telling us how many graphics it wants, and that signal
213
+ * is free, deterministic, and better than any runtime heuristic.
214
+ */
215
+ const ORDINAL_WORDS = [
216
+ "one", "two", "three", "four", "five", "six", "seven", "eight", "nine", "ten",
217
+ ];
218
+ const ORDINAL_ADJECTIVES = [
219
+ "first", "second", "third", "fourth", "fifth",
220
+ "sixth", "seventh", "eighth", "ninth", "tenth",
221
+ ];
222
+
223
+ /**
224
+ * How many distinct enumerated beats the speaker announces — "number one …
225
+ * number two", "first … second", "step 3". Counts DISTINCT ordinals so a
226
+ * speaker who says "number two" twice doesn't inflate the target, and
227
+ * requires at least two so a passing "first of all" isn't read as a list.
228
+ */
229
+ export function countEnumeratedBeats(transcript: Transcript): number {
230
+ const words = transcript.words.map((w) =>
231
+ w.text.toLowerCase().replace(/[^a-z0-9]/g, ""),
232
+ );
233
+ const seen = new Set<number>();
234
+ for (let i = 0; i < words.length; i++) {
235
+ const w = words[i]!;
236
+ const adj = ORDINAL_ADJECTIVES.indexOf(w);
237
+ if (adj !== -1) {
238
+ seen.add(adj + 1);
239
+ continue;
240
+ }
241
+ // "number one" / "step 2" / "point three" — the ordinal must FOLLOW a
242
+ // counting noun, or every stray "one" in the take counts as a beat.
243
+ if (w !== "number" && w !== "step" && w !== "point" && w !== "tip") continue;
244
+ const next = words[i + 1];
245
+ if (!next) continue;
246
+ const spelled = ORDINAL_WORDS.indexOf(next);
247
+ if (spelled !== -1) {
248
+ seen.add(spelled + 1);
249
+ continue;
250
+ }
251
+ const digit = Number.parseInt(next, 10);
252
+ if (Number.isInteger(digit) && digit >= 1 && digit <= 10) seen.add(digit);
253
+ }
254
+ return seen.size >= 2 ? seen.size : 0;
255
+ }
256
+
257
+ /**
258
+ * The number of graphics to ASK for — whichever of structure and runtime is
259
+ * larger. An enumerated take earns its own count plus a hook and a payoff
260
+ * (the virality grammar demands both anyway), but a long take that happens
261
+ * to enumerate three points still has everything else in it, so runtime
262
+ * density is a floor rather than a loser.
263
+ */
264
+ export function graphicsTarget(runtimeSec: number, enumerated: number): number {
265
+ const byRuntime = Math.max(
266
+ SHORT_TAKE_MIN_GRAPHICS,
267
+ Math.round(runtimeSec / SEC_PER_GRAPHIC),
268
+ );
269
+ const byStructure = enumerated > 0 ? enumerated + 2 : 0;
270
+ // Never more than the moment schema can carry once alternation is counted.
271
+ return Math.min(Math.max(byRuntime, byStructure), 12);
272
+ }
273
+
149
274
  /** Fraction of the runtime that should show a graphic (FINDINGS §7). */
150
275
  export const GRAPHICS_COVERAGE_TARGET = 0.45;
151
276
  /**
@@ -157,6 +282,28 @@ export const GRAPHICS_COVERAGE_TARGET = 0.45;
157
282
  export const SHORT_TAKE_SEC = 45;
158
283
  export const SHORT_TAKE_MIN_GRAPHICS = 4;
159
284
 
285
+ /** Why the target was what it was, when the take enumerated itself. */
286
+ function enumeratedNote(transcript: Transcript): string | null {
287
+ const n = countEnumeratedBeats(transcript);
288
+ return n > 0 ? ` — the take enumerates ${n} points` : null;
289
+ }
290
+
291
+ /**
292
+ * The one-line graphics accounting (§118b): delivered vs asked, and why the
293
+ * ask was what it was. One formatter for the console issue and `report.txt`,
294
+ * so the two can never say different things about the same run.
295
+ */
296
+ export function formatGraphicsAccounting(
297
+ delivered: number,
298
+ asked: number,
299
+ transcript: Transcript,
300
+ ): string {
301
+ return (
302
+ `graphics: ${delivered} of ${asked} planned` +
303
+ (enumeratedNote(transcript) ?? ` (target is ~1 per ${SEC_PER_GRAPHIC}s)`)
304
+ );
305
+ }
306
+
160
307
  /** A moment's approximate seconds of speech, from the transcript word stamps. */
161
308
  function momentDuration(m: Moment, transcript: Transcript): number {
162
309
  const first = transcript.words[m.startWord];
@@ -170,10 +317,21 @@ function momentMidpoint(m: Moment, transcript: Transcript): number {
170
317
  return first && last ? (first.start + last.end) / 2 : 0;
171
318
  }
172
319
 
173
- /** Semantic validation beyond the schema; repairs what it can, reports the rest. */
320
+ /**
321
+ * Semantic validation beyond the schema; repairs what it can, reports the rest.
322
+ *
323
+ * `askedGraphics` is the count the PROMPT stated (§118b): pass it so the
324
+ * shortfall check measures against what was actually asked for — on a clip
325
+ * run the internal fallback would measure against the full take's runtime,
326
+ * not the clip target the prompt named. `null` skips the check entirely (the
327
+ * pre-slice pass of a clip run, whose sheet is renormalized after slicing —
328
+ * two passes reporting the same shortfall would say it twice). Omitted, the
329
+ * ask is derived from the transcript's own span.
330
+ */
174
331
  export function normalizeBeatSheet(
175
332
  sheet: BeatSheet,
176
333
  transcript: Transcript,
334
+ askedGraphics?: number | null,
177
335
  ): { sheet: BeatSheet; issues: BeatsValidationIssue[] } {
178
336
  const wordCount = transcript.words.length;
179
337
  const issues: BeatsValidationIssue[] = [];
@@ -225,6 +383,15 @@ export function normalizeBeatSheet(
225
383
 
226
384
  // On a short take the count floor outranks the percentage — never demote
227
385
  // below it, whatever the coverage budget says (§29).
386
+ //
387
+ // §118 decided NOT to extend this floor above 45s, and the reason matters:
388
+ // a floor that outranks the ceiling at every length would fight §114's
389
+ // full-span pricing — more graphics × whole moments blows past 45%, the
390
+ // loop below starts removing what the floor just required, and the two
391
+ // rules oscillate. It would also be treating the wrong failure. When the
392
+ // producer UNDER-plans, this loop never runs at all, so no floor here
393
+ // could have helped; the fix is the target in the prompt. What this layer
394
+ // owes the user instead is to SAY so — see the shortfall issue below.
228
395
  const minGraphics = runtime < SHORT_TAKE_SEC ? SHORT_TAKE_MIN_GRAPHICS : 0;
229
396
 
230
397
  for (;;) {
@@ -293,6 +460,22 @@ export function normalizeBeatSheet(
293
460
  issues.push({ moment: -1, issue: `coverText shortened to "${coverText}"` });
294
461
  }
295
462
 
463
+ // §118b: a run that under-delivers must say so. The producer was asked for
464
+ // a specific number of graphics; if fewer survive, that is a fact about
465
+ // this render the report should carry, exactly as every cut is justified.
466
+ // Silence is what let three graphics on a five-point take look normal.
467
+ const delivered = surviving().length;
468
+ const asked =
469
+ askedGraphics === undefined
470
+ ? graphicsTarget(runtime, countEnumeratedBeats(transcript))
471
+ : askedGraphics;
472
+ if (asked !== null && delivered < asked) {
473
+ issues.push({
474
+ moment: -1,
475
+ issue: formatGraphicsAccounting(delivered, asked, transcript),
476
+ });
477
+ }
478
+
296
479
  return { sheet: { hook: sheet.hook, coverText, moments }, issues };
297
480
  }
298
481
 
@@ -305,10 +488,22 @@ export async function generateBeatSheet(
305
488
  framingBrief?: string,
306
489
  clip?: { targetSec: number },
307
490
  aspect?: "9:16" | "16:9",
308
- ): Promise<{ sheet: BeatSheet; issues: BeatsValidationIssue[]; highlight?: ClipHighlight }> {
491
+ ): Promise<{
492
+ sheet: BeatSheet;
493
+ issues: BeatsValidationIssue[];
494
+ /** The graphic count the prompt asked for (§118b) — what "asked" means everywhere downstream. */
495
+ asked: number;
496
+ highlight?: ClipHighlight;
497
+ }> {
309
498
  const user =
310
499
  (speaker ? `The speaker: ${speaker}\n\n` : "") +
311
500
  buildBeatsUserPrompt(transcript, duration, intent, framingBrief, clip, aspect);
501
+ // The same number `buildBeatsUserPrompt` states — computed from the same
502
+ // inputs by the same pure functions, so the check and the ask agree.
503
+ const asked = graphicsTarget(
504
+ clip?.targetSec ?? duration,
505
+ countEnumeratedBeats(transcript),
506
+ );
312
507
  if (clip) {
313
508
  // Same editorial call, extended schema (R19 §93d) — the highlight and the
314
509
  // beat sheet come from ONE judgement, so they cannot disagree.
@@ -318,7 +513,8 @@ export async function generateBeatSheet(
318
513
  schema: ClipBeatSheetSchema,
319
514
  schemaName: "clip_beat_sheet",
320
515
  });
321
- return { ...normalizeBeatSheet(raw, transcript), highlight: raw.highlight };
516
+ // `null`: the post-slice renormalization owns the shortfall check.
517
+ return { ...normalizeBeatSheet(raw, transcript, null), asked, highlight: raw.highlight };
322
518
  }
323
519
  const raw = await provider.complete({
324
520
  system: PRODUCER_SYSTEM,
@@ -326,5 +522,5 @@ export async function generateBeatSheet(
326
522
  schema: BeatSheetSchema,
327
523
  schemaName: "beat_sheet",
328
524
  });
329
- return normalizeBeatSheet(raw, transcript);
525
+ return { ...normalizeBeatSheet(raw, transcript, asked), asked };
330
526
  }
@@ -106,6 +106,13 @@ export function defaultProviderName(env: NodeJS.ProcessEnv = process.env): Provi
106
106
  export interface ProduceScenesResult {
107
107
  beatSheet: BeatSheet;
108
108
  beatIssues: BeatsValidationIssue[];
109
+ /**
110
+ * The graphics accounting (§118b): how many the prompt asked for and how
111
+ * many survived planning. `delivered` equals the scene count — layout
112
+ * repair never demotes, and a failed props call falls back to a TitleCard
113
+ * rather than dropping the scene.
114
+ */
115
+ graphics: { asked: number; delivered: number };
109
116
  scenes: Scene[];
110
117
  failures: ScenePropsFailure[];
111
118
  /**
@@ -154,7 +161,7 @@ export async function produceScenes(
154
161
  const framingBrief = args.framing
155
162
  ? buildFramingBrief(args.framing, args.transcript)
156
163
  : undefined;
157
- const { sheet, issues, highlight } = await generateBeatSheet(
164
+ const { sheet, issues, asked, highlight } = await generateBeatSheet(
158
165
  provider,
159
166
  args.transcript,
160
167
  args.outputDuration,
@@ -187,6 +194,9 @@ export async function produceScenes(
187
194
  const renorm = normalizeBeatSheet(
188
195
  { hook: sheet.hook, coverText: sheet.coverText, moments: anchored },
189
196
  transcript,
197
+ // The ask the prompt stated — NOT re-derived from the slice, which
198
+ // would compare the model against a number it was never given (§118b).
199
+ asked,
190
200
  );
191
201
  workingSheet = renorm.sheet;
192
202
  issues.push(...renorm.issues);
@@ -213,5 +223,13 @@ export async function produceScenes(
213
223
  const { scenes, failures } = await generateScenes(provider, moments, transcript, {
214
224
  framing: args.framing,
215
225
  });
216
- return { beatSheet: { ...workingSheet, moments }, beatIssues: issues, scenes, failures, clip };
226
+ const delivered = moments.filter((m) => m.sceneKind !== "none").length;
227
+ return {
228
+ beatSheet: { ...workingSheet, moments },
229
+ beatIssues: issues,
230
+ graphics: { asked, delivered },
231
+ scenes,
232
+ failures,
233
+ clip,
234
+ };
217
235
  }
package/src/schema.ts CHANGED
@@ -71,10 +71,22 @@ export type Analysis = z.infer<typeof AnalysisSchema>;
71
71
 
72
72
  export const ProbeSchema = z.object({
73
73
  duration: z.number().positive(),
74
+ /**
75
+ * DISPLAYED dimensions, after the rotation matrix (R27 §119) — not the raw
76
+ * stream's. A phone/camera writes a portrait take as a landscape stream plus
77
+ * a 90° display matrix, and ffmpeg's filter chain auto-rotates, so the raw
78
+ * numbers disagree with every measurement taken through ffmpeg.
79
+ */
74
80
  width: z.number().int().positive(),
75
81
  height: z.number().int().positive(),
76
82
  fps: z.number().positive(),
77
83
  hasAudio: z.boolean(),
84
+ /**
85
+ * The stream's rotation in degrees (0/90/180/270), recorded so a workdir says
86
+ * why its geometry is what it is. Optional: pre-§119 `production.json` files
87
+ * predate it and must still parse.
88
+ */
89
+ rotation: z.number().int().optional(),
78
90
  });
79
91
  export type Probe = z.infer<typeof ProbeSchema>;
80
92