reelkit-cli 0.5.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/README.md +3 -2
  2. package/package.json +7 -2
  3. package/skill/SKILL.md +22 -11
  4. package/skill/THIRD_PARTY.md +102 -0
  5. package/skill/commands/launch-film.md +7 -0
  6. package/skill/reference/asset-reuse.md +13 -2
  7. package/skill/reference/backgrounds.md +63 -0
  8. package/skill/reference/beat-sync.md +32 -17
  9. package/skill/reference/captions.md +11 -5
  10. package/skill/reference/component-authoring.md +12 -1
  11. package/skill/reference/continuity.md +21 -2
  12. package/skill/reference/kit.md +140 -11
  13. package/skill/reference/launch-film.md +190 -0
  14. package/skill/reference/remotion-composition.md +4 -3
  15. package/skill/reference/scene-treatments.md +20 -0
  16. package/skill/reference/scriptwriting.md +4 -1
  17. package/skill/reference/three-d.md +134 -0
  18. package/skill/reference/voice-sync.md +108 -0
  19. package/src/agents.ts +23 -12
  20. package/src/api/client.ts +4 -1
  21. package/src/cli.ts +25 -8
  22. package/src/commands/assets.ts +334 -36
  23. package/src/commands/build.ts +172 -36
  24. package/src/commands/components.ts +220 -0
  25. package/src/commands/init.ts +9 -3
  26. package/src/commands/install.ts +1 -1
  27. package/src/commands/plan.ts +8 -5
  28. package/src/commands/ref.ts +5 -2
  29. package/src/contract/index.ts +27 -5
  30. package/src/pipeline/beatsnap.ts +72 -0
  31. package/src/pipeline/review.ts +67 -7
  32. package/src/pipeline/schema.ts +51 -4
  33. package/src/pipeline/timing.ts +27 -1
  34. package/src/project/background.ts +33 -0
  35. package/src/project/layers.ts +60 -0
  36. package/src/project/loudness.ts +68 -0
  37. package/src/project/manifest.ts +68 -14
  38. package/src/project/music.ts +19 -5
  39. package/src/project/project.ts +7 -2
  40. package/src/project/refmeasure.ts +1 -1
  41. package/src/project/soundreport.ts +347 -0
  42. package/src/project/svgcheck.ts +21 -0
  43. package/src/remotion/kit/Assemble3D.tsx +92 -0
  44. package/src/remotion/kit/BrowserFrame.tsx +83 -0
  45. package/src/remotion/kit/Camera.tsx +6 -4
  46. package/src/remotion/kit/Captions.tsx +33 -17
  47. package/src/remotion/kit/Card3D.tsx +211 -0
  48. package/src/remotion/kit/ChapterFrame.tsx +68 -0
  49. package/src/remotion/kit/CounterRoll.tsx +75 -0
  50. package/src/remotion/kit/GlassPanel.tsx +43 -0
  51. package/src/remotion/kit/Grounds.tsx +177 -0
  52. package/src/remotion/kit/Headline.tsx +97 -0
  53. package/src/remotion/kit/Hero3D.tsx +197 -0
  54. package/src/remotion/kit/HudOverlay.tsx +52 -0
  55. package/src/remotion/kit/ImageLayers.tsx +48 -0
  56. package/src/remotion/kit/Music.tsx +4 -4
  57. package/src/remotion/kit/NamedCursor.tsx +54 -0
  58. package/src/remotion/kit/Orbit3D.tsx +49 -0
  59. package/src/remotion/kit/Particles3D.tsx +74 -0
  60. package/src/remotion/kit/Place.tsx +12 -0
  61. package/src/remotion/kit/PromptBox.tsx +84 -0
  62. package/src/remotion/kit/Scene3D.tsx +70 -0
  63. package/src/remotion/kit/SceneFrame.tsx +88 -11
  64. package/src/remotion/kit/Sfx.tsx +12 -6
  65. package/src/remotion/kit/SoundCues.tsx +22 -0
  66. package/src/remotion/kit/TerminalLog.tsx +98 -0
  67. package/src/remotion/kit/Text3D.tsx +78 -0
  68. package/src/remotion/kit/TextOnImage.tsx +41 -0
  69. package/src/remotion/kit/Warp3D.tsx +59 -0
  70. package/src/remotion/kit/bg-math.ts +179 -0
  71. package/src/remotion/kit/caption-groups.ts +7 -3
  72. package/src/remotion/kit/caption-style.ts +45 -0
  73. package/src/remotion/kit/docs.ts +133 -11
  74. package/src/remotion/kit/image-layers-math.ts +115 -0
  75. package/src/remotion/kit/index.ts +43 -1
  76. package/src/remotion/kit/inter-bold-typeface.ts +3 -0
  77. package/src/remotion/kit/media.ts +5 -3
  78. package/src/remotion/kit/motion-math.ts +36 -2
  79. package/src/remotion/kit/music-math.ts +27 -10
  80. package/src/remotion/kit/quiet-three.ts +11 -0
  81. package/src/remotion/kit/sample-text.ts +55 -0
  82. package/src/remotion/kit/scene3d-context.ts +5 -0
  83. package/src/remotion/kit/seeded.ts +13 -0
  84. package/src/remotion/kit/sound-cues.ts +89 -0
  85. package/src/remotion/kit/sound-kinds.ts +122 -0
  86. package/src/remotion/kit/theme.ts +2 -0
  87. package/src/remotion/kit/three-fx-math.ts +192 -0
  88. package/src/remotion/kit/three-math.ts +145 -0
  89. package/src/remotion/kit/transition-math.ts +116 -0
  90. package/src/remotion/kit/ui-math.ts +145 -0
  91. package/src/remotion/kit/ui-theme.ts +25 -0
  92. package/src/remotion/kit/word-anchor.ts +107 -0
  93. package/src/render/contact-sheet.ts +39 -0
  94. package/src/render/continuity.ts +14 -4
  95. package/src/render/deps.ts +15 -3
  96. package/src/render/master.ts +31 -0
  97. package/src/render/render.ts +15 -8
  98. package/src/render/sound-notes.ts +106 -0
  99. package/src/render/static-check.ts +156 -0
  100. package/src/render/validate.ts +3 -150
  101. package/src/render/word-check.ts +181 -0
  102. package/src/testing/conformance.ts +61 -1
  103. package/src/testing/fake-api.ts +11 -5
  104. package/src/testing/fixtures.ts +3 -0
@@ -0,0 +1,68 @@
1
+ import { audioDuration, tool } from "./refmeasure";
2
+
3
+ // How loud a sound is, measured on this machine, and how much to turn it up or down so that the same `volume` number sounds equally loud
4
+ // for every file.
5
+ // A one-shot (a sound effect under 3 s) is levelled by its loudest moment: its peak is brought to -3 dBFS.
6
+ // A longer sound (music, a long effect) is levelled by how loud it is overall: about -18 LUFS, integrated.
7
+ // The gain is limited to 18 dB either way, so a nearly silent or a broken file is not blown up into noise.
8
+ export const ONE_SHOT_SECONDS = 3;
9
+ export const ONE_SHOT_PEAK_DB = -3;
10
+ export const LONG_TARGET_LUFS = -18;
11
+ export const MAX_GAIN_DB = 18;
12
+
13
+ export type SoundLevel = { durationSec: number; peakDb?: number; lufs?: number };
14
+
15
+ const round = (n: number, digits: number) => Math.round(n * 10 ** digits) / 10 ** digits;
16
+
17
+ // The gain in dB that brings a measured sound to its target, or 0 when it could not be measured (silence).
18
+ export function gainDbFor(level: SoundLevel): number {
19
+ const oneShot = level.durationSec < ONE_SHOT_SECONDS;
20
+ const measured = oneShot ? level.peakDb : level.lufs ?? level.peakDb;
21
+ if (measured === undefined || !Number.isFinite(measured)) return 0;
22
+ const target = oneShot || level.lufs === undefined ? ONE_SHOT_PEAK_DB : LONG_TARGET_LUFS;
23
+ return round(Math.max(-MAX_GAIN_DB, Math.min(MAX_GAIN_DB, target - measured)), 1);
24
+ }
25
+
26
+ export const dbToLinear = (db: number) => round(10 ** (db / 20), 4);
27
+
28
+ // ffmpeg's volumedetect: the loudest sample, in dBFS. Undefined for digital silence.
29
+ export async function peakOf(path: string): Promise<number | undefined> {
30
+ const { stderr } = await tool("ffmpeg", ["-hide_banner", "-nostats", "-i", path, "-vn", "-af", "volumedetect", "-f", "null", "-"]);
31
+ const peak = Number(/max_volume:\s*(-?[0-9.]+)\s*dB/.exec(stderr)?.[1]);
32
+ return Number.isFinite(peak) ? peak : undefined;
33
+ }
34
+
35
+ // What loudnorm's first pass reports; its numbers are text in the JSON it prints, and "-inf" for silence.
36
+ export type LoudnormReport = { input_i: string; input_tp: string; input_lra: string; input_thresh: string; target_offset: string };
37
+
38
+ export function parseLoudnorm(stderr: string): LoudnormReport | undefined {
39
+ const end = stderr.lastIndexOf("}");
40
+ const start = stderr.lastIndexOf("{", end);
41
+ if (start < 0 || end < start) return undefined;
42
+ try {
43
+ const r = JSON.parse(stderr.slice(start, end + 1)) as LoudnormReport;
44
+ return typeof r.input_i === "string" ? r : undefined;
45
+ } catch { return undefined; }
46
+ }
47
+
48
+ export async function measureLoudnorm(path: string, target: { i: number; tp: number }): Promise<LoudnormReport | undefined> {
49
+ const { stderr } = await tool("ffmpeg", ["-hide_banner", "-nostats", "-i", path, "-vn", "-af", `loudnorm=I=${target.i}:TP=${target.tp}:LRA=11:print_format=json`, "-f", "null", "-"]);
50
+ return parseLoudnorm(stderr);
51
+ }
52
+
53
+ // Measures a sound for levelling. A failure to measure is not the caller's failure: it returns undefined and the sound is used as it is.
54
+ export async function levelOf(path: string): Promise<(SoundLevel & { gainDb: number }) | undefined> {
55
+ try {
56
+ const durationSec = await audioDuration(path);
57
+ if (!(durationSec > 0)) return undefined;
58
+ const peakDb = await peakOf(path);
59
+ let lufs: number | undefined;
60
+ if (durationSec >= ONE_SHOT_SECONDS) {
61
+ const r = await measureLoudnorm(path, { i: LONG_TARGET_LUFS, tp: -1.5 });
62
+ const v = Number(r?.input_i);
63
+ if (Number.isFinite(v)) lufs = round(v, 1);
64
+ }
65
+ const level: SoundLevel = { durationSec, ...(peakDb !== undefined ? { peakDb } : {}), ...(lufs !== undefined ? { lufs } : {}) };
66
+ return { ...level, gainDb: gainDbFor(level) };
67
+ } catch { return undefined; }
68
+ }
@@ -1,6 +1,9 @@
1
- import { PACE_SPEED, type AssetManifest, type ScenePlan, type WordTiming } from "../pipeline/schema";
2
- import { extendBeats, snapToBeats, toBeatFrames } from "../pipeline/beatsnap";
3
- import { dimensionsFor, FPS, layoutScenes } from "../pipeline/timing";
1
+ import { gapSec, isVoiceless, PACE_SPEED, type AssetManifest, type ScenePlan, type WordTiming } from "../pipeline/schema";
2
+ import { extendBeats, MIN_TAIL_SEC, snapNarrated, snapToBeats, toBeatFrames } from "../pipeline/beatsnap";
3
+ import { dimensionsFor, FPS, lastWordEndSec, layoutNarrated, layoutScenes, secondsToFrames } from "../pipeline/timing";
4
+ import { readBackground } from "./background";
5
+ import { toImageLayers, type LayerEntry } from "./layers";
6
+ import { dbToLinear } from "./loudness";
4
7
  import type { MusicRecord } from "./music";
5
8
  import { FILES, type Project } from "./project";
6
9
 
@@ -16,26 +19,48 @@ export function voiceoverStale(vo: Voiceover, scene: ScenePlan["scenes"][number]
16
19
  return vo.text !== scene.narration || vo.voiceId !== plan.voiceId || vo.speed !== PACE_SPEED[plan.pace ?? "normal"];
17
20
  }
18
21
 
22
+ // What the length of a film with no voice comes to once its cuts are on the beat of the pulled track, in seconds. Undefined without a track
23
+ // whose tempo can be trusted. Used by `plan check`, which has no manifest yet.
24
+ export function snappedSeconds(project: Project, plan: ScenePlan): number | undefined {
25
+ if (!isVoiceless(plan)) return undefined;
26
+ const music = project.readJsonOr<MusicRecord | undefined>(FILES.music, undefined);
27
+ if (!music?.bpm || !music.beats.length) return undefined;
28
+ const layout = layoutScenes(plan.scenes.map((s) => s.seconds ?? 0), FPS, 0);
29
+ const beatFrames = toBeatFrames(extendBeats(music.beats, music.bpm, layout.totalFrames / FPS + 10), FPS);
30
+ const lengths = snapToBeats({ naturalFrames: layout.scenes.map((s) => s.durationFrames), lastWordEndFrame: 0, beatFrames, fps: FPS, grid: "nearest" });
31
+ return lengths.reduce((a, b) => a + b, 0) / FPS;
32
+ }
33
+
19
34
  // Lays the scenes out on the timeline from the real voiceover lengths. Paths are relative to the project folder.
20
35
  // Returns undefined, and writes nothing, until every scene has its voiceover.
21
36
  export function buildManifest(project: Project, plan: ScenePlan): AssetManifest | undefined {
22
- const voiceovers = project.readJsonOr<Record<string, Voiceover>>(FILES.voiceovers, {});
23
- if (plan.scenes.some((s) => !voiceovers[s.id])) return undefined;
37
+ const voiceless = isVoiceless(plan);
38
+ const voiceovers = voiceless ? {} : project.readJsonOr<Record<string, Voiceover>>(FILES.voiceovers, {});
39
+ if (!voiceless && plan.scenes.some((s) => !voiceovers[s.id])) return undefined;
24
40
  const images = project.readJsonOr<Record<string, string>>(FILES.images, {});
25
41
  const clips = project.readJsonOr<Record<string, ClipRecord>>(FILES.clips, {});
26
42
  const footage = project.footage();
27
43
  const byId = new Map(project.assets().map((a) => [a.id, a]));
28
44
 
29
- let layout = layoutScenes(plan.scenes.map((s) => voiceovers[s.id].durationSec));
45
+ // With no voice the plan's own seconds are the lengths, with no padding after them: the plan already says how long each scene holds.
46
+ // With a voice each scene lasts until its last word has ended plus the plan's gap (the silence before the next sentence); the last keeps a longer tail.
47
+ let layout = voiceless ? layoutScenes(plan.scenes.map((s) => s.seconds ?? 0), FPS, 0) : layoutNarrated(plan.scenes.map((s) => voiceovers[s.id]!), gapSec(plan), FPS);
30
48
  const music = project.readJsonOr<MusicRecord | undefined>(FILES.music, undefined);
31
49
  let beatFrames: number[] = [];
32
50
  if (music) {
33
51
  beatFrames = toBeatFrames(extendBeats(music.beats, music.bpm ?? 120, layout.totalFrames / FPS + 10), FPS);
34
52
  // Scene changes land on beats only when the track has a trusted tempo. Footage has a fixed length that a longer scene must not overrun.
35
53
  if (music.bpm && music.beats.length) {
36
- const last = voiceovers[plan.scenes[plan.scenes.length - 1].id];
37
- const lastWords = last.words.length ? Math.max(...last.words.map((w) => w.endSec)) : last.durationSec;
38
- const lengths = snapToBeats({ naturalFrames: layout.scenes.map((s) => s.durationFrames), lastWordEndFrame: Math.ceil(lastWords * FPS), beatFrames, fps: FPS });
54
+ const naturalFrames = layout.scenes.map((s) => s.durationFrames);
55
+ let lengths: number[];
56
+ if (voiceless) {
57
+ // No word to leave a tail after: the last scene is as long as the plan says, so the tail rule must not stretch it.
58
+ const lastWordEndFrame = Math.max(0, naturalFrames[naturalFrames.length - 1]! - Math.ceil(MIN_TAIL_SEC * FPS));
59
+ lengths = snapToBeats({ naturalFrames, lastWordEndFrame, beatFrames, fps: FPS, grid: "nearest" });
60
+ } else {
61
+ // The voice is the clock: a change moves to a beat only when that is a few frames away and cuts into no word.
62
+ lengths = snapNarrated({ naturalFrames, lastWordEndFrames: plan.scenes.map((s) => Math.ceil(lastWordEndSec(voiceovers[s.id]!) * FPS - 1e-9)), beatFrames }).frames;
63
+ }
39
64
  let cursor = 0;
40
65
  const snapped = { scenes: lengths.map((durationFrames) => { const r = { startFrame: cursor, durationFrames }; cursor += durationFrames; return r; }), totalFrames: 0 };
41
66
  snapped.totalFrames = cursor;
@@ -46,26 +71,41 @@ export function buildManifest(project: Project, plan: ScenePlan): AssetManifest
46
71
  if (footage) {
47
72
  const footageFrames = Math.floor((footage.durationSec ?? 0) * FPS);
48
73
  if (layout.totalFrames > footageFrames) {
49
- throw new Error(`The voiceover (${(layout.totalFrames / FPS).toFixed(1)}s) is longer than the footage (${(footageFrames / FPS).toFixed(1)}s). Use a longer clip or shorten the script.`);
74
+ throw new Error(voiceless
75
+ ? `The scenes (${(layout.totalFrames / FPS).toFixed(1)}s) are longer than the footage (${(footageFrames / FPS).toFixed(1)}s). Use a longer clip or shorten the scenes' seconds.`
76
+ : `The voiceover (${(layout.totalFrames / FPS).toFixed(1)}s) is longer than the footage (${(footageFrames / FPS).toFixed(1)}s). Use a longer clip or shorten the script.`);
50
77
  }
51
78
  totalFrames = footageFrames;
52
79
  }
53
80
 
81
+ // The levelling of every sound that was measured when it was pulled, as linear gains by path.
82
+ const soundGain: Record<string, number> = {};
83
+ for (const e of Object.values(project.readJsonOr<Record<string, { path: string; kind: string; gainDb?: number }>>(FILES.library, {}))) {
84
+ if ((e.kind === "sfx" || e.kind === "music") && typeof e.gainDb === "number") soundGain[e.path] = dbToLinear(e.gainDb);
85
+ }
86
+ if (music) soundGain[music.key] = dbToLinear(music.gainDb);
87
+
88
+ const background = readBackground(project);
89
+ const layerEntries = project.readJsonOr<Record<string, LayerEntry>>(FILES.layers, {});
54
90
  const manifest: AssetManifest = {
55
91
  fps: FPS,
56
92
  ...dimensionsFor(plan.aspect, footage),
57
93
  totalFrames,
58
- captions: plan.captions ?? "phrase",
94
+ captions: plan.captions ?? (voiceless ? "none" : "phrase"),
95
+ ...(Object.keys(soundGain).length ? { soundGain } : {}),
59
96
  ...(footage ? { footageKey: footage.key } : {}),
97
+ ...(background.film ? { background: { key: background.film.key, ...(background.film.durationSec ? { durationSec: background.film.durationSec } : {}) } } : {}),
60
98
  ...(music ? { music: { key: music.key, ...(music.bpm ? { bpm: music.bpm } : {}), beatFrames: music.bpm ? beatFrames.filter((f) => f < totalFrames) : [] } } : {}),
61
99
  scenes: plan.scenes.map((scene, i) => ({
62
100
  id: scene.id,
63
101
  startFrame: layout.scenes[i].startFrame,
64
102
  durationFrames: layout.scenes[i].durationFrames,
65
- voiceoverKey: voiceovers[scene.id].key,
66
- words: voiceovers[scene.id].words,
103
+ ...(voiceless ? {} : { voiceoverKey: voiceovers[scene.id]!.key }),
104
+ words: voiceless ? [] : voiceovers[scene.id]!.words,
67
105
  ...(scene.treatment === "illustration" && images[scene.id] ? { imageKey: images[scene.id] } : {}),
106
+ ...(scene.treatment === "illustration" && images[scene.id] && layerEntries[images[scene.id]!] ? { imageLayers: toImageLayers(images[scene.id]!, layerEntries[images[scene.id]!]!) } : {}),
68
107
  ...(scene.treatment === "clip" && clips[scene.id] ? { clipKey: clips[scene.id].key, ...(clips[scene.id].keyedKey ? { clipKeyedKey: clips[scene.id].keyedKey } : {}) } : {}),
108
+ ...(background.scenes[scene.id] ? { background: { key: background.scenes[scene.id]!.key } } : {}),
69
109
  userAssetKeys: scene.userAssetIds.map((id) => {
70
110
  const a = byId.get(id);
71
111
  if (!a) throw new Error(`Scene ${scene.id} references unknown asset ${id}. Add it with \`reelkit assets upload\`.`);
@@ -83,10 +123,12 @@ export function missingAssets(project: Project, plan: ScenePlan): string[] {
83
123
  const images = project.readJsonOr<Record<string, string>>(FILES.images, {});
84
124
  const clips = project.readJsonOr<Record<string, ClipRecord>>(FILES.clips, {});
85
125
  const problems: string[] = [];
126
+ const voiceless = isVoiceless(plan);
86
127
  const music = project.readJsonOr<MusicRecord | undefined>(FILES.music, undefined);
87
128
  if (music && !project.exists(music.key)) problems.push(`Missing file: ${music.key}. Pull the track again with \`reelkit assets pull ${music.id} --music\`.`);
88
129
  for (const s of plan.scenes) {
89
- if (!voiceovers[s.id]) problems.push(`Scene ${s.id} has no voiceover.`);
130
+ if (voiceless) { /* nothing is recorded for a video with no voice */ }
131
+ else if (!voiceovers[s.id]) problems.push(`Scene ${s.id} has no voiceover.`);
90
132
  else if (voiceoverStale(voiceovers[s.id], s, plan)) problems.push(`Scene ${s.id}'s voiceover is out of date. Run \`reelkit assets voiceover --all\`.`);
91
133
  else if (!project.exists(voiceovers[s.id].key)) problems.push(`Missing file: ${voiceovers[s.id].key}`);
92
134
  // Only an illustration scene uses an image; one left over from an earlier plan is not needed.
@@ -102,3 +144,15 @@ export function missingAssets(project: Project, plan: ScenePlan): string[] {
102
144
  }
103
145
  return problems;
104
146
  }
147
+
148
+ // The silence between each scene's last word and the next scene's first word, in seconds, as the manifest lays them out. This is what a listener hears
149
+ // between two sentences, and it is what the plan's `gap` sets (the beat can move a change by a few frames).
150
+ export function sentenceGaps(manifest: AssetManifest): number[] {
151
+ return manifest.scenes.slice(0, -1).map((s, i) => {
152
+ const next = manifest.scenes[i + 1]!;
153
+ if (!s.words.length || !next.words.length) return Math.max(0, (next.startFrame - s.startFrame) / manifest.fps);
154
+ const end = s.startFrame / manifest.fps + Math.max(...s.words.map((w) => w.endSec));
155
+ const start = next.startFrame / manifest.fps + Math.min(...next.words.map((w) => w.startSec));
156
+ return Math.round((start - end) * 100) / 100;
157
+ });
158
+ }
@@ -1,4 +1,5 @@
1
1
  import type { AssetManifest } from "../pipeline/schema";
2
+ import { gridPoints } from "../pipeline/beatsnap";
2
3
  import { analyzeBeats } from "./beats";
3
4
  import { audioDuration, decodeMono } from "./refmeasure";
4
5
 
@@ -14,12 +15,25 @@ export async function measureMusic(path: string): Promise<Pick<MusicRecord, "dur
14
15
  return { durationSec, beatConfidence: r.confidence, beats: r.tempoBpm !== undefined && r.beats ? r.beats : [], ...(r.tempoBpm !== undefined ? { bpm: r.tempoBpm } : {}) };
15
16
  }
16
17
 
17
- // How many scene changes fall on a beat, for the preview.
18
- export function beatReport(manifest: AssetManifest): { bpm: number; boundariesOnBeat: number; boundaries: number; line: string } | undefined {
18
+ // How many scene changes fall on a beat, for the preview. A film with no voice puts its cuts on a finer grid, so there a change on a half or
19
+ // a quarter beat counts too, and the line says which grid was needed.
20
+ export function beatReport(manifest: AssetManifest, opts: { finer?: boolean } = {}): { bpm: number; boundariesOnBeat: number; boundaries: number; line: string } | undefined {
19
21
  const m = manifest.music;
20
22
  if (!m?.bpm || manifest.scenes.length < 2) return undefined;
21
- const beats = new Set(m.beatFrames);
22
23
  const starts = manifest.scenes.slice(1).map((s) => s.startFrame);
23
- const on = starts.filter((f) => beats.has(f)).length;
24
- return { bpm: m.bpm, boundariesOnBeat: on, boundaries: starts.length, line: `${on} of ${starts.length} scene changes land on the beat at ${Math.round(m.bpm)} BPM` };
24
+ const beats = new Set(m.beatFrames);
25
+ const bpm = Math.round(m.bpm);
26
+ if (!opts.finer) {
27
+ const on = starts.filter((f) => beats.has(f)).length;
28
+ // With a narrator the voice sets where a scene ends: a change is moved to a beat only when that is a few frames away, so say honestly how many are.
29
+ const rest = starts.length - on;
30
+ return { bpm: m.bpm, boundariesOnBeat: on, boundaries: starts.length, line: `${on} of ${starts.length} scene changes land on the beat at ${bpm} BPM${rest ? `; the other${rest === 1 ? "" : "s"} follow${rest === 1 ? "s" : ""} the voice` : ""}` };
31
+ }
32
+ const halves = new Set(gridPoints(m.beatFrames, 2)), quarters = new Set(gridPoints(m.beatFrames, 4));
33
+ const onBeat = starts.filter((f) => beats.has(f)).length;
34
+ const onHalf = starts.filter((f) => !beats.has(f) && halves.has(f)).length;
35
+ const onQuarter = starts.filter((f) => !beats.has(f) && !halves.has(f) && quarters.has(f)).length;
36
+ const on = onBeat + onHalf + onQuarter;
37
+ const grid = onQuarter ? "the beat, a half beat or a quarter beat" : onHalf ? "the beat or a half beat" : "the beat";
38
+ return { bpm: m.bpm, boundariesOnBeat: on, boundaries: starts.length, line: `${on} of ${starts.length} scene changes land on ${grid} at ${bpm} BPM` };
25
39
  }
@@ -5,16 +5,21 @@ import { AspectSchema, type AssetRecord } from "../pipeline/schema";
5
5
 
6
6
  export const FILES = {
7
7
  config: "reelkit.json", plan: "plan.json", manifest: "manifest.json",
8
- assetIndex: "assets/index.json", voiceovers: "assets/voiceovers.json", images: "assets/images.json", clips: "assets/clips.json", library: "assets/library.json", music: "assets/music.json",
8
+ assetIndex: "assets/index.json", voiceovers: "assets/voiceovers.json", images: "assets/images.json", clips: "assets/clips.json", library: "assets/library.json", music: "assets/music.json", background: "assets/background.json", layers: "assets/layers.json", searches: "assets/searches.json", shared: "assets/shared.json",
9
9
  } as const;
10
10
 
11
11
  // "path: message", or just the message for a problem at the root of the file.
12
12
  export const issueLines = (e: { issues: { path: PropertyKey[]; message: string }[] }) =>
13
13
  e.issues.map((i) => (i.path.length ? `${i.path.map(String).join(".")}: ${i.message}` : i.message));
14
14
 
15
- export const ProjectConfigSchema = z.object({ aspect: AspectSchema, name: z.string().optional(), footage: z.string().optional() });
15
+ export const ProjectConfigSchema = z.object({ aspect: AspectSchema, name: z.string().optional(), footage: z.string().optional(),
16
+ // false when the project was made with --private: nothing is shared with the library automatically.
17
+ shareComponents: z.boolean().optional() });
16
18
  export type ProjectConfig = z.infer<typeof ProjectConfigSchema>;
17
19
 
20
+ // True for a project made with `reelkit init --private` (shareComponents: false): nothing in it is sent to the library, whatever the plan or a flag says.
21
+ export const isPrivateProject = (project: { config(): ProjectConfig }): boolean => project.config().shareComponents === false;
22
+
18
23
  // One video's working folder. Every command reads and writes its state here.
19
24
  export class Project {
20
25
  constructor(readonly dir: string) {}
@@ -9,7 +9,7 @@ const exec = promisify(execFile);
9
9
  export const REF_ASPECTS = ["9:16", "16:9", "1:1", "4:5"] as const;
10
10
  export type RefAspect = (typeof REF_ASPECTS)[number];
11
11
 
12
- async function tool(cmd: "ffmpeg" | "ffprobe", args: string[]): Promise<{ stdout: Buffer; stderr: string }> {
12
+ export async function tool(cmd: "ffmpeg" | "ffprobe", args: string[]): Promise<{ stdout: Buffer; stderr: string }> {
13
13
  try {
14
14
  const r = await exec(cmd, args, { maxBuffer: 256 * 1024 * 1024, encoding: "buffer" });
15
15
  return { stdout: r.stdout, stderr: r.stderr.toString("utf8") };
@@ -0,0 +1,347 @@
1
+ // Measures how a finished film sounds against what it shows: how often something hits, how often a swell rises, and how many of the big
2
+ // changes in the picture have a hit beside them. Everything but the two decoders is a pure function over arrays.
3
+ import { loudnessLufs, meanAbsDiffs, probeVideo, tool } from "./refmeasure";
4
+
5
+ export const AUDIO_RATE = 22050;
6
+ export const PICTURE_FPS = 30;
7
+ export const PICTURE_WIDTH = 64;
8
+ const WINDOW_SEC = 0.046, HOP_SEC = 0.012;
9
+ // A hit this close to a picture change belongs to it.
10
+ export const HIT_NEAR_SEC = 0.1;
11
+ // Nothing quieter than about -50 dBFS counts as sound, so a track that is only dither does not start the film.
12
+ const AUDIBLE_RMS = 0.003;
13
+ // A swell lives in the 0.5 to 10 kHz band, the air of a whoosh or a riser. Measured on the library's own sounds, whooshes keep their energy between
14
+ // 0.2 and 5 kHz (the Cinematic and Quantum Motion ones put 40 to 90 percent of it between 0.5 and 5 kHz, risers 40 to 95 percent between 2 and 10 kHz)
15
+ // and only a riser has a real share above 5 kHz, so a 5 to 10 kHz band could not hear a whoosh at all. Bass and kicks stay below it.
16
+ export const SWELL_BAND: [number, number] = [500, 10000];
17
+ // The band must be at least this loud (about -50 dBFS) to count as a swell, so that silence and dither are not counted.
18
+ const BAND_FLOOR = 0.003;
19
+ // A swell is a rise of the band above its own local level: the level is the median of the band over about 1.5 s around it, the rise must reach
20
+ // SWELL_RISE_DB above it, stay above SWELL_EDGE_DB for at least SWELL_MIN_SEC, and take at least SWELL_RAMP_SEC to build (a click or a hit does not).
21
+ const FLOOR_SEC = 1.5, SWELL_EDGE_DB = 2, SWELL_RISE_DB = 4, SWELL_MIN_SEC = 0.12, SWELL_RAMP_SEC = 0.08;
22
+
23
+ const mean = (a: ArrayLike<number>) => { let s = 0; for (let i = 0; i < a.length; i++) s += a[i]!; return a.length ? s / a.length : 0; };
24
+ const sd = (a: ArrayLike<number>, m = mean(a)) => { let s = 0; for (let i = 0; i < a.length; i++) s += (a[i]! - m) ** 2; return a.length ? Math.sqrt(s / a.length) : 0; };
25
+
26
+ // The root-mean-square level of each ~46 ms window, every ~12 ms.
27
+ export function envelope(samples: ArrayLike<number>, rate: number): { values: Float64Array; hopSec: number } {
28
+ const win = Math.max(1, Math.round(WINDOW_SEC * rate)), hop = Math.max(1, Math.round(HOP_SEC * rate));
29
+ const n = samples.length >= win ? Math.floor((samples.length - win) / hop) + 1 : 0;
30
+ const values = new Float64Array(n);
31
+ for (let f = 0; f < n; f++) {
32
+ let s = 0;
33
+ for (let i = f * hop, end = i + win; i < end; i++) s += samples[i]! * samples[i]!;
34
+ values[f] = Math.sqrt(s / win);
35
+ }
36
+ return { values, hopSec: hop / rate };
37
+ }
38
+
39
+ // Peaks of a series that rise above `threshold`, at least `gap` entries apart (the higher of two close ones is kept).
40
+ function pickPeaks(series: ArrayLike<number>, threshold: number, gap: number): number[] {
41
+ const peaks: number[] = [];
42
+ for (let i = 0; i < series.length; i++) {
43
+ const v = series[i]!;
44
+ if (!(v > threshold) || v < (series[i - 1] ?? -Infinity) || v < (series[i + 1] ?? -Infinity)) continue;
45
+ const last = peaks[peaks.length - 1];
46
+ if (last !== undefined && i - last < gap) { if (v > series[last]!) peaks[peaks.length - 1] = i; } else peaks.push(i);
47
+ }
48
+ return peaks;
49
+ }
50
+
51
+ // The times in seconds of the hits: sudden rises of the full-band level. A hit is where the level of an ~46 ms window jumps by at least HIT_JUMP_DB
52
+ // over what it was 48 ms before and ends up audible, at least 90 ms apart. The jump is relative to the sound just before it, so a quiet tap is a hit
53
+ // whatever else the film holds: a loud drop does not raise the bar for the rest (a threshold taken from the whole film's statistics did).
54
+ const HIT_JUMP_DB = 8, HIT_LOOKBACK = 4, HIT_FLOOR = 0.001;
55
+ export function hitTimes(samples: ArrayLike<number>, rate: number): number[] {
56
+ const { values, hopSec } = envelope(samples, rate);
57
+ const jump = new Float64Array(values.length);
58
+ for (let i = HIT_LOOKBACK; i < values.length; i++) {
59
+ if (values[i]! < HIT_FLOOR) continue;
60
+ jump[i] = 20 * Math.log10(values[i]! / Math.max(values[i - HIT_LOOKBACK]!, 1e-5));
61
+ }
62
+ return pickPeaks(jump, HIT_JUMP_DB, Math.max(1, Math.round(0.09 / hopSec))).map((i) => (i * hopSec) + WINDOW_SEC / 2);
63
+ }
64
+
65
+ // One second-order section (RBJ cookbook). Two of them in a row make each edge of the band steeper.
66
+ function biquad(x: ArrayLike<number>, type: "high" | "low", freq: number, rate: number): Float64Array {
67
+ const w0 = (2 * Math.PI * freq) / rate, cos = Math.cos(w0), alpha = Math.sin(w0) / (2 * Math.SQRT1_2);
68
+ const [b0, b1, b2] = type === "high" ? [(1 + cos) / 2, -(1 + cos), (1 + cos) / 2] : [(1 - cos) / 2, 1 - cos, (1 - cos) / 2];
69
+ const a0 = 1 + alpha, a1 = -2 * cos, a2 = 1 - alpha;
70
+ const y = new Float64Array(x.length);
71
+ let x1 = 0, x2 = 0, y1 = 0, y2 = 0;
72
+ for (let i = 0; i < x.length; i++) {
73
+ const v = (b0 * x[i]! + b1 * x1 + b2 * x2 - a1 * y1 - a2 * y2) / a0;
74
+ x2 = x1; x1 = x[i]!; y2 = y1; y1 = v; y[i] = v;
75
+ }
76
+ return y;
77
+ }
78
+
79
+ export type Swell = { start: number; peak: number; length: number };
80
+
81
+ // The swells of a sound: where the 0.5 to 10 kHz band rises ramping up above its own local level (see the constants above). Each is a start (where it
82
+ // leaves the local level), a peak (its loudest moment) and a length, all in seconds. The level is local, so one loud riser does not hide the quiet
83
+ // whooshes around it, and a steady bed of music, which is its own level everywhere, gives none.
84
+ export function swells(samples: ArrayLike<number>, rate: number): Swell[] {
85
+ if (samples.length < rate / 10) return [];
86
+ const [lo, hi] = [SWELL_BAND[0], Math.min(SWELL_BAND[1], rate * 0.45)];
87
+ const band = biquad(biquad(biquad(biquad(samples, "high", lo, rate), "high", lo, rate), "low", hi, rate), "low", hi, rate);
88
+ const fine = envelope(band, rate);
89
+ // One value every four hops (about 48 ms) is enough resolution and keeps the median cheap.
90
+ const step = 4, hopSec = fine.hopSec * step, n = Math.floor(fine.values.length / step);
91
+ const level = new Float64Array(n), db = new Float64Array(n);
92
+ for (let i = 0; i < n; i++) { level[i] = fine.values[i * step]!; db[i] = 20 * Math.log10(Math.max(level[i]!, 1e-6)); }
93
+ const half = Math.max(1, Math.round(FLOOR_SEC / 2 / hopSec));
94
+ const excess = new Float64Array(n);
95
+ for (let i = 0; i < n; i++) {
96
+ const win = Array.from(db.subarray(Math.max(0, i - half), Math.min(n, i + half + 1))).sort((x, y) => x - y);
97
+ excess[i] = db[i]! - win[win.length >> 1]!;
98
+ }
99
+ const out: Swell[] = [];
100
+ for (let i = 0; i < n;) {
101
+ if (!(excess[i]! > SWELL_EDGE_DB)) { i++; continue; }
102
+ let j = i;
103
+ while (j < n && excess[j]! > SWELL_EDGE_DB) j++;
104
+ let top = i, loudest = i;
105
+ for (let k = i; k < j; k++) { if (excess[k]! > excess[top]!) top = k; if (level[k]! > level[loudest]!) loudest = k; }
106
+ // How long the rise took: from leaving the local level to coming within 3 dB of the run's own top.
107
+ let near = i;
108
+ while (near < top && excess[near]! < excess[top]! - 3) near++;
109
+ const loud = level[loudest]! >= BAND_FLOOR;
110
+ if (loud && excess[top]! >= SWELL_RISE_DB && (j - i) * hopSec >= SWELL_MIN_SEC && (near - i + 1) * hopSec >= SWELL_RAMP_SEC) {
111
+ out.push({ start: i * hopSec, peak: loudest * hopSec, length: (j - i) * hopSec });
112
+ }
113
+ i = j;
114
+ }
115
+ return out;
116
+ }
117
+
118
+ // The start times in seconds of the swells.
119
+ export const swellTimes = (samples: ArrayLike<number>, rate: number): number[] => swells(samples, rate).map((w) => w.start);
120
+
121
+ // A big change of the picture. A transition is a span of frames with a high difference, not a single frame: `start` and `end` (seconds) bound the
122
+ // run of frames above the change threshold around the peak, `time` and `frame` are the peak itself, and `strength` is the peak's mean difference
123
+ // between neighbouring frames (0 to 255, on a small grey picture).
124
+ export type PictureChange = { time: number; frame: number; strength: number; start: number; end: number };
125
+
126
+ // The big picture changes: peaks of the mean absolute difference between neighbouring frames above its mean plus 1.5 standard deviations, at least
127
+ // 0.2 s apart, each with the contiguous run of frames above that same threshold around it. `diffs[i]` is the change from frame i to frame i + 1.
128
+ export function pictureChanges(diffs: ArrayLike<number>, fps: number): PictureChange[] {
129
+ const m = mean(diffs);
130
+ const threshold = Math.max(0.5, m + 1.5 * sd(diffs, m));
131
+ return pickPeaks(diffs, threshold, Math.max(1, Math.round(0.2 * fps))).map((i) => {
132
+ let a = i, b = i;
133
+ while (a > 0 && diffs[a - 1]! > threshold) a--;
134
+ while (b < diffs.length - 1 && diffs[b + 1]! > threshold) b++;
135
+ return { time: (i + 1) / fps, frame: i + 1, strength: Math.round(diffs[i]! * 10) / 10, start: (a + 1) / fps, end: (b + 1) / fps };
136
+ });
137
+ }
138
+
139
+ export const pictureChangeTimes = (diffs: ArrayLike<number>, fps: number): number[] => pictureChanges(diffs, fps).map((c) => c.time);
140
+
141
+ // The hit nearest to a change, measured to the nearest point of the change's own span: 0 when the hit lies inside it. `side` says which way it is.
142
+ export type HitMatch = { hit: number; distance: number; side: "inside" | "before" | "after" };
143
+ export function nearestHit(change: PictureChange | number, hits: readonly number[]): HitMatch | undefined {
144
+ const [a, b] = typeof change === "number" ? [change, change] : [change.start, change.end];
145
+ let best: HitMatch | undefined;
146
+ for (const h of hits) {
147
+ const d = h < a ? a - h : h > b ? h - b : 0;
148
+ if (!best || d < best.distance) best = { hit: h, distance: d, side: h < a ? "before" : h > b ? "after" : "inside" };
149
+ }
150
+ return best;
151
+ }
152
+
153
+ // How many changes have a hit within `near` seconds of their span.
154
+ export const countWithHit = (changes: readonly (PictureChange | number)[], hits: readonly number[], near = HIT_NEAR_SEC): number =>
155
+ changes.filter((c) => (nearestHit(c, hits)?.distance ?? Infinity) <= near + 1e-9).length;
156
+
157
+ // The first moment the full-band level reaches audible, or null for a film that is silent.
158
+ export function firstSoundSec(samples: ArrayLike<number>, rate: number): number | null {
159
+ const { values, hopSec } = envelope(samples, rate);
160
+ const i = values.findIndex((v) => v > AUDIBLE_RMS);
161
+ return i < 0 ? null : Math.round(i * hopSec * 100) / 100;
162
+ }
163
+
164
+ export type SoundDetail = {
165
+ changes: (PictureChange & { hit: number | null; distance: number | null; side: HitMatch["side"] | null; matched: boolean })[];
166
+ swells: Swell[];
167
+ hits: number[];
168
+ firstSoundSec: number | null;
169
+ };
170
+ export type SoundReport = { hitsPerSecond: number; swellsPerSecond: number; pictureChanges: number; changesWithHit: number; firstSoundSec: number | null; loudnessLufs?: number; detail?: SoundDetail; voice?: VoiceMusic };
171
+ const round = (n: number, d = 2) => Math.round(n * 10 ** d) / 10 ** d;
172
+
173
+ // The whole measurement from decoded audio and the picture's frame differences. `detail` lists every event the counts were made from.
174
+ export function soundReport(samples: ArrayLike<number>, rate: number, pictureDiffs: ArrayLike<number>, fps: number): SoundReport {
175
+ const seconds = samples.length / rate;
176
+ const hits = hitTimes(samples, rate), found = swells(samples, rate), changes = pictureChanges(pictureDiffs, fps), first = firstSoundSec(samples, rate);
177
+ return {
178
+ hitsPerSecond: seconds > 0 ? round(hits.length / seconds) : 0, swellsPerSecond: seconds > 0 ? round(found.length / seconds, 3) : 0,
179
+ pictureChanges: changes.length, changesWithHit: countWithHit(changes, hits), firstSoundSec: first,
180
+ detail: {
181
+ changes: changes.map((c) => { const m = nearestHit(c, hits); return { ...c, hit: m ? round(m.hit, 3) : null, distance: m ? round(m.distance, 3) : null, side: m?.side ?? null, matched: (m?.distance ?? Infinity) <= HIT_NEAR_SEC + 1e-9 }; }),
182
+ swells: found.map((w) => ({ start: round(w.start, 2), peak: round(w.peak, 2), length: round(w.length, 2) })),
183
+ hits: hits.map((h) => round(h, 3)),
184
+ firstSoundSec: first,
185
+ },
186
+ };
187
+ }
188
+
189
+ // The same report as lines of text, one per event, for `reelkit sound --detail`.
190
+ export function detailLines(d: SoundDetail): string[] {
191
+ const t = (n: number) => `${n.toFixed(2)}s`;
192
+ const lines = ["Picture changes (the time and frame of the peak, how strong, and the nearest hit measured to the changing span):"];
193
+ for (const c of d.changes) {
194
+ const span = c.end > c.start ? ` (changing ${t(c.start)} to ${t(c.end)})` : "";
195
+ const hit = c.hit === null ? "no hit at all" : c.side === "inside" ? `hit at ${t(c.hit)}, inside the change` : `nearest hit at ${t(c.hit)}, ${t(c.distance!)} ${c.side} it`;
196
+ lines.push(` ${t(c.time)} frame ${c.frame} strength ${c.strength}${span}: ${hit}${c.matched ? "" : " <- no sound within 0.1s"}`);
197
+ }
198
+ if (!d.changes.length) lines.push(" none found");
199
+ lines.push("Swells (start, peak, length):");
200
+ for (const w of d.swells) lines.push(` ${t(w.start)} peak ${t(w.peak)} length ${t(w.length)}`);
201
+ if (!d.swells.length) lines.push(" none found");
202
+ lines.push(`Hits: ${d.hits.length}${d.hits.length ? ` (${d.hits.slice(0, 40).map((h) => h.toFixed(2)).join(", ")}${d.hits.length > 40 ? ", ..." : ""})` : ""}`);
203
+ lines.push(`First sound: ${d.firstSoundSec === null ? "never" : t(d.firstSoundSec)}`);
204
+ return lines;
205
+ }
206
+
207
+ // Plain-word advice for a film with no voice, against the targets in reference/launch-film.md. It is a measurement and never a verdict.
208
+ export function soundAdvice(r: SoundReport): string {
209
+ const hits = `${r.hitsPerSecond.toFixed(1)} hits a second`;
210
+ const every = r.swellsPerSecond > 0 ? 1 / r.swellsPerSecond : undefined;
211
+ const swell = every === undefined ? "no swell at all" : `a swell every ${every < 10 ? every.toFixed(1) : Math.round(every)} seconds`;
212
+ const swellFix = r.swellsPerSecond < 0.5 ? " (films like this have one every 1 to 2 seconds: add a whoosh to each move)" : r.swellsPerSecond > 1.3 ? " (films like this have one every 1 to 2 seconds: that is busy, so leave some moves silent)" : "";
213
+ const aim = Math.max(1, Math.ceil(r.pictureChanges * 0.6));
214
+ const changes = r.pictureChanges ? `${r.changesWithHit} of ${r.pictureChanges} picture changes have a sound within a tenth of a second (aim for ${aim} or more)` : "no clear picture change was found to compare with";
215
+ const start = r.firstSoundSec === null ? " The film is silent." : r.firstSoundSec > 0.1 ? ` The first sound is at ${r.firstSoundSec.toFixed(2)}s; the films started within 0.05s.` : "";
216
+ return `Sound: ${hits}; ${swell}${swellFix}; ${changes}.${start} This is a measurement of the file, not a judgement of how it sounds: listen to it too.`;
217
+ }
218
+
219
+ // ---- The voice against the music, for a narrated film ----------------------------------------------------------------------------------------
220
+ // The mix is one channel, so the music cannot be measured alone. What can be measured is the mix while the voice speaks, the mix in the pauses
221
+ // (where only the music and effects are heard), and the mix in the short pauses between words, where the music is still held down: that last level
222
+ // is the estimate of "the music under the voice". The numbers say "estimated" for that reason.
223
+
224
+ // A word's time in the film, in seconds. `sentenceEnd` is true for the last word of a sentence (it ends with . ? ! or ends its scene).
225
+ export type SpokenWord = { start: number; end: number; sentenceEnd?: boolean };
226
+ export type VoiceMusic = {
227
+ // The mix's level while the voice speaks, dBFS (RMS), and in pauses of at least 0.5 s where it does not.
228
+ speechDb: number;
229
+ pauseDb?: number;
230
+ // speechDb minus pauseDb: how far the voice stands above what is heard in a real pause.
231
+ voiceOverPauseDb?: number;
232
+ // The music's level under the voice, from the short pauses between words, and how far below the voice it is (estimated).
233
+ underVoiceDb?: number;
234
+ musicBelowVoiceDb?: number;
235
+ // How many times the level between two sentences rose more than 6 dB above the level under the voice, and how many such gaps were looked at.
236
+ pumps?: number;
237
+ pumpsPer30s?: number;
238
+ sentenceGaps?: number;
239
+ estimated: true;
240
+ };
241
+
242
+ export const PUMP_RISE_DB = 6;
243
+ // The music under the voice should sit this far below it (dB): closer masks the voice, farther is not heard.
244
+ export const MUSIC_BELOW_VOICE_DB: [number, number] = [10, 22];
245
+ export const MAX_PUMPS_PER_30S = 2;
246
+ const SPEECH_EDGE = 0.02, PAUSE_MIN = 0.5, PAUSE_EDGE = 0.1, SHORT_PAUSE_MIN = 0.12, SHORT_PAUSE_START = 0.05, SHORT_PAUSE_END = 0.03;
247
+ const SENTENCE_GAP_MIN = 0.2, SENTENCE_GAP_START = 0.1, SENTENCE_GAP_END = 0.05, PUMP_WINDOW = 0.1, PUMP_STEP = 0.05;
248
+
249
+ const toDb = (power: number) => (power > 1e-12 ? 10 * Math.log10(power) : -120);
250
+ const r1 = (n: number) => Math.round(n * 10) / 10;
251
+
252
+ // The measurement itself, from the decoded mix and the words' times. Returns undefined for a film with no words.
253
+ export function voiceAgainstMusic(samples: ArrayLike<number>, rate: number, words: SpokenWord[]): VoiceMusic | undefined {
254
+ const ws = words.filter((w) => w.end > w.start).sort((a, b) => a.start - b.start);
255
+ if (!ws.length || samples.length < rate) return undefined;
256
+ const seconds = samples.length / rate;
257
+ // Prefix sums of the squares, so the power of any stretch is two lookups.
258
+ const acc = new Float64Array(samples.length + 1);
259
+ for (let i = 0; i < samples.length; i++) acc[i + 1] = acc[i]! + samples[i]! * samples[i]!;
260
+ const sumOf = (a: number, b: number): { sum: number; n: number } => {
261
+ const i = Math.max(0, Math.min(samples.length, Math.round(a * rate))), j = Math.max(0, Math.min(samples.length, Math.round(b * rate)));
262
+ return j > i ? { sum: acc[j]! - acc[i]!, n: j - i } : { sum: 0, n: 0 };
263
+ };
264
+ const powerOver = (spans: [number, number][]): number | undefined => {
265
+ let sum = 0, n = 0;
266
+ for (const [a, b] of spans) { const r = sumOf(a, b); sum += r.sum; n += r.n; }
267
+ return n > 0 ? sum / n : undefined;
268
+ };
269
+
270
+ const speech = powerOver(ws.map((w): [number, number] => [w.start + SPEECH_EDGE, w.end - SPEECH_EDGE]));
271
+ if (speech === undefined) return undefined;
272
+
273
+ // The gaps between consecutive words, with the start and the end of the film counted as gaps too (the last second is a fade-out, so it is left out).
274
+ type Gap = { from: number; to: number; after?: SpokenWord };
275
+ const gaps: Gap[] = [{ from: 0, to: ws[0]!.start }];
276
+ for (let i = 0; i + 1 < ws.length; i++) gaps.push({ from: ws[i]!.end, to: ws[i + 1]!.start, after: ws[i] });
277
+ gaps.push({ from: ws[ws.length - 1]!.end, to: Math.max(ws[ws.length - 1]!.end, seconds - 1) });
278
+
279
+ const pauses = gaps.filter((g) => g.to - g.from >= PAUSE_MIN).map((g): [number, number] => [g.from + PAUSE_EDGE, g.to - PAUSE_EDGE]).filter(([a, b]) => b - a >= 0.2);
280
+ const pause = powerOver(pauses);
281
+ // The music held down under the voice: the quiet moments between words (and, with tight gaps, between sentences), the voice's own tail trimmed off.
282
+ // Word times are rounded, so a short pause may still hold the end of a word: take the quietest quarter of its 50 ms blocks, which is the music (its median is what the gaps between sentences are compared with).
283
+ const blocks: number[] = [];
284
+ for (const g of gaps) {
285
+ if (!g.after || g.to - g.from < SHORT_PAUSE_MIN || g.to - g.from >= PAUSE_MIN) continue;
286
+ for (let t = g.from + SHORT_PAUSE_START; t + 0.05 <= g.to - SHORT_PAUSE_END + 1e-9; t += 0.05) { const r = sumOf(t, t + 0.05); if (r.n) blocks.push(r.sum / r.n); }
287
+ }
288
+ blocks.sort((a, b) => a - b);
289
+ const short = blocks.length ? blocks[Math.floor((blocks.length - 1) * 0.25)] : undefined;
290
+ // Between two sentences the music is compared with the typical level of those pauses.
291
+ const typical = blocks.length ? blocks[Math.floor((blocks.length - 1) * 0.5)] : undefined;
292
+
293
+ const out: VoiceMusic = { speechDb: r1(toDb(speech)), estimated: true };
294
+ if (pause !== undefined) { out.pauseDb = r1(toDb(pause)); out.voiceOverPauseDb = r1(toDb(speech) - toDb(pause)); }
295
+ if (short !== undefined) {
296
+ out.underVoiceDb = r1(toDb(short));
297
+ // The voice alone is the speech power less the music's: a level only a few dB above the music leaves little voice, so the floor is a tenth of it.
298
+ out.musicBelowVoiceDb = r1(toDb(Math.max(speech - short, speech * 0.1)) - toDb(short));
299
+ // Between two sentences: the loudest 0.1 s of the gap, apart from the voice's tail at its start and the next word's breath at its end.
300
+ let pumps = 0, looked = 0;
301
+ for (const g of gaps) {
302
+ if (!g.after?.sentenceEnd || g.to - g.from < SENTENCE_GAP_MIN) continue;
303
+ const a = g.from + SENTENCE_GAP_START, b = g.to - SENTENCE_GAP_END;
304
+ if (b - a < PUMP_WINDOW) continue;
305
+ looked++;
306
+ let loudest = -Infinity;
307
+ for (let t = a; t + PUMP_WINDOW <= b + 1e-9; t += PUMP_STEP) { const r = sumOf(t, t + PUMP_WINDOW); if (r.n) loudest = Math.max(loudest, toDb(r.sum / r.n)); }
308
+ if (loudest - toDb(typical ?? short) > PUMP_RISE_DB) pumps++;
309
+ }
310
+ out.pumps = pumps; out.sentenceGaps = looked; out.pumpsPer30s = r1((pumps * 30) / seconds);
311
+ }
312
+ return out;
313
+ }
314
+
315
+ // The words of a narrated film as times in the film, from its manifest: a word ends a sentence when it ends with . ? ! … or is a scene's last.
316
+ export function spokenWords(manifest: { fps: number; scenes: { startFrame: number; words: { word: string; startSec: number; endSec: number }[] }[] }): SpokenWord[] {
317
+ return manifest.scenes.flatMap((s) => s.words.map((w, i) => ({ start: s.startFrame / manifest.fps + w.startSec, end: s.startFrame / manifest.fps + w.endSec, sentenceEnd: i === s.words.length - 1 || /[.!?…؟׃。!?]["'”’»)\]]*$/.test(w.word.trim()) })));
318
+ }
319
+
320
+ // Plain-word lines about the voice and the music, ending with advice when the music is too near the voice or too far from it, or pumps.
321
+ export function voiceAdvice(v: VoiceMusic): string {
322
+ const parts: string[] = [];
323
+ if (v.voiceOverPauseDb !== undefined) parts.push(`the voice speaks ${v.voiceOverPauseDb.toFixed(1)} dB above the level of the pauses (${v.speechDb.toFixed(1)} against ${v.pauseDb!.toFixed(1)} dBFS)`);
324
+ else parts.push(`the voice speaks at ${v.speechDb.toFixed(1)} dBFS and there is no pause of 0.5 s to compare it with`);
325
+ if (v.musicBelowVoiceDb !== undefined) parts.push(`the music under the voice is about ${v.musicBelowVoiceDb.toFixed(0)} dB below it`);
326
+ if (v.pumps !== undefined) parts.push(`${v.pumps} pump${v.pumps === 1 ? "" : "s"} (${v.pumpsPer30s!.toFixed(1)} per 30 s) in ${v.sentenceGaps} gap${v.sentenceGaps === 1 ? "" : "s"} between sentences`);
327
+ const advice: string[] = [];
328
+ if (v.musicBelowVoiceDb !== undefined && v.musicBelowVoiceDb < MUSIC_BELOW_VOICE_DB[0]) advice.push(`The music is too close to the voice (under ${MUSIC_BELOW_VOICE_DB[0]} dB below it) and will compete with it: lower duckTo on <Music>.`);
329
+ if (v.musicBelowVoiceDb !== undefined && v.musicBelowVoiceDb > MUSIC_BELOW_VOICE_DB[1]) advice.push(`The music is hardly there under the voice (over ${MUSIC_BELOW_VOICE_DB[1]} dB below it): raise duckTo on <Music>.`);
330
+ if (v.pumpsPer30s !== undefined && v.pumpsPer30s > MAX_PUMPS_PER_30S) advice.push(`The music pumps: its level rises between sentences more than ${MAX_PUMPS_PER_30S} times in 30 s. Keep the gaps tight (plan gap), leave Music's defaults, and give no volume per scene.`);
331
+ return `Voice against music (estimated from the mix, which has the music and the voice in one signal): ${parts.join("; ")}. ${advice.length ? advice.join(" ") : "Both are inside the targets (the music 10 to 22 dB below the voice, at most 2 pumps in 30 s)."}`;
332
+ }
333
+
334
+ // Decodes the audio and the picture of a finished film and measures them. The loudness is passed in when mastering already measured it.
335
+ export async function measureSound(path: string, loudness?: number, words?: SpokenWord[]): Promise<SoundReport> {
336
+ const audio = await tool("ffmpeg", ["-v", "error", "-i", path, "-vn", "-ac", "1", "-ar", String(AUDIO_RATE), "-f", "f32le", "-"]);
337
+ const samples = new Float32Array(audio.stdout.buffer.slice(audio.stdout.byteOffset, audio.stdout.byteOffset + Math.floor(audio.stdout.length / 4) * 4));
338
+ const info = await probeVideo(path, path);
339
+ const w = PICTURE_WIDTH, h = Math.max(2, Math.round((w * info.height) / info.width / 2) * 2);
340
+ const frames = await tool("ffmpeg", ["-v", "error", "-i", path, "-an", "-vf", `fps=${PICTURE_FPS},scale=${w}:${h}:flags=area,format=gray`, "-f", "rawvideo", "-pix_fmt", "gray", "-"]);
341
+ const diffs = meanAbsDiffs(new Uint8Array(frames.stdout.buffer, frames.stdout.byteOffset, frames.stdout.length), w * h);
342
+ const base = soundReport(samples, AUDIO_RATE, diffs, PICTURE_FPS);
343
+ const voice = words?.length ? voiceAgainstMusic(samples, AUDIO_RATE, words) : undefined;
344
+ const report: SoundReport = voice ? { ...base, voice } : base;
345
+ const lufs = loudness ?? (await loudnessLufs(path));
346
+ return lufs === undefined ? report : { ...report, loudnessLufs: lufs };
347
+ }