reelkit-cli 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/README.md +3 -2
  2. package/package.json +3 -2
  3. package/skill/SKILL.md +11 -5
  4. package/skill/reference/beat-sync.md +54 -0
  5. package/skill/reference/captions.md +18 -2
  6. package/skill/reference/component-authoring.md +12 -1
  7. package/skill/reference/continuity.md +99 -0
  8. package/skill/reference/kit.md +19 -1
  9. package/skill/reference/motion-design.md +9 -0
  10. package/skill/reference/references.md +11 -0
  11. package/skill/reference/remotion-composition.md +2 -1
  12. package/skill/reference/scene-treatments.md +1 -1
  13. package/skill/reference/scriptwriting.md +39 -5
  14. package/skill/reference/sound-design.md +11 -0
  15. package/skill/reference/styles.md +9 -0
  16. package/src/cli.ts +10 -4
  17. package/src/commands/assets.ts +46 -5
  18. package/src/commands/build.ts +99 -11
  19. package/src/commands/components.ts +220 -0
  20. package/src/commands/init.ts +9 -3
  21. package/src/commands/plan.ts +4 -2
  22. package/src/commands/ref.ts +11 -3
  23. package/src/contract/index.ts +23 -3
  24. package/src/pipeline/beatsnap.ts +52 -0
  25. package/src/pipeline/review.ts +54 -1
  26. package/src/pipeline/schema.ts +8 -0
  27. package/src/project/loudness.ts +68 -0
  28. package/src/project/manifest.ts +31 -1
  29. package/src/project/music.ts +25 -0
  30. package/src/project/project.ts +4 -2
  31. package/src/project/refmeasure.ts +45 -1
  32. package/src/remotion/Root.tsx +4 -2
  33. package/src/remotion/kit/Camera.tsx +22 -0
  34. package/src/remotion/kit/Captions.tsx +25 -8
  35. package/src/remotion/kit/Carry.tsx +38 -0
  36. package/src/remotion/kit/Music.tsx +19 -0
  37. package/src/remotion/kit/Sfx.tsx +12 -6
  38. package/src/remotion/kit/beat.ts +23 -0
  39. package/src/remotion/kit/caption-groups.ts +65 -0
  40. package/src/remotion/kit/docs.ts +19 -1
  41. package/src/remotion/kit/index.ts +7 -0
  42. package/src/remotion/kit/media.ts +17 -0
  43. package/src/remotion/kit/motion-math.ts +113 -0
  44. package/src/remotion/kit/music-math.ts +42 -0
  45. package/src/render/continuity.ts +87 -0
  46. package/src/render/master.ts +31 -0
  47. package/src/render/static-check.ts +156 -0
  48. package/src/render/validate.ts +30 -150
  49. package/src/testing/conformance.ts +49 -1
  50. package/src/testing/fake-api.ts +7 -1
@@ -13,7 +13,12 @@
13
13
  // - A voiceover `text` must contain at least one non-space character: whitespace alone is 400 invalid_request.
14
14
  // - A library item `id` (pull, commit) matches `^[A-Za-z0-9][A-Za-z0-9._-]{0,127}$`; a `deviceCode` is 1 to 200 characters of plain text.
15
15
  // - A `voiceId` is 1 to 64 letters and digits and must be one of the voices the server lists (`voices`), otherwise 400 invalid_request.
16
- // - Components are added to the library only by the owner's scripts, never through this API (UploadKindSchema has no `component`).
16
+ // - A user's component is uploaded for review through the same two calls as any other file (`kind: "component"`). The server checks its source
17
+ // at commit with the same static check a pulled component gets (only react, remotion and reelkit/kit imports, no network, no globals that reach
18
+ // outside the video) and refuses a source that fails it with 400 `invalid_request` and the first problem. It stays in `review`, is never found
19
+ // by search or the public listing, and is published only by the owner. A component upload is: `contentType` `text/plain`, `bytes` up to 65,536,
20
+ // `shareable` true, `filename` matching `^[A-Z][A-Za-z0-9]*\.tsx$`, a `description` of at least 20 characters, and `meta.example` a string of at most
21
+ // 2,000 characters that starts with `<` and the file's name (`<StatCard value="42" />` for StatCard.tsx). Anything else is 400 `invalid_request`.
17
22
  // - Upload is a single `PUT` to `uploadUrl` with exactly the declared content type and byte length, and no auth header.
18
23
  // Anything else is refused with a non-2xx status and stores nothing. That refusal comes from the storage service, so its body
19
24
  // is not an API error and a client must not parse it.
@@ -105,6 +110,11 @@ const Meta = z.record(z.string(), z.unknown()).refine((m) => new TextEncoder().e
105
110
 
106
111
  // The most a single upload may be: 200 MB.
107
112
  export const MAX_UPLOAD_BYTES = 209_715_200;
113
+ // A component's source is at most 64 KB, its file name is PascalCase, and its description and example have the sizes below.
114
+ export const MAX_COMPONENT_BYTES = 65_536;
115
+ export const COMPONENT_FILE = /^[A-Z][A-Za-z0-9]*\.tsx$/;
116
+ export const MIN_COMPONENT_DESCRIPTION = 20;
117
+ export const MAX_COMPONENT_EXAMPLE = 2000;
108
118
  // The video types a cutout takes.
109
119
  export const CutoutTypeSchema = z.enum(["video/mp4", "video/quicktime", "video/webm"]);
110
120
  export type CutoutType = z.infer<typeof CutoutTypeSchema>;
@@ -124,8 +134,8 @@ export const ApiErrorSchema = z.object({ error: z.object({ code: z.string(), mes
124
134
 
125
135
  export const LibraryKindSchema = z.enum(["image", "overlay", "sfx", "music", "component", "clip"]);
126
136
  export type LibraryKind = z.infer<typeof LibraryKindSchema>;
127
- // What a user may upload: everything except components, which are code and come only from Reelkit.
128
- export const UploadKindSchema = LibraryKindSchema.exclude(["component"]);
137
+ // What a user may upload: every kind. A component is code and is held for review (see the rules at the top).
138
+ export const UploadKindSchema = LibraryKindSchema;
129
139
  export type UploadKind = z.infer<typeof UploadKindSchema>;
130
140
 
131
141
  export const LibraryItemSchema = z.object({
@@ -183,6 +193,16 @@ export const routes = {
183
193
  kind: UploadKindSchema, title: text(200).min(1), description: text(2000), tags: z.array(text(40)).max(20),
184
194
  meta: Meta, filename: text(200).min(1), contentType: text(100), bytes: z.number().int().positive().max(MAX_UPLOAD_BYTES),
185
195
  shareable: z.boolean(),
196
+ }).superRefine((r, ctx) => {
197
+ if (r.kind !== "component") return;
198
+ const bad = (path: string, message: string) => ctx.addIssue({ code: "custom", path: [path], message });
199
+ if (r.contentType.toLowerCase() !== "text/plain") bad("contentType", "a component must be text/plain");
200
+ if (r.bytes > MAX_COMPONENT_BYTES) bad("bytes", `a component is at most ${MAX_COMPONENT_BYTES} bytes`);
201
+ if (!COMPONENT_FILE.test(r.filename)) bad("filename", "a component file is named like StatCard.tsx");
202
+ if (!r.shareable) bad("shareable", "a component is only uploaded to be shared for review");
203
+ if (r.description.trim().length < MIN_COMPONENT_DESCRIPTION) bad("description", `a component needs a description of at least ${MIN_COMPONENT_DESCRIPTION} characters`);
204
+ const example = r.meta.example;
205
+ if (typeof example !== "string" || example.length > MAX_COMPONENT_EXAMPLE || !example.startsWith(`<${r.filename.replace(/\.tsx$/, "")}`)) bad("meta.example", `a component needs meta.example, at most ${MAX_COMPONENT_EXAMPLE} characters, that starts with <Name`);
186
206
  }),
187
207
  z.object({ id: z.string(), uploadUrl: z.string() })),
188
208
  libraryCommit: route("POST", "/library/upload/commit", true, z.object({ id: ItemId }), z.object({ item: LibraryItemSchema })),
@@ -0,0 +1,52 @@
1
+ // Puts scene changes on the beat. A scene is held a little longer at its end so that the next one starts exactly on a beat.
2
+ // The voiceover and its word times never move: only a scene's length, and so where the later scenes start, changes.
3
+
4
+ // The last scene must keep at least this long after its last word, as the natural padding already does (0.4 s) plus a little.
5
+ export const MIN_TAIL_SEC = 0.5;
6
+
7
+ export type SnapInput = {
8
+ // Each scene's natural length in frames.
9
+ naturalFrames: number[];
10
+ // When the last word of the last scene ends, in frames from the start of that scene.
11
+ lastWordEndFrame: number;
12
+ // Beats as frame numbers, ascending. They should reach past the end of the video.
13
+ beatFrames: number[];
14
+ fps: number;
15
+ };
16
+
17
+ const firstAtOrAfter = (beats: number[], frame: number): number | undefined => beats.find((b) => b >= frame);
18
+
19
+ // The new length of every scene. A scene is only ever made longer, by less than one beat (the last by a few frames more, to keep its tail).
20
+ // A boundary that has no beat close enough keeps its natural place, and the scenes after it follow from there.
21
+ export function snapToBeats(input: SnapInput): number[] {
22
+ const { naturalFrames, beatFrames, fps } = input;
23
+ if (beatFrames.length < 2) return [...naturalFrames];
24
+ const gaps = beatFrames.slice(1).map((b, i) => b - beatFrames[i]!).sort((a, b) => a - b);
25
+ const period = Math.max(1, gaps[Math.floor(gaps.length / 2)]!);
26
+ let cursor = 0;
27
+ return naturalFrames.map((natural, i) => {
28
+ const last = i === naturalFrames.length - 1;
29
+ let end = cursor + natural;
30
+ let limit = period;
31
+ if (last) {
32
+ const tail = cursor + input.lastWordEndFrame + Math.ceil(MIN_TAIL_SEC * fps);
33
+ if (tail > end) { limit += tail - end; end = tail; }
34
+ }
35
+ const beat = firstAtOrAfter(beatFrames, end);
36
+ const target = beat !== undefined && beat - (cursor + natural) <= limit ? beat : end;
37
+ const frames = target - cursor;
38
+ cursor = target;
39
+ return frames;
40
+ });
41
+ }
42
+
43
+ // Beats in seconds, continued at the tempo's own spacing until `untilSec`, for a video longer than the track (the track loops).
44
+ export function extendBeats(beats: number[], bpm: number, untilSec: number): number[] {
45
+ if (!beats.length) return [];
46
+ const period = 60 / bpm;
47
+ const out = [...beats];
48
+ while (out[out.length - 1]! < untilSec) out.push(Math.round((out[out.length - 1]! + period) * 1000) / 1000);
49
+ return out;
50
+ }
51
+
52
+ export const toBeatFrames = (beatsSec: number[], fps: number): number[] => [...new Set(beatsSec.map((t) => Math.round(t * fps)))].sort((a, b) => a - b);
@@ -8,6 +8,30 @@ export function estimateLength(plan: ScenePlan): { words: number; seconds: numbe
8
8
  return { words: total, seconds: total / WORDS_PER_SEC };
9
9
  }
10
10
 
11
+ // Each scene's estimated length in seconds, from its narration at the same pace as the whole-video estimate above.
12
+ export function sceneSeconds(plan: ScenePlan): { id: string; seconds: number }[] {
13
+ return plan.scenes.map((s) => ({ id: s.id, seconds: words(s.narration) / WORDS_PER_SEC }));
14
+ }
15
+
16
+ // How uneven the scene lengths are: the standard deviation over the mean (0 when they are all the same).
17
+ export function lengthVariation(seconds: number[]): number {
18
+ const mean = seconds.reduce((a, b) => a + b, 0) / (seconds.length || 1);
19
+ if (!(mean > 0)) return 0;
20
+ return Math.sqrt(seconds.reduce((a, b) => a + (b - mean) ** 2, 0) / seconds.length) / mean;
21
+ }
22
+
23
+ // A video whose scenes all last about as long feels like a slideshow. With four scenes or more, the lengths should differ a lot.
24
+ export function rhythmNote(plan: ScenePlan): string | undefined {
25
+ if (plan.scenes.length < 4) return undefined;
26
+ const lengths = sceneSeconds(plan);
27
+ const values = lengths.map((l) => l.seconds);
28
+ const mean = values.reduce((a, b) => a + b, 0) / values.length;
29
+ const cv = lengthVariation(values);
30
+ if (cv >= 0.25 && values.some((v) => v < mean / 2)) return undefined;
31
+ const shortest = lengths.reduce((a, b) => (b.seconds < a.seconds ? b : a)), longest = lengths.reduce((a, b) => (b.seconds > a.seconds ? b : a));
32
+ return `The scenes are too even in length (the shortest, ${shortest.id}, is about ${shortest.seconds.toFixed(1)}s and the longest, ${longest.id}, about ${longest.seconds.toFixed(1)}s). Make one or two scenes much shorter (a hit of a few words) and let one run long.`;
33
+ }
34
+
11
35
  // Soft quality checks on a plan that already passes the hard rules in validatePlan.
12
36
  // These are things worth improving in the script; they never fail a video on their own.
13
37
  export function reviewPlan(plan: ScenePlan, ctx: { footage?: AssetRecord }): string[] {
@@ -26,10 +50,39 @@ export function reviewPlan(plan: ScenePlan, ctx: { footage?: AssetRecord }): str
26
50
 
27
51
  for (const s of plan.scenes) {
28
52
  const n = words(s.narration);
29
- if (!ctx.footage && n < 5) issues.push(`Scene ${s.id} has only ${n} words of narration; give it one full sentence.`);
53
+ // A hit of a few words is wanted in the rhythm of a video, so only a scene with almost nothing to say is flagged.
54
+ if (!ctx.footage && n < 3) issues.push(`Scene ${s.id} has only ${n} word${n === 1 ? "" : "s"} of narration; give it at least a short phrase.`);
30
55
  if (n > 45) issues.push(`Scene ${s.id} has ${n} words of narration; split it or cut it to under 45.`);
31
56
  if (s.onScreenText.length > 3) issues.push(`Scene ${s.id} has ${s.onScreenText.length} on-screen text items; use at most 3.`);
32
57
  for (const t of s.onScreenText) if (words(t) > 7) issues.push(`Scene ${s.id}: on-screen text "${t}" is too long; keep each item to 7 words or fewer.`);
33
58
  }
59
+ const rhythm = rhythmNote(plan);
60
+ if (rhythm) issues.push(rhythm);
61
+ issues.push(...structureNotes(plan, ctx));
34
62
  return issues;
35
63
  }
64
+
65
+ const hasPicture = (s: ScenePlan["scenes"][number]) => s.treatment === "illustration" || s.treatment === "clip" || s.userAssetIds.length > 0;
66
+
67
+ // The first sentence of some narration, up to . ! ? … or their Hebrew and Arabic forms.
68
+ export function firstSentence(text: string): string {
69
+ const m = /^[\s\S]*?[.!?…؟׃。!?]+(?=\s|$)/.exec(text.trim());
70
+ return (m ? m[0] : text).trim();
71
+ }
72
+
73
+ // What the plan alone says about whether the video will look like something: pictures, an opening that shows, an opening that gets to the point.
74
+ // Used by `plan check` and, because the composition is built from the plan, by `check` too.
75
+ export function structureNotes(plan: ScenePlan, ctx: { footage?: AssetRecord }): string[] {
76
+ const notes: string[] = [];
77
+ if (ctx.footage) return notes;
78
+ if (plan.scenes.length >= 4 && !plan.scenes.some(hasPicture)) {
79
+ notes.push("Every scene is type and shapes: no illustration, no clip and none of the user's own files. Give at least one scene a picture so the video has something to look at; see reference/scene-treatments.md.");
80
+ }
81
+ const first = plan.scenes[0]!;
82
+ if (!hasPicture(first)) {
83
+ notes.push(`The opening has no picture, clip or user asset (scene ${first.id}). The first second decides whether the video is watched, so open on something to look at; see reference/scriptwriting.md.`);
84
+ }
85
+ const n = words(firstSentence(first.narration));
86
+ if (n > 12) notes.push(`The first sentence of the opening is ${n} words. Open with a sentence of 12 words or fewer; see reference/scriptwriting.md.`);
87
+ return notes;
88
+ }
@@ -28,6 +28,8 @@ export const ScenePlanSchema = z.object({
28
28
  // Optional only so plans stored before voices existed still load; a new plan must choose one.
29
29
  voiceId: z.string().optional().describe("id of the narration voice, one of the ids from `reelkit assets voices`, chosen to suit the idea, audience and language"),
30
30
  pace: z.enum(["slow", "normal", "fast"]).optional().describe("speaking pace: slow for calm or emotional, normal by default, fast for high-energy"),
31
+ // Whether the video has captions and of which kind: none, one word at a time, or a phrase at a time. Absent means "phrase".
32
+ captions: z.enum(["none", "word", "phrase"]).optional().describe("captions: none, one word at a time (word), or a phrase at a time (phrase, the default), as the user chose"),
31
33
  // A video the new one is made "like": what is taken from it is how it feels (structure, pacing, motion), never its footage, music or words.
32
34
  // Optional, so plans stored before references existed still load.
33
35
  reference: z.object({
@@ -83,6 +85,12 @@ export const AssetManifestSchema = z.object({
83
85
  height: z.number(),
84
86
  totalFrames: z.number(),
85
87
  footageKey: z.string().optional(),
88
+ // The video's music track, when one was pulled with `reelkit assets pull <id> --music`. beatFrames are the beats that fall inside the video.
89
+ music: z.object({ key: z.string(), bpm: z.number().optional(), beatFrames: z.array(z.number()) }).optional(),
90
+ // The plan's caption choice, with the default filled in, so the composition can pass it to <Captions group={...}>.
91
+ captions: z.enum(["none", "word", "phrase"]).optional(),
92
+ // Linear gain per sound file path that brings every pulled sound to a common level; the kit's Sfx and Music apply it.
93
+ soundGain: z.record(z.string(), z.number()).optional(),
86
94
  scenes: z.array(ManifestSceneSchema),
87
95
  });
88
96
  export type AssetManifest = z.infer<typeof AssetManifestSchema>;
@@ -0,0 +1,68 @@
1
+ import { audioDuration, tool } from "./refmeasure";
2
+
3
+ // How loud a sound is, measured on this machine, and how much to turn it up or down so that the same `volume` number sounds equally loud
4
+ // for every file.
5
+ // A one-shot (a sound effect under 3 s) is levelled by its loudest moment: its peak is brought to -3 dBFS.
6
+ // A longer sound (music, a long effect) is levelled by how loud it is overall: about -18 LUFS, integrated.
7
+ // The gain is limited to 18 dB either way, so a nearly silent or a broken file is not blown up into noise.
8
+ export const ONE_SHOT_SECONDS = 3;
9
+ export const ONE_SHOT_PEAK_DB = -3;
10
+ export const LONG_TARGET_LUFS = -18;
11
+ export const MAX_GAIN_DB = 18;
12
+
13
+ export type SoundLevel = { durationSec: number; peakDb?: number; lufs?: number };
14
+
15
+ const round = (n: number, digits: number) => Math.round(n * 10 ** digits) / 10 ** digits;
16
+
17
+ // The gain in dB that brings a measured sound to its target, or 0 when it could not be measured (silence).
18
+ export function gainDbFor(level: SoundLevel): number {
19
+ const oneShot = level.durationSec < ONE_SHOT_SECONDS;
20
+ const measured = oneShot ? level.peakDb : level.lufs ?? level.peakDb;
21
+ if (measured === undefined || !Number.isFinite(measured)) return 0;
22
+ const target = oneShot || level.lufs === undefined ? ONE_SHOT_PEAK_DB : LONG_TARGET_LUFS;
23
+ return round(Math.max(-MAX_GAIN_DB, Math.min(MAX_GAIN_DB, target - measured)), 1);
24
+ }
25
+
26
+ export const dbToLinear = (db: number) => round(10 ** (db / 20), 4);
27
+
28
+ // ffmpeg's volumedetect: the loudest sample, in dBFS. Undefined for digital silence.
29
+ export async function peakOf(path: string): Promise<number | undefined> {
30
+ const { stderr } = await tool("ffmpeg", ["-hide_banner", "-nostats", "-i", path, "-vn", "-af", "volumedetect", "-f", "null", "-"]);
31
+ const peak = Number(/max_volume:\s*(-?[0-9.]+)\s*dB/.exec(stderr)?.[1]);
32
+ return Number.isFinite(peak) ? peak : undefined;
33
+ }
34
+
35
+ // What loudnorm's first pass reports; its numbers are text in the JSON it prints, and "-inf" for silence.
36
+ export type LoudnormReport = { input_i: string; input_tp: string; input_lra: string; input_thresh: string; target_offset: string };
37
+
38
+ export function parseLoudnorm(stderr: string): LoudnormReport | undefined {
39
+ const end = stderr.lastIndexOf("}");
40
+ const start = stderr.lastIndexOf("{", end);
41
+ if (start < 0 || end < start) return undefined;
42
+ try {
43
+ const r = JSON.parse(stderr.slice(start, end + 1)) as LoudnormReport;
44
+ return typeof r.input_i === "string" ? r : undefined;
45
+ } catch { return undefined; }
46
+ }
47
+
48
+ export async function measureLoudnorm(path: string, target: { i: number; tp: number }): Promise<LoudnormReport | undefined> {
49
+ const { stderr } = await tool("ffmpeg", ["-hide_banner", "-nostats", "-i", path, "-vn", "-af", `loudnorm=I=${target.i}:TP=${target.tp}:LRA=11:print_format=json`, "-f", "null", "-"]);
50
+ return parseLoudnorm(stderr);
51
+ }
52
+
53
+ // Measures a sound for levelling. A failure to measure is not the caller's failure: it returns undefined and the sound is used as it is.
54
+ export async function levelOf(path: string): Promise<(SoundLevel & { gainDb: number }) | undefined> {
55
+ try {
56
+ const durationSec = await audioDuration(path);
57
+ if (!(durationSec > 0)) return undefined;
58
+ const peakDb = await peakOf(path);
59
+ let lufs: number | undefined;
60
+ if (durationSec >= ONE_SHOT_SECONDS) {
61
+ const r = await measureLoudnorm(path, { i: LONG_TARGET_LUFS, tp: -1.5 });
62
+ const v = Number(r?.input_i);
63
+ if (Number.isFinite(v)) lufs = round(v, 1);
64
+ }
65
+ const level: SoundLevel = { durationSec, ...(peakDb !== undefined ? { peakDb } : {}), ...(lufs !== undefined ? { lufs } : {}) };
66
+ return { ...level, gainDb: gainDbFor(level) };
67
+ } catch { return undefined; }
68
+ }
@@ -1,5 +1,8 @@
1
1
  import { PACE_SPEED, type AssetManifest, type ScenePlan, type WordTiming } from "../pipeline/schema";
2
+ import { extendBeats, snapToBeats, toBeatFrames } from "../pipeline/beatsnap";
2
3
  import { dimensionsFor, FPS, layoutScenes } from "../pipeline/timing";
4
+ import { dbToLinear } from "./loudness";
5
+ import type { MusicRecord } from "./music";
3
6
  import { FILES, type Project } from "./project";
4
7
 
5
8
  // What a scene was recorded from is stored with it, so a changed script or voice is noticed.
@@ -24,7 +27,22 @@ export function buildManifest(project: Project, plan: ScenePlan): AssetManifest
24
27
  const footage = project.footage();
25
28
  const byId = new Map(project.assets().map((a) => [a.id, a]));
26
29
 
27
- const layout = layoutScenes(plan.scenes.map((s) => voiceovers[s.id].durationSec));
30
+ let layout = layoutScenes(plan.scenes.map((s) => voiceovers[s.id].durationSec));
31
+ const music = project.readJsonOr<MusicRecord | undefined>(FILES.music, undefined);
32
+ let beatFrames: number[] = [];
33
+ if (music) {
34
+ beatFrames = toBeatFrames(extendBeats(music.beats, music.bpm ?? 120, layout.totalFrames / FPS + 10), FPS);
35
+ // Scene changes land on beats only when the track has a trusted tempo. Footage has a fixed length that a longer scene must not overrun.
36
+ if (music.bpm && music.beats.length) {
37
+ const last = voiceovers[plan.scenes[plan.scenes.length - 1].id];
38
+ const lastWords = last.words.length ? Math.max(...last.words.map((w) => w.endSec)) : last.durationSec;
39
+ const lengths = snapToBeats({ naturalFrames: layout.scenes.map((s) => s.durationFrames), lastWordEndFrame: Math.ceil(lastWords * FPS), beatFrames, fps: FPS });
40
+ let cursor = 0;
41
+ const snapped = { scenes: lengths.map((durationFrames) => { const r = { startFrame: cursor, durationFrames }; cursor += durationFrames; return r; }), totalFrames: 0 };
42
+ snapped.totalFrames = cursor;
43
+ if (!footage || cursor <= Math.floor((footage.durationSec ?? 0) * FPS)) layout = snapped;
44
+ }
45
+ }
28
46
  let totalFrames = layout.totalFrames;
29
47
  if (footage) {
30
48
  const footageFrames = Math.floor((footage.durationSec ?? 0) * FPS);
@@ -34,11 +52,21 @@ export function buildManifest(project: Project, plan: ScenePlan): AssetManifest
34
52
  totalFrames = footageFrames;
35
53
  }
36
54
 
55
+ // The levelling of every sound that was measured when it was pulled, as linear gains by path.
56
+ const soundGain: Record<string, number> = {};
57
+ for (const e of Object.values(project.readJsonOr<Record<string, { path: string; kind: string; gainDb?: number }>>(FILES.library, {}))) {
58
+ if ((e.kind === "sfx" || e.kind === "music") && typeof e.gainDb === "number") soundGain[e.path] = dbToLinear(e.gainDb);
59
+ }
60
+ if (music) soundGain[music.key] = dbToLinear(music.gainDb);
61
+
37
62
  const manifest: AssetManifest = {
38
63
  fps: FPS,
39
64
  ...dimensionsFor(plan.aspect, footage),
40
65
  totalFrames,
66
+ captions: plan.captions ?? "phrase",
67
+ ...(Object.keys(soundGain).length ? { soundGain } : {}),
41
68
  ...(footage ? { footageKey: footage.key } : {}),
69
+ ...(music ? { music: { key: music.key, ...(music.bpm ? { bpm: music.bpm } : {}), beatFrames: music.bpm ? beatFrames.filter((f) => f < totalFrames) : [] } } : {}),
42
70
  scenes: plan.scenes.map((scene, i) => ({
43
71
  id: scene.id,
44
72
  startFrame: layout.scenes[i].startFrame,
@@ -64,6 +92,8 @@ export function missingAssets(project: Project, plan: ScenePlan): string[] {
64
92
  const images = project.readJsonOr<Record<string, string>>(FILES.images, {});
65
93
  const clips = project.readJsonOr<Record<string, ClipRecord>>(FILES.clips, {});
66
94
  const problems: string[] = [];
95
+ const music = project.readJsonOr<MusicRecord | undefined>(FILES.music, undefined);
96
+ if (music && !project.exists(music.key)) problems.push(`Missing file: ${music.key}. Pull the track again with \`reelkit assets pull ${music.id} --music\`.`);
67
97
  for (const s of plan.scenes) {
68
98
  if (!voiceovers[s.id]) problems.push(`Scene ${s.id} has no voiceover.`);
69
99
  else if (voiceoverStale(voiceovers[s.id], s, plan)) problems.push(`Scene ${s.id}'s voiceover is out of date. Run \`reelkit assets voiceover --all\`.`);
@@ -0,0 +1,25 @@
1
+ import type { AssetManifest } from "../pipeline/schema";
2
+ import { analyzeBeats } from "./beats";
3
+ import { audioDuration, decodeMono } from "./refmeasure";
4
+
5
+ // The video's one music track, stored in assets/music.json. `bpm` and `beats` are present only when the pulse was clear enough to trust.
6
+ export type MusicRecord = { key: string; id: string; title: string; durationSec: number; bpm?: number; beatConfidence: number; beats: number[]; gainDb: number };
7
+
8
+ const RATE = 11025;
9
+
10
+ // Measures a pulled track on this machine: its length, its tempo and where its beats fall.
11
+ export async function measureMusic(path: string): Promise<Pick<MusicRecord, "durationSec" | "bpm" | "beatConfidence" | "beats">> {
12
+ const durationSec = Math.round((await audioDuration(path)) * 1000) / 1000;
13
+ const r = analyzeBeats(await decodeMono(path, RATE), RATE);
14
+ return { durationSec, beatConfidence: r.confidence, beats: r.tempoBpm !== undefined && r.beats ? r.beats : [], ...(r.tempoBpm !== undefined ? { bpm: r.tempoBpm } : {}) };
15
+ }
16
+
17
+ // How many scene changes fall on a beat, for the preview.
18
+ export function beatReport(manifest: AssetManifest): { bpm: number; boundariesOnBeat: number; boundaries: number; line: string } | undefined {
19
+ const m = manifest.music;
20
+ if (!m?.bpm || manifest.scenes.length < 2) return undefined;
21
+ const beats = new Set(m.beatFrames);
22
+ const starts = manifest.scenes.slice(1).map((s) => s.startFrame);
23
+ const on = starts.filter((f) => beats.has(f)).length;
24
+ return { bpm: m.bpm, boundariesOnBeat: on, boundaries: starts.length, line: `${on} of ${starts.length} scene changes land on the beat at ${Math.round(m.bpm)} BPM` };
25
+ }
@@ -5,14 +5,16 @@ import { AspectSchema, type AssetRecord } from "../pipeline/schema";
5
5
 
6
6
  export const FILES = {
7
7
  config: "reelkit.json", plan: "plan.json", manifest: "manifest.json",
8
- assetIndex: "assets/index.json", voiceovers: "assets/voiceovers.json", images: "assets/images.json", clips: "assets/clips.json", library: "assets/library.json",
8
+ assetIndex: "assets/index.json", voiceovers: "assets/voiceovers.json", images: "assets/images.json", clips: "assets/clips.json", library: "assets/library.json", music: "assets/music.json", searches: "assets/searches.json", shared: "assets/shared.json",
9
9
  } as const;
10
10
 
11
11
  // "path: message", or just the message for a problem at the root of the file.
12
12
  export const issueLines = (e: { issues: { path: PropertyKey[]; message: string }[] }) =>
13
13
  e.issues.map((i) => (i.path.length ? `${i.path.map(String).join(".")}: ${i.message}` : i.message));
14
14
 
15
- export const ProjectConfigSchema = z.object({ aspect: AspectSchema, name: z.string().optional(), footage: z.string().optional() });
15
+ export const ProjectConfigSchema = z.object({ aspect: AspectSchema, name: z.string().optional(), footage: z.string().optional(),
16
+ // false when the project was made with --private: nothing is shared with the library automatically.
17
+ shareComponents: z.boolean().optional() });
16
18
  export type ProjectConfig = z.infer<typeof ProjectConfigSchema>;
17
19
 
18
20
  // One video's working folder. Every command reads and writes its state here.
@@ -9,7 +9,7 @@ const exec = promisify(execFile);
9
9
  export const REF_ASPECTS = ["9:16", "16:9", "1:1", "4:5"] as const;
10
10
  export type RefAspect = (typeof REF_ASPECTS)[number];
11
11
 
12
- async function tool(cmd: "ffmpeg" | "ffprobe", args: string[]): Promise<{ stdout: Buffer; stderr: string }> {
12
+ export async function tool(cmd: "ffmpeg" | "ffprobe", args: string[]): Promise<{ stdout: Buffer; stderr: string }> {
13
13
  try {
14
14
  const r = await exec(cmd, args, { maxBuffer: 256 * 1024 * 1024, encoding: "buffer" });
15
15
  return { stdout: r.stdout, stderr: r.stderr.toString("utf8") };
@@ -146,3 +146,47 @@ export async function audioDuration(path: string): Promise<number> {
146
146
  const { stdout } = await tool("ffprobe", ["-v", "error", "-show_entries", "format=duration", "-of", "csv=p=0", path]);
147
147
  return Number(stdout.toString("utf8").trim());
148
148
  }
149
+
150
+ // The mean absolute difference, from 0 to 255, between each frame of a run of grey frames and the next. frameSize is the pixels in one frame.
151
+ export function meanAbsDiffs(frames: Uint8Array, frameSize: number): number[] {
152
+ const count = Math.floor(frames.length / frameSize);
153
+ const out: number[] = [];
154
+ for (let f = 0; f + 1 < count; f++) {
155
+ let sum = 0;
156
+ for (let i = 0, a = f * frameSize, b = a + frameSize; i < frameSize; i++) sum += Math.abs(frames[a + i]! - frames[b + i]!);
157
+ out.push(sum / frameSize);
158
+ }
159
+ return out;
160
+ }
161
+
162
+ // Two frames whose pictures differ by less than this, on average over 255 grey levels, are the same picture for a viewer: video
163
+ // compression noise on a held shot stays well under it, and a slow push still stays above it.
164
+ const STILL_BELOW = 1;
165
+
166
+ // The share of the video's time in which the picture barely changes from one sampled frame to the next, and the longest unbroken stretch
167
+ // of it. diffs has one entry for each pair of consecutive samples, taken `rate` times a second.
168
+ export function stillness(diffs: number[], rate: number): { share: number; longestSec: number } {
169
+ if (!diffs.length) return { share: 0, longestSec: 0 };
170
+ let still = 0, run = 0, longest = 0;
171
+ for (const d of diffs) {
172
+ if (d < STILL_BELOW) { still++; longest = Math.max(longest, ++run); } else run = 0;
173
+ }
174
+ const round = (n: number) => Math.round(n * 1000) / 1000;
175
+ return { share: round(still / diffs.length), longestSec: round(longest / rate) };
176
+ }
177
+
178
+ // Sampled at 10 frames a second and 64 pixels wide, a ten-minute video is about 44 MB of grey pixels and is measured in a second or two.
179
+ export const STILL_SAMPLE = { rate: 10, width: 64 };
180
+
181
+ export async function stillnessOfVideo(video: string, width: number, height: number): Promise<{ share: number; longestSec: number }> {
182
+ const w = STILL_SAMPLE.width, h = Math.max(2, Math.round((w * height) / width / 2) * 2);
183
+ const { stdout } = await tool("ffmpeg", ["-v", "error", "-i", video, "-an", "-vf", `fps=${STILL_SAMPLE.rate},scale=${w}:${h}:flags=area,format=gray`, "-f", "rawvideo", "-pix_fmt", "gray", "-"]);
184
+ return stillness(meanAbsDiffs(new Uint8Array(stdout.buffer, stdout.byteOffset, stdout.length), w * h), STILL_SAMPLE.rate);
185
+ }
186
+
187
+ // The standard deviation of the values over their mean: 0 when they are all equal.
188
+ export function coefficientOfVariation(values: number[]): number {
189
+ const mean = values.reduce((a, b) => a + b, 0) / (values.length || 1);
190
+ if (!(mean > 0)) return 0;
191
+ return Math.sqrt(values.reduce((a, b) => a + (b - mean) ** 2, 0) / values.length) / mean;
192
+ }
@@ -1,6 +1,6 @@
1
1
  import React from "react";
2
2
  import { Composition } from "remotion";
3
- import { UrlsContext } from "./kit/media";
3
+ import { ManifestContext, UrlsContext } from "./kit/media";
4
4
  import type { VideoProps } from "./types";
5
5
 
6
6
  const EMPTY: VideoProps = { manifest: { fps: 30, width: 1080, height: 1920, totalFrames: 1, scenes: [] }, urls: {} };
@@ -8,7 +8,9 @@ const EMPTY: VideoProps = { manifest: { fps: 30, width: 1080, height: 1920, tota
8
8
  export const makeRoot = (Video: React.FC<VideoProps>): React.FC => {
9
9
  const WithMedia: React.FC<VideoProps> = (props) => (
10
10
  <UrlsContext.Provider value={props.urls}>
11
- <Video {...props} />
11
+ <ManifestContext.Provider value={props.manifest}>
12
+ <Video {...props} />
13
+ </ManifestContext.Provider>
12
14
  </UrlsContext.Provider>
13
15
  );
14
16
  return () => (
@@ -0,0 +1,22 @@
1
+ import React from "react";
2
+ import { AbsoluteFill, useCurrentFrame, useVideoConfig } from "remotion";
3
+ import { cameraAt, cameraTransform, type CameraKey } from "./motion-math";
4
+
5
+ // Moves the whole picture. Wrap all the scenes in it for one camera that never cuts, or wrap one scene's content.
6
+ // Each key gives the point of the content, as fractions of the frame, that sits at the middle of the screen (x, y), the zoom (1 shows
7
+ // the content as it fits) and a turn in degrees; what a key leaves out stays as it was. The move toward a key starts `lead` frames
8
+ // before its frame (12 unless set) on a spring and lands on the key's frame. drift is a very slow push on top, as a fraction of the
9
+ // zoom per second (0.02 is two percent a second), so a held shot is never perfectly still.
10
+ export const Camera: React.FC<{
11
+ keys: CameraKey[];
12
+ drift?: number;
13
+ lead?: number;
14
+ stiffness?: number;
15
+ damping?: number;
16
+ children: React.ReactNode;
17
+ }> = ({ keys, drift, lead, stiffness, damping, children }) => {
18
+ const frame = useCurrentFrame();
19
+ const { fps, width, height } = useVideoConfig();
20
+ const { origin, transform } = cameraTransform(cameraAt(keys, frame, fps, { lead, stiffness, damping, drift }), width, height);
21
+ return <AbsoluteFill style={{ transformOrigin: origin, transform }}>{children}</AbsoluteFill>;
22
+ };
@@ -1,6 +1,7 @@
1
1
  import React from "react";
2
2
  import { AbsoluteFill, interpolate, spring, useCurrentFrame, useVideoConfig } from "remotion";
3
3
  import type { WordTiming } from "../../pipeline/schema";
4
+ import { captionGroups, groupAt, type CaptionGroup } from "./caption-groups";
4
5
  import { fonts, springs } from "./theme";
5
6
 
6
7
  export type CaptionMode = "highlight" | "pop" | "karaoke";
@@ -9,34 +10,50 @@ export type CaptionMode = "highlight" | "pop" | "karaoke";
9
10
  // highlight: a line of words, the spoken one in the highlight colour (calm, educational)
10
11
  // pop: one to three words at a time, each popping in as it is spoken (high energy)
11
12
  // karaoke: a line that fills with the highlight colour as it is spoken (voiceover, music)
13
+ // group, when given, decides which words share the screen and replaces perLine:
14
+ // "word": exactly one word at a time
15
+ // "phrase": words grouped as they are spoken: a group ends at a sentence end, at a comma or dash after three words, or at a pause of
16
+ // 0.35 s, and never holds more than 6 words or about 32 characters
17
+ // "none": nothing is drawn (the plan's `captions: "none"`), so the manifest's value can be passed straight through
12
18
  export const Captions: React.FC<{
13
19
  words: WordTiming[];
14
20
  mode?: CaptionMode;
15
21
  highlight?: string;
16
22
  color?: string;
17
23
  perLine?: number;
24
+ group?: CaptionGroup;
18
25
  uppercase?: boolean;
19
26
  // Distance of the caption baseline from the bottom, as a fraction of the height. Default keeps it clear of platform UI.
20
27
  bottom?: number;
21
28
  // Font family. Set to font("heebo") or font("rubik") for Hebrew; rtl also reverses the word order on the line.
22
29
  face?: string;
23
30
  rtl?: boolean;
24
- }> = ({ words, mode = "highlight", highlight = "#ffe14d", color = "#ffffff", perLine, uppercase = false, bottom = 0.16, face, rtl = false }) => {
31
+ }> = ({ words, mode = "highlight", highlight = "#ffe14d", color = "#ffffff", perLine, group, uppercase = false, bottom = 0.16, face, rtl = false }) => {
25
32
  const frame = useCurrentFrame();
26
33
  const { fps, width, height } = useVideoConfig();
27
34
  const t = frame / fps;
28
- if (words.length === 0) return null;
35
+ if (words.length === 0 || group === "none") return null;
29
36
 
30
- const size = perLine ?? (mode === "pop" ? 2 : 4);
31
- let active = words.findIndex((w) => t < w.endSec);
32
- if (active === -1) active = words.length - 1;
33
- const lineStart = Math.floor(active / size) * size;
34
- const line = words.slice(lineStart, lineStart + size);
37
+ if (group === "word" && t < words[0]!.startSec) return null;
38
+ let active: number, lineStart: number, line: WordTiming[];
39
+ if (group) {
40
+ const groups = captionGroups(words, group);
41
+ const at = groupAt(words, groups, t);
42
+ active = at.active;
43
+ [lineStart] = groups[at.index]!;
44
+ line = words.slice(lineStart, groups[at.index]![1]);
45
+ } else {
46
+ const size = perLine ?? (mode === "pop" ? 2 : 4);
47
+ active = words.findIndex((w) => t < w.endSec);
48
+ if (active === -1) active = words.length - 1;
49
+ lineStart = Math.floor(active / size) * size;
50
+ line = words.slice(lineStart, lineStart + size);
51
+ }
35
52
  const show = (w: WordTiming) => (uppercase ? w.word.toUpperCase() : w.word);
36
53
 
37
54
  const base: React.CSSProperties = {
38
55
  fontFamily: face ?? fonts.body, fontWeight: 800, textAlign: "center", direction: rtl ? "rtl" : "ltr", padding: `0 ${width * 0.06}px`,
39
- fontSize: width * (mode === "pop" ? 0.075 : 0.055), lineHeight: 1.15, color,
56
+ fontSize: width * (mode === "pop" || group === "word" ? 0.075 : 0.055), lineHeight: 1.15, color,
40
57
  // A dark outline plus shadow keeps contrast above 4.5:1 on any background.
41
58
  WebkitTextStroke: `${width * 0.006}px rgba(0,0,0,0.85)`, paintOrder: "stroke fill",
42
59
  textShadow: "0 4px 18px rgba(0,0,0,0.7)",
@@ -0,0 +1,38 @@
1
+ import React from "react";
2
+ import { useCurrentFrame, useVideoConfig } from "remotion";
3
+ import { boxAt, type Box, type BoxKey } from "./motion-math";
4
+
5
+ // What a function child is given: the box now, as fractions of the frame, and its size in pixels.
6
+ export type CarryBox = Box & { widthPx: number; heightPx: number };
7
+
8
+ // One element that lives outside the scenes and is held through several of them, so that it is still on screen when a scene changes
9
+ // and moves, grows or turns into what the next scene needs. Place it once, beside the SceneFrames and above them.
10
+ // Each key says where the box must have arrived by the given absolute frame: x and y are its centre, width and height are
11
+ // fractions of the frame, radius is the corner radius as a fraction of the frame width, rotate is in degrees. The move toward a key
12
+ // starts `lead` frames before its frame (12 unless set) on a spring and lands on the key's frame exactly. Before the first key nothing is
13
+ // drawn unless that key sets an opacity; after the last the box holds.
14
+ export const Carry: React.FC<{
15
+ keys: BoxKey[];
16
+ lead?: number;
17
+ stiffness?: number;
18
+ damping?: number;
19
+ children: React.ReactNode | ((box: CarryBox) => React.ReactNode);
20
+ }> = ({ keys, lead, stiffness, damping, children }) => {
21
+ const frame = useCurrentFrame();
22
+ const { fps, width, height } = useVideoConfig();
23
+ const box = boxAt(keys, frame, fps, { lead, stiffness, damping });
24
+ if (!box || box.opacity <= 0) return null;
25
+ const widthPx = box.width * width, heightPx = box.height * height;
26
+ return (
27
+ <div
28
+ style={{
29
+ position: "absolute", left: (box.x - box.width / 2) * width, top: (box.y - box.height / 2) * height, width: widthPx, height: heightPx,
30
+ opacity: box.opacity, transform: box.rotate ? `rotate(${box.rotate}deg)` : undefined, borderRadius: box.radius * width, overflow: box.radius > 0 ? "hidden" : undefined,
31
+ // The layer sits above the scenes; it must not be the thing a pointer-like overlay thinks it hits.
32
+ pointerEvents: "none",
33
+ }}
34
+ >
35
+ {typeof children === "function" ? children({ ...box, widthPx, heightPx }) : children}
36
+ </div>
37
+ );
38
+ };
@@ -0,0 +1,19 @@
1
+ import React from "react";
2
+ import { Audio, useVideoConfig } from "remotion";
3
+ import { musicVolume, speechSpans, type SpeechScene } from "./music-math";
4
+ import { useGain, useManifest, useMedia } from "./media";
5
+
6
+ // The video's music track. Place it once, outside the scenes. It sits at `duckTo` while the voiceover is heard and comes up to `volume`
7
+ // in the gaps, loops when the video is longer than the track, and fades out over the last second. `scenes` defaults to the manifest's.
8
+ export const Music: React.FC<{ src: string; volume?: number; duckTo?: number; scenes?: SpeechScene[] }> = ({ src, volume = 0.5, duckTo = 0.12, scenes }) => {
9
+ const { fps, durationInFrames } = useVideoConfig();
10
+ const manifest = useManifest();
11
+ const gain = useGain(src);
12
+ const spans = React.useMemo(() => speechSpans(scenes ?? manifest?.scenes ?? [], fps), [scenes, manifest, fps]);
13
+ return (
14
+ <Audio
15
+ src={useMedia(src)} loop loopVolumeCurveBehavior="extend"
16
+ volume={(f) => musicVolume(f, { fps, totalFrames: durationInFrames, volume, duckTo, spans, gain })}
17
+ />
18
+ );
19
+ };