reelkit-cli 0.5.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -2
- package/package.json +7 -2
- package/skill/SKILL.md +22 -11
- package/skill/THIRD_PARTY.md +102 -0
- package/skill/commands/launch-film.md +7 -0
- package/skill/reference/asset-reuse.md +13 -2
- package/skill/reference/backgrounds.md +63 -0
- package/skill/reference/beat-sync.md +32 -17
- package/skill/reference/captions.md +11 -5
- package/skill/reference/component-authoring.md +12 -1
- package/skill/reference/continuity.md +21 -2
- package/skill/reference/kit.md +140 -11
- package/skill/reference/launch-film.md +190 -0
- package/skill/reference/remotion-composition.md +4 -3
- package/skill/reference/scene-treatments.md +20 -0
- package/skill/reference/scriptwriting.md +4 -1
- package/skill/reference/three-d.md +134 -0
- package/skill/reference/voice-sync.md +108 -0
- package/src/agents.ts +23 -12
- package/src/api/client.ts +4 -1
- package/src/cli.ts +25 -8
- package/src/commands/assets.ts +334 -36
- package/src/commands/build.ts +172 -36
- package/src/commands/components.ts +220 -0
- package/src/commands/init.ts +9 -3
- package/src/commands/install.ts +1 -1
- package/src/commands/plan.ts +8 -5
- package/src/commands/ref.ts +5 -2
- package/src/contract/index.ts +27 -5
- package/src/pipeline/beatsnap.ts +72 -0
- package/src/pipeline/review.ts +67 -7
- package/src/pipeline/schema.ts +51 -4
- package/src/pipeline/timing.ts +27 -1
- package/src/project/background.ts +33 -0
- package/src/project/layers.ts +60 -0
- package/src/project/loudness.ts +68 -0
- package/src/project/manifest.ts +68 -14
- package/src/project/music.ts +19 -5
- package/src/project/project.ts +7 -2
- package/src/project/refmeasure.ts +1 -1
- package/src/project/soundreport.ts +347 -0
- package/src/project/svgcheck.ts +21 -0
- package/src/remotion/kit/Assemble3D.tsx +92 -0
- package/src/remotion/kit/BrowserFrame.tsx +83 -0
- package/src/remotion/kit/Camera.tsx +6 -4
- package/src/remotion/kit/Captions.tsx +33 -17
- package/src/remotion/kit/Card3D.tsx +211 -0
- package/src/remotion/kit/ChapterFrame.tsx +68 -0
- package/src/remotion/kit/CounterRoll.tsx +75 -0
- package/src/remotion/kit/GlassPanel.tsx +43 -0
- package/src/remotion/kit/Grounds.tsx +177 -0
- package/src/remotion/kit/Headline.tsx +97 -0
- package/src/remotion/kit/Hero3D.tsx +197 -0
- package/src/remotion/kit/HudOverlay.tsx +52 -0
- package/src/remotion/kit/ImageLayers.tsx +48 -0
- package/src/remotion/kit/Music.tsx +4 -4
- package/src/remotion/kit/NamedCursor.tsx +54 -0
- package/src/remotion/kit/Orbit3D.tsx +49 -0
- package/src/remotion/kit/Particles3D.tsx +74 -0
- package/src/remotion/kit/Place.tsx +12 -0
- package/src/remotion/kit/PromptBox.tsx +84 -0
- package/src/remotion/kit/Scene3D.tsx +70 -0
- package/src/remotion/kit/SceneFrame.tsx +88 -11
- package/src/remotion/kit/Sfx.tsx +12 -6
- package/src/remotion/kit/SoundCues.tsx +22 -0
- package/src/remotion/kit/TerminalLog.tsx +98 -0
- package/src/remotion/kit/Text3D.tsx +78 -0
- package/src/remotion/kit/TextOnImage.tsx +41 -0
- package/src/remotion/kit/Warp3D.tsx +59 -0
- package/src/remotion/kit/bg-math.ts +179 -0
- package/src/remotion/kit/caption-groups.ts +7 -3
- package/src/remotion/kit/caption-style.ts +45 -0
- package/src/remotion/kit/docs.ts +133 -11
- package/src/remotion/kit/image-layers-math.ts +115 -0
- package/src/remotion/kit/index.ts +43 -1
- package/src/remotion/kit/inter-bold-typeface.ts +3 -0
- package/src/remotion/kit/media.ts +5 -3
- package/src/remotion/kit/motion-math.ts +36 -2
- package/src/remotion/kit/music-math.ts +27 -10
- package/src/remotion/kit/quiet-three.ts +11 -0
- package/src/remotion/kit/sample-text.ts +55 -0
- package/src/remotion/kit/scene3d-context.ts +5 -0
- package/src/remotion/kit/seeded.ts +13 -0
- package/src/remotion/kit/sound-cues.ts +89 -0
- package/src/remotion/kit/sound-kinds.ts +122 -0
- package/src/remotion/kit/theme.ts +2 -0
- package/src/remotion/kit/three-fx-math.ts +192 -0
- package/src/remotion/kit/three-math.ts +145 -0
- package/src/remotion/kit/transition-math.ts +116 -0
- package/src/remotion/kit/ui-math.ts +145 -0
- package/src/remotion/kit/ui-theme.ts +25 -0
- package/src/remotion/kit/word-anchor.ts +107 -0
- package/src/render/contact-sheet.ts +39 -0
- package/src/render/continuity.ts +14 -4
- package/src/render/deps.ts +15 -3
- package/src/render/master.ts +31 -0
- package/src/render/render.ts +15 -8
- package/src/render/sound-notes.ts +106 -0
- package/src/render/static-check.ts +156 -0
- package/src/render/validate.ts +3 -150
- package/src/render/word-check.ts +181 -0
- package/src/testing/conformance.ts +61 -1
- package/src/testing/fake-api.ts +11 -5
- package/src/testing/fixtures.ts +3 -0
package/src/commands/plan.ts
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import { client, type Ctx, type Result } from "../context";
|
|
2
2
|
import { loadCredentials } from "../credentials";
|
|
3
3
|
import { estimateLength, lengthVariation, reviewPlan, sceneSeconds } from "../pipeline/review";
|
|
4
|
-
import { ScenePlanSchema, validatePlan, type ScenePlan } from "../pipeline/schema";
|
|
4
|
+
import { isVoiceless, ScenePlanSchema, validatePlan, type ScenePlan } from "../pipeline/schema";
|
|
5
|
+
import { snappedSeconds } from "../project/manifest";
|
|
5
6
|
import { FILES, issueLines, openProject, type Project } from "../project/project";
|
|
6
7
|
|
|
7
8
|
export function loadPlan(project: Project): ScenePlan {
|
|
@@ -52,7 +53,8 @@ export async function planCheck(ctx: Ctx): Promise<Result> {
|
|
|
52
53
|
}
|
|
53
54
|
const footage = project.footage();
|
|
54
55
|
const assets = project.assets().filter((a) => a.id !== footage?.id);
|
|
55
|
-
const
|
|
56
|
+
const voiceless = isVoiceless(parsed.data);
|
|
57
|
+
const voiceIds = !voiceless && loadCredentials(ctx.env) ? (await client(ctx)("voices", {})).voices.map((v) => v.id) : undefined;
|
|
56
58
|
const ref = referenceChecks(project, parsed.data);
|
|
57
59
|
const mustFix = [...validatePlan(parsed.data, { aspect: project.config().aspect, footage, assets, voiceIds }), ...ref.mustFix];
|
|
58
60
|
const shouldImprove = mustFix.length ? [] : [...reviewPlan(parsed.data, { footage }), ...ref.improve];
|
|
@@ -60,11 +62,12 @@ export async function planCheck(ctx: Ctx): Promise<Result> {
|
|
|
60
62
|
const estimatedSeconds = Math.round(length.seconds);
|
|
61
63
|
const perScene = sceneSeconds(parsed.data).map((s) => ({ id: s.id, seconds: Math.round(s.seconds * 10) / 10 }));
|
|
62
64
|
const variation = Math.round(lengthVariation(sceneSeconds(parsed.data).map((s) => s.seconds)) * 100) / 100;
|
|
65
|
+
const snapped = voiceless ? snappedSeconds(project, parsed.data) : undefined;
|
|
63
66
|
const lines = [
|
|
64
67
|
mustFix.length ? `Fix these:\n- ${mustFix.join("\n- ")}` : "The plan is valid.",
|
|
65
|
-
`Estimated length: about ${estimatedSeconds}s (${length.words} words).`,
|
|
68
|
+
voiceless ? `Length: ${(Math.round(length.seconds * 10) / 10).toFixed(1)}s over ${parsed.data.scenes.length} scenes${snapped !== undefined ? `; about ${snapped.toFixed(1)}s once the cuts are on the beat` : ""}.` : `Estimated length: about ${estimatedSeconds}s (${length.words} words).`,
|
|
66
69
|
...(shouldImprove.length ? [`Worth improving:\n- ${shouldImprove.join("\n- ")}`] : []),
|
|
67
|
-
...(voiceIds ? [] : ["The voice was not checked because you are not logged in. Run `reelkit auth login`."]),
|
|
70
|
+
...(voiceIds || voiceless ? [] : ["The voice was not checked because you are not logged in. Run `reelkit auth login`."]),
|
|
68
71
|
];
|
|
69
|
-
return { ok: mustFix.length === 0, data: { mustFix, shouldImprove, estimatedSeconds, words: length.words, sceneSeconds: perScene, lengthVariation: variation }, summary: lines.join("\n") };
|
|
72
|
+
return { ok: mustFix.length === 0, data: { mustFix, shouldImprove, estimatedSeconds, words: length.words, sceneSeconds: perScene, lengthVariation: variation, ...(snapped !== undefined ? { snappedSeconds: Math.round(snapped * 10) / 10 } : {}) }, summary: lines.join("\n") };
|
|
70
73
|
}
|
package/src/commands/ref.ts
CHANGED
|
@@ -60,11 +60,14 @@ const realYtdlp: YtdlpRunner = {
|
|
|
60
60
|
|
|
61
61
|
const YTDLP_MISSING = "yt-dlp is not installed, and links need it. Install it (macOS: `brew install yt-dlp`, or `pipx install yt-dlp`) and run the command again; a video file on this machine works without it.";
|
|
62
62
|
|
|
63
|
+
// The duration filter uses "<=?": some sites (Instagram) give no duration before the download, and without the "?" such a video is
|
|
64
|
+
// skipped as if it were too long. A video that turns out longer is refused after it is probed.
|
|
65
|
+
// No "--max-downloads 1": yt-dlp exits with an error code after reaching it even when the one video was saved.
|
|
63
66
|
// One video, at most 1080p, as mp4, and nothing longer than the limit: the flags that keep a link from fetching a playlist, a channel or a huge file.
|
|
64
67
|
export function ytdlpArgs(url: string, dest: string): string[] {
|
|
65
68
|
return [
|
|
66
|
-
"--ignore-config", "--no-playlist", "--
|
|
67
|
-
"--match-filter", `duration
|
|
69
|
+
"--ignore-config", "--no-playlist", "--no-progress", "--no-warnings",
|
|
70
|
+
"--match-filter", `duration<=?${MAX_REF_SECONDS}`,
|
|
68
71
|
"-f", "bv*[height<=1080][ext=mp4]+ba[ext=m4a]/b[height<=1080][ext=mp4]/bv*[height<=1080]+ba/b[height<=1080]",
|
|
69
72
|
"--merge-output-format", "mp4", "-o", dest, "--", url,
|
|
70
73
|
];
|
package/src/contract/index.ts
CHANGED
|
@@ -13,7 +13,12 @@
|
|
|
13
13
|
// - A voiceover `text` must contain at least one non-space character: whitespace alone is 400 invalid_request.
|
|
14
14
|
// - A library item `id` (pull, commit) matches `^[A-Za-z0-9][A-Za-z0-9._-]{0,127}$`; a `deviceCode` is 1 to 200 characters of plain text.
|
|
15
15
|
// - A `voiceId` is 1 to 64 letters and digits and must be one of the voices the server lists (`voices`), otherwise 400 invalid_request.
|
|
16
|
-
// -
|
|
16
|
+
// - A user's component is uploaded for review through the same two calls as any other file (`kind: "component"`). The server checks its source
|
|
17
|
+
// at commit with the same static check a pulled component gets (only react, remotion and reelkit/kit imports, no network, no globals that reach
|
|
18
|
+
// outside the video) and refuses a source that fails it with 400 `invalid_request` and the first problem. It stays in `review`, is never found
|
|
19
|
+
// by search or the public listing, and is published only by the owner. A component upload is: `contentType` `text/plain`, `bytes` up to 65,536,
|
|
20
|
+
// `shareable` true, `filename` matching `^[A-Z][A-Za-z0-9]*\.tsx$`, a `description` of at least 20 characters, and `meta.example` a string of at most
|
|
21
|
+
// 2,000 characters that starts with `<` and the file's name (`<StatCard value="42" />` for StatCard.tsx). Anything else is 400 `invalid_request`.
|
|
17
22
|
// - Upload is a single `PUT` to `uploadUrl` with exactly the declared content type and byte length, and no auth header.
|
|
18
23
|
// Anything else is refused with a non-2xx status and stores nothing. That refusal comes from the storage service, so its body
|
|
19
24
|
// is not an API error and a client must not parse it.
|
|
@@ -105,6 +110,11 @@ const Meta = z.record(z.string(), z.unknown()).refine((m) => new TextEncoder().e
|
|
|
105
110
|
|
|
106
111
|
// The most a single upload may be: 200 MB.
|
|
107
112
|
export const MAX_UPLOAD_BYTES = 209_715_200;
|
|
113
|
+
// A component's source is at most 64 KB, its file name is PascalCase, and its description and example have the sizes below.
|
|
114
|
+
export const MAX_COMPONENT_BYTES = 65_536;
|
|
115
|
+
export const COMPONENT_FILE = /^[A-Z][A-Za-z0-9]*\.tsx$/;
|
|
116
|
+
export const MIN_COMPONENT_DESCRIPTION = 20;
|
|
117
|
+
export const MAX_COMPONENT_EXAMPLE = 2000;
|
|
108
118
|
// The video types a cutout takes.
|
|
109
119
|
export const CutoutTypeSchema = z.enum(["video/mp4", "video/quicktime", "video/webm"]);
|
|
110
120
|
export type CutoutType = z.infer<typeof CutoutTypeSchema>;
|
|
@@ -124,8 +134,8 @@ export const ApiErrorSchema = z.object({ error: z.object({ code: z.string(), mes
|
|
|
124
134
|
|
|
125
135
|
export const LibraryKindSchema = z.enum(["image", "overlay", "sfx", "music", "component", "clip"]);
|
|
126
136
|
export type LibraryKind = z.infer<typeof LibraryKindSchema>;
|
|
127
|
-
// What a user may upload:
|
|
128
|
-
export const UploadKindSchema = LibraryKindSchema
|
|
137
|
+
// What a user may upload: every kind. A component is code and is held for review (see the rules at the top).
|
|
138
|
+
export const UploadKindSchema = LibraryKindSchema;
|
|
129
139
|
export type UploadKind = z.infer<typeof UploadKindSchema>;
|
|
130
140
|
|
|
131
141
|
export const LibraryItemSchema = z.object({
|
|
@@ -183,6 +193,16 @@ export const routes = {
|
|
|
183
193
|
kind: UploadKindSchema, title: text(200).min(1), description: text(2000), tags: z.array(text(40)).max(20),
|
|
184
194
|
meta: Meta, filename: text(200).min(1), contentType: text(100), bytes: z.number().int().positive().max(MAX_UPLOAD_BYTES),
|
|
185
195
|
shareable: z.boolean(),
|
|
196
|
+
}).superRefine((r, ctx) => {
|
|
197
|
+
if (r.kind !== "component") return;
|
|
198
|
+
const bad = (path: string, message: string) => ctx.addIssue({ code: "custom", path: [path], message });
|
|
199
|
+
if (r.contentType.toLowerCase() !== "text/plain") bad("contentType", "a component must be text/plain");
|
|
200
|
+
if (r.bytes > MAX_COMPONENT_BYTES) bad("bytes", `a component is at most ${MAX_COMPONENT_BYTES} bytes`);
|
|
201
|
+
if (!COMPONENT_FILE.test(r.filename)) bad("filename", "a component file is named like StatCard.tsx");
|
|
202
|
+
if (!r.shareable) bad("shareable", "a component is only uploaded to be shared for review");
|
|
203
|
+
if (r.description.trim().length < MIN_COMPONENT_DESCRIPTION) bad("description", `a component needs a description of at least ${MIN_COMPONENT_DESCRIPTION} characters`);
|
|
204
|
+
const example = r.meta.example;
|
|
205
|
+
if (typeof example !== "string" || example.length > MAX_COMPONENT_EXAMPLE || !example.startsWith(`<${r.filename.replace(/\.tsx$/, "")}`)) bad("meta.example", `a component needs meta.example, at most ${MAX_COMPONENT_EXAMPLE} characters, that starts with <Name`);
|
|
186
206
|
}),
|
|
187
207
|
z.object({ id: z.string(), uploadUrl: z.string() })),
|
|
188
208
|
libraryCommit: route("POST", "/library/upload/commit", true, z.object({ id: ItemId }), z.object({ item: LibraryItemSchema })),
|
|
@@ -190,10 +210,12 @@ export const routes = {
|
|
|
190
210
|
voiceover: route("POST", "/voiceover", true,
|
|
191
211
|
z.object({ text: text(5000).refine((t) => /\S/.test(t), "must contain a non-space character"), voiceId: z.string().regex(/^[A-Za-z0-9]{1,64}$/).optional(), speed: z.number().min(0.7).max(1.3).optional() }),
|
|
192
212
|
z.object({ url: z.string(), ext: z.enum(["mp3", "wav", "m4a"]), contentType: z.string(), durationSec: z.number().positive(), words: z.array(WordTimingSchema), chars: z.number() })),
|
|
213
|
+
// `format` is "png" (a raster image, the default) or "svg" (a vector graphic). An SVG counts as one image against the image quota, takes about 45 s, and the
|
|
214
|
+
// server returns only a cleaned SVG: no scripts, no event attributes, no outside references. Its `ext` is "svg" and its contentType `image/svg+xml`.
|
|
193
215
|
images: route("POST", "/images", true,
|
|
194
|
-
z.object({ prompt: text(2000).min(1), aspect: AspectSchema, shareable: z.boolean(), tags: z.array(text(40)).max(12) }),
|
|
216
|
+
z.object({ prompt: text(2000).min(1), aspect: AspectSchema, shareable: z.boolean(), tags: z.array(text(40)).max(12), format: z.enum(["png", "svg"]).default("png") }),
|
|
195
217
|
// libraryId is set when the image was also added to the shared library.
|
|
196
|
-
z.object({ url: z.string(), ext: z.enum(["png", "jpg", "jpeg", "webp"]), contentType: z.string(), libraryId: z.string().optional() })),
|
|
218
|
+
z.object({ url: z.string(), ext: z.enum(["png", "jpg", "jpeg", "webp", "svg"]), contentType: z.string(), libraryId: z.string().optional() })),
|
|
197
219
|
clipStart: route("POST", "/clips", true,
|
|
198
220
|
z.object({
|
|
199
221
|
prompt: text(2000).min(1), aspect: AspectSchema, greenScreen: z.boolean().default(false), durationSec: z.union([z.literal(5), z.literal(10)]).default(5),
|
package/src/pipeline/beatsnap.ts
CHANGED
|
@@ -12,6 +12,9 @@ export type SnapInput = {
|
|
|
12
12
|
// Beats as frame numbers, ascending. They should reach past the end of the video.
|
|
13
13
|
beatFrames: number[];
|
|
14
14
|
fps: number;
|
|
15
|
+
// "next" holds a scene to the next whole beat (narration cannot be cut). "nearest" moves each end to the closest point on a finer grid
|
|
16
|
+
// (a film with no voice can be shortened as well as lengthened, so its total stays near the plan).
|
|
17
|
+
grid?: "next" | "nearest";
|
|
15
18
|
};
|
|
16
19
|
|
|
17
20
|
const firstAtOrAfter = (beats: number[], frame: number): number | undefined => beats.find((b) => b >= frame);
|
|
@@ -21,6 +24,7 @@ const firstAtOrAfter = (beats: number[], frame: number): number | undefined => b
|
|
|
21
24
|
export function snapToBeats(input: SnapInput): number[] {
|
|
22
25
|
const { naturalFrames, beatFrames, fps } = input;
|
|
23
26
|
if (beatFrames.length < 2) return [...naturalFrames];
|
|
27
|
+
if (input.grid === "nearest") return snapToNearest(naturalFrames, beatFrames);
|
|
24
28
|
const gaps = beatFrames.slice(1).map((b, i) => b - beatFrames[i]!).sort((a, b) => a - b);
|
|
25
29
|
const period = Math.max(1, gaps[Math.floor(gaps.length / 2)]!);
|
|
26
30
|
let cursor = 0;
|
|
@@ -40,6 +44,74 @@ export function snapToBeats(input: SnapInput): number[] {
|
|
|
40
44
|
});
|
|
41
45
|
}
|
|
42
46
|
|
|
47
|
+
// How far a scene change of a narrated film may move to reach a beat, in frames, either way.
|
|
48
|
+
export const NARRATED_SNAP_FRAMES = 4;
|
|
49
|
+
// A boundary never comes closer than this to the last word of the scene it ends, in frames.
|
|
50
|
+
export const NARRATED_WORD_CLEARANCE = 2;
|
|
51
|
+
|
|
52
|
+
// Scene changes for a narrated film. The voice sets where a scene ends; a boundary is moved to the nearest beat only when that is at most
|
|
53
|
+
// `maxMove` frames away either way and leaves at least `clearance` frames after the scene's last word (a boundary never cuts into a word). The
|
|
54
|
+
// next scene's audio starts on its first frame, so it moves with the boundary. Every other boundary stays where the narration puts it, and the
|
|
55
|
+
// last scene is not touched: the film's end is not a scene change. Each boundary is judged from where its scene actually starts, so
|
|
56
|
+
// nothing adds up. `lastWordEndFrames[i]` is when scene i's last word ends, in frames from its own start.
|
|
57
|
+
export function snapNarrated(input: { naturalFrames: number[]; lastWordEndFrames: number[]; beatFrames: number[]; maxMove?: number; clearance?: number }): { frames: number[]; onBeat: boolean[] } {
|
|
58
|
+
const { naturalFrames, lastWordEndFrames, beatFrames } = input;
|
|
59
|
+
const maxMove = input.maxMove ?? NARRATED_SNAP_FRAMES, clearance = input.clearance ?? NARRATED_WORD_CLEARANCE;
|
|
60
|
+
let cursor = 0;
|
|
61
|
+
const onBeat: boolean[] = [];
|
|
62
|
+
const frames = naturalFrames.map((natural, i) => {
|
|
63
|
+
let end = cursor + natural;
|
|
64
|
+
if (i < naturalFrames.length - 1) {
|
|
65
|
+
const floor = cursor + (lastWordEndFrames[i] ?? 0) + clearance;
|
|
66
|
+
let best: number | undefined;
|
|
67
|
+
for (const b of beatFrames) {
|
|
68
|
+
if (Math.abs(b - end) > maxMove || b < floor) continue;
|
|
69
|
+
if (best === undefined || Math.abs(b - end) < Math.abs(best - end) || (Math.abs(b - end) === Math.abs(best - end) && b > best)) best = b;
|
|
70
|
+
}
|
|
71
|
+
if (best !== undefined) end = best;
|
|
72
|
+
onBeat.push(beatFrames.includes(end));
|
|
73
|
+
}
|
|
74
|
+
const length = end - cursor;
|
|
75
|
+
cursor = end;
|
|
76
|
+
return length;
|
|
77
|
+
});
|
|
78
|
+
return { frames, onBeat };
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
// The beat frames with `divisions - 1` evenly spaced points between each pair, rounded to frames. The last beat is followed by one more
|
|
82
|
+
// at the median spacing, so a point just after it exists too.
|
|
83
|
+
export function gridPoints(beatFrames: number[], divisions: number): number[] {
|
|
84
|
+
if (beatFrames.length < 2) return [...beatFrames];
|
|
85
|
+
const gaps = beatFrames.slice(1).map((b, i) => b - beatFrames[i]!).sort((a, b) => a - b);
|
|
86
|
+
const period = Math.max(1, gaps[Math.floor(gaps.length / 2)]!);
|
|
87
|
+
const all = [...beatFrames, beatFrames[beatFrames.length - 1]! + period];
|
|
88
|
+
const out = new Set<number>();
|
|
89
|
+
for (let i = 0; i + 1 < all.length; i++) for (let k = 0; k < divisions; k++) out.add(Math.round(all[i]! + ((all[i + 1]! - all[i]!) * k) / divisions));
|
|
90
|
+
out.add(all[all.length - 1]!);
|
|
91
|
+
return [...out].sort((a, b) => a - b);
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// A scene of at least 1.5 beats ends on the nearest whole beat; a shorter one on the nearest half beat, and under half a beat the nearest
|
|
95
|
+
// quarter. The target is the planned end (the sum of the natural lengths), so rounding errors do not add up; a scene is never made empty.
|
|
96
|
+
function snapToNearest(naturalFrames: number[], beatFrames: number[]): number[] {
|
|
97
|
+
const gaps = beatFrames.slice(1).map((b, i) => b - beatFrames[i]!).sort((a, b) => a - b);
|
|
98
|
+
const period = Math.max(1, gaps[Math.floor(gaps.length / 2)]!);
|
|
99
|
+
const grids = new Map<number, number[]>();
|
|
100
|
+
const grid = (d: number) => grids.get(d) ?? (grids.set(d, gridPoints(beatFrames, d)), grids.get(d)!);
|
|
101
|
+
let cursor = 0, planned = 0;
|
|
102
|
+
return naturalFrames.map((natural) => {
|
|
103
|
+
planned += natural;
|
|
104
|
+
const beats = natural / period;
|
|
105
|
+
const points = grid(beats >= 1.5 ? 1 : beats >= 0.5 ? 2 : 4).filter((f) => f > cursor);
|
|
106
|
+
let target = planned;
|
|
107
|
+
if (points.length) target = points.reduce((best, f) => (Math.abs(f - planned) < Math.abs(best - planned) ? f : best));
|
|
108
|
+
else target = Math.max(planned, cursor + 1);
|
|
109
|
+
const frames = target - cursor;
|
|
110
|
+
cursor = target;
|
|
111
|
+
return frames;
|
|
112
|
+
});
|
|
113
|
+
}
|
|
114
|
+
|
|
43
115
|
// Beats in seconds, continued at the tempo's own spacing until `untilSec`, for a video longer than the track (the track loops).
|
|
44
116
|
export function extendBeats(beats: number[], bpm: number, untilSec: number): number[] {
|
|
45
117
|
if (!beats.length) return [];
|
package/src/pipeline/review.ts
CHANGED
|
@@ -1,16 +1,23 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { gapSec, isVoiceless, LAST_TAIL_SEC, PACE_SPEED, VOICE_WORDS_PER_SEC, type AssetRecord, type ScenePlan } from "./schema";
|
|
2
2
|
|
|
3
3
|
const words = (s: string) => s.trim().split(/\s+/).filter(Boolean).length;
|
|
4
4
|
|
|
5
|
-
//
|
|
5
|
+
// What the voice speaks at, in words a second, for this plan's pace.
|
|
6
|
+
const speechRate = (plan: ScenePlan) => VOICE_WORDS_PER_SEC * PACE_SPEED[plan.pace ?? "normal"];
|
|
7
|
+
|
|
8
|
+
// The estimate the length advice below is based on: each scene's words at the voice's real speed (2.25 words a second at the normal pace, the
|
|
9
|
+
// pauses within a sentence included), the plan's silence between sentences after every scene but the last, and the last scene's tail after its
|
|
10
|
+
// last word. It matches the manifest `reelkit assets voiceover` builds. A video with no voice has no words to count: its length is the sum of the
|
|
11
|
+
// scenes' own seconds.
|
|
6
12
|
export function estimateLength(plan: ScenePlan): { words: number; seconds: number } {
|
|
13
|
+
if (isVoiceless(plan)) return { words: 0, seconds: plan.scenes.reduce((n, s) => n + (s.seconds ?? 0), 0) };
|
|
7
14
|
const total = plan.scenes.reduce((n, s) => n + words(s.narration), 0);
|
|
8
|
-
return { words: total, seconds: total /
|
|
15
|
+
return { words: total, seconds: total / speechRate(plan) + (plan.scenes.length - 1) * gapSec(plan) + LAST_TAIL_SEC };
|
|
9
16
|
}
|
|
10
17
|
|
|
11
|
-
// Each scene's estimated length in seconds,
|
|
18
|
+
// Each scene's estimated length in seconds: its speech and what follows it (the gap, or the tail for the last scene), so they add up to the whole estimate.
|
|
12
19
|
export function sceneSeconds(plan: ScenePlan): { id: string; seconds: number }[] {
|
|
13
|
-
return plan.scenes.map((s) => ({ id: s.id, seconds: words(s.narration) /
|
|
20
|
+
return plan.scenes.map((s, i) => ({ id: s.id, seconds: isVoiceless(plan) ? (s.seconds ?? 0) : words(s.narration) / speechRate(plan) + (i === plan.scenes.length - 1 ? LAST_TAIL_SEC : gapSec(plan)) }));
|
|
14
21
|
}
|
|
15
22
|
|
|
16
23
|
// How uneven the scene lengths are: the standard deviation over the mean (0 when they are all the same).
|
|
@@ -29,12 +36,14 @@ export function rhythmNote(plan: ScenePlan): string | undefined {
|
|
|
29
36
|
const cv = lengthVariation(values);
|
|
30
37
|
if (cv >= 0.25 && values.some((v) => v < mean / 2)) return undefined;
|
|
31
38
|
const shortest = lengths.reduce((a, b) => (b.seconds < a.seconds ? b : a)), longest = lengths.reduce((a, b) => (b.seconds > a.seconds ? b : a));
|
|
32
|
-
|
|
39
|
+
const fix = isVoiceless(plan) ? "Make one or two scenes much shorter (under a second) and let one run long." : "Make one or two scenes much shorter (a hit of a few words) and let one run long.";
|
|
40
|
+
return `The scenes are too even in length (the shortest, ${shortest.id}, is about ${shortest.seconds.toFixed(1)}s and the longest, ${longest.id}, about ${longest.seconds.toFixed(1)}s). ${fix}`;
|
|
33
41
|
}
|
|
34
42
|
|
|
35
43
|
// Soft quality checks on a plan that already passes the hard rules in validatePlan.
|
|
36
44
|
// These are things worth improving in the script; they never fail a video on their own.
|
|
37
45
|
export function reviewPlan(plan: ScenePlan, ctx: { footage?: AssetRecord }): string[] {
|
|
46
|
+
if (isVoiceless(plan)) return reviewVoicelessPlan(plan, ctx);
|
|
38
47
|
const issues: string[] = [];
|
|
39
48
|
const { words: total, seconds } = estimateLength(plan);
|
|
40
49
|
|
|
@@ -53,10 +62,61 @@ export function reviewPlan(plan: ScenePlan, ctx: { footage?: AssetRecord }): str
|
|
|
53
62
|
// A hit of a few words is wanted in the rhythm of a video, so only a scene with almost nothing to say is flagged.
|
|
54
63
|
if (!ctx.footage && n < 3) issues.push(`Scene ${s.id} has only ${n} word${n === 1 ? "" : "s"} of narration; give it at least a short phrase.`);
|
|
55
64
|
if (n > 45) issues.push(`Scene ${s.id} has ${n} words of narration; split it or cut it to under 45.`);
|
|
56
|
-
if (s.onScreenText.length >
|
|
65
|
+
if (s.onScreenText.length > 5) issues.push(`Scene ${s.id} has ${s.onScreenText.length} on-screen text items; use at most 5 (the items of one list on screen count as one text, but the plan cannot say which they are).`);
|
|
66
|
+
for (const t of s.onScreenText) if (words(t) > 7) issues.push(`Scene ${s.id}: on-screen text "${t}" is too long; keep each item to 7 words or fewer.`);
|
|
67
|
+
}
|
|
68
|
+
const rhythm = rhythmNote(plan);
|
|
69
|
+
if (rhythm) issues.push(rhythm);
|
|
70
|
+
issues.push(...structureNotes(plan, ctx));
|
|
71
|
+
return issues;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
// The same kind of advice for a video with no voice: no words to count, so length and rhythm come from `seconds`, and every word the
|
|
75
|
+
// viewer reads is on screen. A video of this kind needs at least one quick hit, so a plan with no scene under a second is flagged.
|
|
76
|
+
function reviewVoicelessPlan(plan: ScenePlan, ctx: { footage?: AssetRecord }): string[] {
|
|
77
|
+
const issues: string[] = [];
|
|
78
|
+
const { seconds } = estimateLength(plan);
|
|
79
|
+
if (seconds < 10) issues.push(`The video runs ${seconds.toFixed(1)}s. A video with no voice needs at least 10 seconds to get across; aim for 15 to 45.`);
|
|
80
|
+
if (seconds > 60) issues.push(`The video runs ${seconds.toFixed(1)}s. Cut it to 60 seconds or less; aim for 15 to 45.`);
|
|
81
|
+
if (!plan.scenes.some((s) => (s.seconds ?? 0) < 1)) issues.push("No scene is under 1 second. A film like this needs at least one quick hit; shorten one or two scenes to under a second.");
|
|
82
|
+
if (!ctx.footage) {
|
|
83
|
+
const illustrations = plan.scenes.filter((s) => s.treatment === "illustration").length;
|
|
84
|
+
if (illustrations > Math.ceil(plan.scenes.length / 2)) issues.push(`${illustrations} of ${plan.scenes.length} scenes are illustrations. Use illustrations for at most half the scenes.`);
|
|
85
|
+
}
|
|
86
|
+
for (const s of plan.scenes) {
|
|
87
|
+
if (!s.onScreenText.length && !s.notes.trim()) issues.push(`Scene ${s.id} has neither onScreenText nor notes. Say what is on screen, since no voice explains it.`);
|
|
88
|
+
if (s.onScreenText.length > 5) issues.push(`Scene ${s.id} has ${s.onScreenText.length} on-screen text items; use at most 5 (the items of one list on screen count as one text, but the plan cannot say which they are).`);
|
|
57
89
|
for (const t of s.onScreenText) if (words(t) > 7) issues.push(`Scene ${s.id}: on-screen text "${t}" is too long; keep each item to 7 words or fewer.`);
|
|
58
90
|
}
|
|
59
91
|
const rhythm = rhythmNote(plan);
|
|
60
92
|
if (rhythm) issues.push(rhythm);
|
|
93
|
+
issues.push(...structureNotes(plan, ctx));
|
|
61
94
|
return issues;
|
|
62
95
|
}
|
|
96
|
+
|
|
97
|
+
const hasPicture = (s: ScenePlan["scenes"][number]) => s.treatment === "illustration" || s.treatment === "clip" || s.userAssetIds.length > 0;
|
|
98
|
+
|
|
99
|
+
// The first sentence of some narration, up to . ! ? … or their Hebrew and Arabic forms.
|
|
100
|
+
export function firstSentence(text: string): string {
|
|
101
|
+
const m = /^[\s\S]*?[.!?…؟׃。!?]+(?=\s|$)/.exec(text.trim());
|
|
102
|
+
return (m ? m[0] : text).trim();
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// What the plan alone says about whether the video will look like something: pictures, an opening that shows, an opening that gets to the point.
|
|
106
|
+
// Used by `plan check` and, because the composition is built from the plan, by `check` too.
|
|
107
|
+
export function structureNotes(plan: ScenePlan, ctx: { footage?: AssetRecord }): string[] {
|
|
108
|
+
const notes: string[] = [];
|
|
109
|
+
if (ctx.footage) return notes;
|
|
110
|
+
// A film with no voice is made of type and interface pieces on purpose (reference/launch-film.md), so it is not asked for a picture.
|
|
111
|
+
const voiceless = isVoiceless(plan);
|
|
112
|
+
if (!voiceless && plan.scenes.length >= 4 && !plan.scenes.some(hasPicture)) {
|
|
113
|
+
notes.push("Every scene is type and shapes: no illustration, no clip and none of the user's own files. Give at least one scene a picture so the video has something to look at; see reference/scene-treatments.md.");
|
|
114
|
+
}
|
|
115
|
+
const first = plan.scenes[0]!;
|
|
116
|
+
if (!voiceless && !hasPicture(first)) {
|
|
117
|
+
notes.push(`The opening has no picture, clip or user asset (scene ${first.id}). The first second decides whether the video is watched, so open on something to look at; see reference/scriptwriting.md.`);
|
|
118
|
+
}
|
|
119
|
+
const n = isVoiceless(plan) ? 0 : words(firstSentence(first.narration));
|
|
120
|
+
if (n > 12) notes.push(`The first sentence of the opening is ${n} words. Open with a sentence of 12 words or fewer; see reference/scriptwriting.md.`);
|
|
121
|
+
return notes;
|
|
122
|
+
}
|
package/src/pipeline/schema.ts
CHANGED
|
@@ -7,7 +7,10 @@ export type Aspect = z.infer<typeof AspectSchema>;
|
|
|
7
7
|
|
|
8
8
|
export const SceneSchema = z.object({
|
|
9
9
|
id: z.string().regex(/^[a-z0-9-]+$/).describe("short unique id, lowercase letters, digits, dashes"),
|
|
10
|
-
|
|
10
|
+
// The minimum of one character is enforced for a narrated plan in ScenePlanSchema below, so that a video without a voice may leave it empty.
|
|
11
|
+
narration: z.string().describe("the words spoken in this scene; an empty string when the plan has voice \"none\""),
|
|
12
|
+
// How long the scene lasts. Used only when the plan has voice "none"; with a voice the recording sets the length.
|
|
13
|
+
seconds: z.number().min(0.3).max(15).optional().describe("the scene's length in seconds, from 0.3 to 15; required when the plan has voice \"none\""),
|
|
11
14
|
// clip: a generated or reused video clip is the scene's picture.
|
|
12
15
|
treatment: z.enum(["motion-graphic", "illustration", "footage-overlay", "clip"]),
|
|
13
16
|
onScreenText: z.array(z.string()),
|
|
@@ -18,9 +21,14 @@ export const SceneSchema = z.object({
|
|
|
18
21
|
imageTags: z.array(z.string()).describe("3 to 6 short tags describing the illustration; empty when imagePrompt is null"),
|
|
19
22
|
userAssetIds: z.array(z.string()),
|
|
20
23
|
notes: z.string().describe("visual direction for whoever writes the composition"),
|
|
24
|
+
// Says that this scene begins on a deliberate hard cut, so `preview` does not count the missing carry-over against the film.
|
|
25
|
+
cutIn: z.boolean().optional().describe("true when this scene begins on a deliberate hard cut (nothing is meant to carry over into it)"),
|
|
21
26
|
});
|
|
22
27
|
export type Scene = z.infer<typeof SceneSchema>;
|
|
23
28
|
|
|
29
|
+
export const MAX_NARRATED_SCENES = 8;
|
|
30
|
+
export const MAX_VOICELESS_SCENES = 16;
|
|
31
|
+
|
|
24
32
|
export const ScenePlanSchema = z.object({
|
|
25
33
|
title: z.string(),
|
|
26
34
|
aspect: AspectSchema,
|
|
@@ -28,18 +36,45 @@ export const ScenePlanSchema = z.object({
|
|
|
28
36
|
// Optional only so plans stored before voices existed still load; a new plan must choose one.
|
|
29
37
|
voiceId: z.string().optional().describe("id of the narration voice, one of the ids from `reelkit assets voices`, chosen to suit the idea, audience and language"),
|
|
30
38
|
pace: z.enum(["slow", "normal", "fast"]).optional().describe("speaking pace: slow for calm or emotional, normal by default, fast for high-energy"),
|
|
39
|
+
// The silence between one scene's last word and the next scene's first word: tight 0.2 s, normal 0.3 s (the default), relaxed 0.5 s.
|
|
40
|
+
// (`pace` is the speed of the voice itself, so the gap has its own setting.)
|
|
41
|
+
gap: z.enum(["tight", "normal", "relaxed"]).optional().describe("silence between one sentence and the next: tight 0.2 s, normal 0.3 s (the default), relaxed 0.5 s; a tight promo is tight"),
|
|
42
|
+
// Absent means narrated. "none" is a video carried by music and sound effects alone: scenes get their length from `seconds`.
|
|
43
|
+
voice: z.enum(["narrated", "none"]).optional().describe("narrated (the default), or none for a video with no voice"),
|
|
31
44
|
// Whether the video has captions and of which kind: none, one word at a time, or a phrase at a time. Absent means "phrase".
|
|
32
|
-
captions: z.enum(["none", "word", "phrase"]).optional().describe("captions: none, one word at a time (word), or a
|
|
45
|
+
captions: z.enum(["none", "word", "phrase"]).optional().describe("captions: none, one word at a time (word), or a few words at a time (phrase, the default; a few words at a time is phrase, not word), as the user chose"),
|
|
33
46
|
// A video the new one is made "like": what is taken from it is how it feels (structure, pacing, motion), never its footage, music or words.
|
|
34
47
|
// Optional, so plans stored before references existed still load.
|
|
35
48
|
reference: z.object({
|
|
36
49
|
id: z.string().describe("the id of a reference in this project, from `reelkit ref list`"),
|
|
37
50
|
take: z.array(z.string()).min(1).max(6).describe("short notes on what is taken from it, e.g. \"fast cuts every ~1.2s\", \"big type on colour fields\""),
|
|
38
51
|
}).optional(),
|
|
39
|
-
scenes: z.array(SceneSchema).min(3).max(
|
|
52
|
+
scenes: z.array(SceneSchema).min(3).max(MAX_VOICELESS_SCENES),
|
|
53
|
+
}).superRefine((plan, ctx) => {
|
|
54
|
+
if (plan.voice === "none") {
|
|
55
|
+
plan.scenes.forEach((s, i) => {
|
|
56
|
+
if (s.seconds === undefined) ctx.addIssue({ code: "custom", path: ["scenes", i, "seconds"], message: `Scene ${s.id} needs seconds (a number from 0.3 to 15) because the video has no voice. Add it, or remove voice from the plan.` });
|
|
57
|
+
});
|
|
58
|
+
return;
|
|
59
|
+
}
|
|
60
|
+
// The narrated rules, as they were before voice existed.
|
|
61
|
+
if (plan.scenes.length > MAX_NARRATED_SCENES) ctx.addIssue({ code: "too_big", origin: "array", maximum: MAX_NARRATED_SCENES, inclusive: true, input: plan.scenes, path: ["scenes"] });
|
|
62
|
+
plan.scenes.forEach((s, i) => {
|
|
63
|
+
if (s.narration.length < 1) ctx.addIssue({ code: "too_small", origin: "string", minimum: 1, inclusive: true, input: s.narration, path: ["scenes", i, "narration"] });
|
|
64
|
+
});
|
|
40
65
|
});
|
|
41
66
|
|
|
67
|
+
export const isVoiceless = (plan: { voice?: "narrated" | "none" }): boolean => plan.voice === "none";
|
|
68
|
+
|
|
42
69
|
export const PACE_SPEED = { slow: 0.92, normal: 1, fast: 1.1 } as const;
|
|
70
|
+
// The silence, in seconds, between one scene's last word and the next scene's first word, by the plan's `gap`.
|
|
71
|
+
export const GAP_SEC = { tight: 0.2, normal: 0.3, relaxed: 0.5 } as const;
|
|
72
|
+
export const gapSec = (plan: { gap?: keyof typeof GAP_SEC }): number => GAP_SEC[plan.gap ?? "normal"];
|
|
73
|
+
// The last scene keeps this long after its last word, so that the film does not end on the last syllable.
|
|
74
|
+
export const LAST_TAIL_SEC = 0.7;
|
|
75
|
+
// The voice speaks about this many words a second at `pace: "normal"` (measured on a recorded film: 2.0 to 2.4, 2.25 over the whole of it, the
|
|
76
|
+
// pauses inside a sentence included). `plan check` estimates the length from it.
|
|
77
|
+
export const VOICE_WORDS_PER_SEC = 2.25;
|
|
43
78
|
export type ScenePlan = z.infer<typeof ScenePlanSchema>;
|
|
44
79
|
|
|
45
80
|
export const AssetRecordSchema = z.object({
|
|
@@ -76,6 +111,16 @@ export const ManifestSceneSchema = z.object({
|
|
|
76
111
|
clipKey: z.string().optional(),
|
|
77
112
|
clipKeyedKey: z.string().optional(),
|
|
78
113
|
userAssetKeys: z.array(z.string()),
|
|
114
|
+
// The scene's picture in layers, from `reelkit assets layers`: the picture itself, the subject cut out of it when it has a clear one (a PNG with alpha beside it) and its box,
|
|
115
|
+
// and where words can sit (`region`, how light and how busy). Pass it to <ImageLayers layers={...}> and <TextOnImage layers={...}>.
|
|
116
|
+
imageLayers: z.object({
|
|
117
|
+
back: z.string(),
|
|
118
|
+
subject: z.string().optional(),
|
|
119
|
+
subjectBox: z.object({ x: z.number(), y: z.number(), w: z.number(), h: z.number() }).optional(),
|
|
120
|
+
textZone: z.object({ region: z.enum(["top", "middle", "bottom", "left", "right"]), luminance: z.number(), busy: z.number() }),
|
|
121
|
+
}).optional(),
|
|
122
|
+
// The picture chosen as this scene's ground, with `reelkit assets pull <id> --background --scene <id>`.
|
|
123
|
+
background: z.object({ key: z.string() }).optional(),
|
|
79
124
|
});
|
|
80
125
|
export type ManifestScene = z.infer<typeof ManifestSceneSchema>;
|
|
81
126
|
|
|
@@ -91,6 +136,8 @@ export const AssetManifestSchema = z.object({
|
|
|
91
136
|
captions: z.enum(["none", "word", "phrase"]).optional(),
|
|
92
137
|
// Linear gain per sound file path that brings every pulled sound to a common level; the kit's Sfx and Music apply it.
|
|
93
138
|
soundGain: z.record(z.string(), z.number()).optional(),
|
|
139
|
+
// The film's ground (a video or a picture), chosen with `--background`; durationSec is set for a video.
|
|
140
|
+
background: z.object({ key: z.string(), durationSec: z.number().optional() }).optional(),
|
|
94
141
|
scenes: z.array(ManifestSceneSchema),
|
|
95
142
|
});
|
|
96
143
|
export type AssetManifest = z.infer<typeof AssetManifestSchema>;
|
|
@@ -102,7 +149,7 @@ export function validatePlan(
|
|
|
102
149
|
ctx: { aspect: Aspect; footage?: AssetRecord; assets: AssetRecord[]; voiceIds?: string[] },
|
|
103
150
|
): string[] {
|
|
104
151
|
const errors: string[] = [];
|
|
105
|
-
if (ctx.voiceIds) {
|
|
152
|
+
if (ctx.voiceIds && !isVoiceless(plan)) {
|
|
106
153
|
if (!plan.voiceId) errors.push("Choose a narration voice: set voiceId to one of the ids from `reelkit assets voices`.");
|
|
107
154
|
else if (!ctx.voiceIds.includes(plan.voiceId)) errors.push(`voiceId ${plan.voiceId} is not one of the available voices.`);
|
|
108
155
|
}
|
package/src/pipeline/timing.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type
|
|
1
|
+
import { LAST_TAIL_SEC, type Aspect, type AssetRecord, type WordTiming } from "./schema";
|
|
2
2
|
|
|
3
3
|
export const FPS = 30;
|
|
4
4
|
export const SCENE_PADDING_SEC = 0.4;
|
|
@@ -28,3 +28,29 @@ export function dimensionsFor(aspect: Aspect, footage?: AssetRecord) {
|
|
|
28
28
|
if (aspect === "1:1") return { width: 1080, height: 1080 };
|
|
29
29
|
return { width: 1080, height: 1920 };
|
|
30
30
|
}
|
|
31
|
+
|
|
32
|
+
// A narrated scene's audio: when its last word ends (seconds from its start) and when its first word starts. A recording with no word times
|
|
33
|
+
// counts as one stretch of speech as long as the file.
|
|
34
|
+
export type SpokenScene = { durationSec: number; words: WordTiming[] };
|
|
35
|
+
export const lastWordEndSec = (v: SpokenScene): number => (v.words.length ? Math.max(...v.words.map((w) => w.endSec)) : v.durationSec);
|
|
36
|
+
export const firstWordStartSec = (v: SpokenScene): number => (v.words.length ? Math.min(...v.words.map((w) => w.startSec)) : 0);
|
|
37
|
+
|
|
38
|
+
// The boundary never comes closer than this after a scene's last word, however tight the gap is.
|
|
39
|
+
export const MIN_AFTER_WORD_SEC = 0.1;
|
|
40
|
+
|
|
41
|
+
// Lays narrated scenes end to end so that the silence between one scene's last word and the next scene's first word is `gapSec`. Each scene's
|
|
42
|
+
// audio starts on its first frame, so a scene lasts until its last word has ended plus the gap, less the silence its successor opens with. The
|
|
43
|
+
// last scene keeps `tailSec` after its last word. Returns the natural lengths, before any snapping to the beat.
|
|
44
|
+
export function layoutNarrated(scenes: SpokenScene[], gapSec: number, fps = FPS, tailSec = LAST_TAIL_SEC) {
|
|
45
|
+
let cursor = 0;
|
|
46
|
+
const out = scenes.map((v, i) => {
|
|
47
|
+
const end = lastWordEndSec(v);
|
|
48
|
+
const next = scenes[i + 1];
|
|
49
|
+
const seconds = next ? Math.max(end + gapSec - firstWordStartSec(next), end + MIN_AFTER_WORD_SEC) : end + tailSec;
|
|
50
|
+
const durationFrames = secondsToFrames(seconds, fps);
|
|
51
|
+
const scene = { startFrame: cursor, durationFrames };
|
|
52
|
+
cursor += durationFrames;
|
|
53
|
+
return scene;
|
|
54
|
+
});
|
|
55
|
+
return { scenes: out, totalFrames: cursor };
|
|
56
|
+
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import { FILES, type Project } from "./project";
|
|
2
|
+
|
|
3
|
+
// The film's ground, when one was chosen: a video or a picture for the whole film, and pictures for single scenes (a video is only ever for the film).
|
|
4
|
+
// Stored in assets/background.json and shown to the composition as manifest.background and manifest.scenes[i].background.
|
|
5
|
+
export type BackgroundEntry = { key: string; id?: string; title?: string; durationSec?: number };
|
|
6
|
+
export type BackgroundFile = { film?: BackgroundEntry; scenes: Record<string, BackgroundEntry> };
|
|
7
|
+
|
|
8
|
+
export const readBackground = (project: Project): BackgroundFile => {
|
|
9
|
+
const raw = project.readJsonOr<Partial<BackgroundFile>>(FILES.background, {});
|
|
10
|
+
return { ...(raw.film ? { film: raw.film } : {}), scenes: raw.scenes ?? {} };
|
|
11
|
+
};
|
|
12
|
+
|
|
13
|
+
// One ground for the film: setting another replaces it. A scene's picture is set the same way, by scene id.
|
|
14
|
+
export function setBackground(project: Project, entry: BackgroundEntry, sceneId?: string): BackgroundFile {
|
|
15
|
+
const current = readBackground(project);
|
|
16
|
+
const next: BackgroundFile = sceneId ? { ...current, scenes: { ...current.scenes, [sceneId]: entry } } : { ...current, film: entry };
|
|
17
|
+
project.writeJson(FILES.background, next);
|
|
18
|
+
return next;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export const VIDEO_EXT = /\.(mp4|mov|webm|m4v)$/i;
|
|
22
|
+
export const isVideoPath = (path: string): boolean => VIDEO_EXT.test(path);
|
|
23
|
+
|
|
24
|
+
// What is added to a prompt for a video or picture meant to sit behind everything, so that it stays calm and leaves room for type.
|
|
25
|
+
export const BACKGROUND_CLIP_SENTENCE = "An abstract, slow, seamless ambient background with no people, no objects in focus, no text and no logos, soft and low in contrast so text stays readable on top.";
|
|
26
|
+
export const BACKGROUND_IMAGE_SENTENCE = "An abstract, soft, low-contrast background with no people, no text and no logos, with empty space for type on top.";
|
|
27
|
+
export const DEFAULT_BACKGROUND_LOOK = "soft, slowly shifting colour fields in a muted dark palette";
|
|
28
|
+
|
|
29
|
+
// A look for a background that was asked for with no prompt: the colours the first scene's notes name, when they name any, else a neutral default.
|
|
30
|
+
export function lookFromNotes(notes: string | undefined): string {
|
|
31
|
+
const named = [...new Set((notes ?? "").toLowerCase().match(/\b(?:black|white|navy|blue|teal|cyan|green|lime|yellow|amber|orange|red|crimson|pink|magenta|purple|violet|indigo|brown|beige|cream|grey|gray|gold|silver)\b|#[0-9a-f]{6}\b/g) ?? [])].slice(0, 4);
|
|
32
|
+
return named.length ? `soft, slowly shifting colour fields in ${named.join(", ")}` : DEFAULT_BACKGROUND_LOOK;
|
|
33
|
+
}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import { execFile } from "node:child_process";
|
|
2
|
+
import { mkdtempSync, rmSync } from "node:fs";
|
|
3
|
+
import { tmpdir } from "node:os";
|
|
4
|
+
import { join } from "node:path";
|
|
5
|
+
import { promisify } from "node:util";
|
|
6
|
+
import { coverageOf, subjectBoxOf, textZoneOf, type ImageLayersRecord, type Rect, type TextZone } from "../remotion/kit/image-layers-math";
|
|
7
|
+
|
|
8
|
+
const run = promisify(execFile);
|
|
9
|
+
const GRID = 48;
|
|
10
|
+
|
|
11
|
+
// The picture's size is capped, so that the 1-second video a cutout is made from stays small.
|
|
12
|
+
const MAX_SIDE = 1920;
|
|
13
|
+
|
|
14
|
+
async function raw(args: string[]): Promise<Buffer> {
|
|
15
|
+
try {
|
|
16
|
+
const { stdout } = await run("ffmpeg", ["-v", "error", ...args], { encoding: "buffer", maxBuffer: 1 << 24 });
|
|
17
|
+
return stdout;
|
|
18
|
+
} catch (e) {
|
|
19
|
+
if ((e as NodeJS.ErrnoException)?.code === "ENOENT") throw new Error("ffmpeg was not found. Install ffmpeg (macOS: `brew install ffmpeg`) and run the command again.");
|
|
20
|
+
throw new Error(`ffmpeg could not read the picture: ${e instanceof Error ? e.message.split("\n")[0] : String(e)}`);
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
// A picture as a GRID by GRID grid of grey values.
|
|
25
|
+
export async function greyGrid(image: string): Promise<Uint8Array> {
|
|
26
|
+
return new Uint8Array(await raw(["-i", image, "-frames:v", "1", "-vf", `scale=${GRID}:${GRID}:flags=area,format=gray`, "-f", "rawvideo", "-"]));
|
|
27
|
+
}
|
|
28
|
+
// The alpha channel of a picture as a GRID by GRID grid (255 opaque), 255 everywhere when it has none.
|
|
29
|
+
export async function alphaGrid(image: string): Promise<Uint8Array> {
|
|
30
|
+
return new Uint8Array(await raw(["-i", image, "-frames:v", "1", "-vf", `format=rgba,alphaextract,scale=${GRID}:${GRID}:flags=area,format=gray`, "-f", "rawvideo", "-"]));
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
// Where words can sit in this picture: the calmest region, how light it is and how busy. `subject` is the subject's box when there is one.
|
|
34
|
+
export async function measureTextZone(image: string, subject?: Rect): Promise<TextZone> {
|
|
35
|
+
return textZoneOf(await greyGrid(image), GRID, GRID, subject);
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
// The one-second video a still is sent to the cutout service as: 2 frames a second, even sides, never over 1920 on the long side.
|
|
39
|
+
export async function stillToVideo(image: string, out: string): Promise<void> {
|
|
40
|
+
const vf = `scale='min(iw,${MAX_SIDE})':'min(ih,${MAX_SIDE})':force_original_aspect_ratio=decrease,scale=trunc(iw/2)*2:trunc(ih/2)*2,format=yuv420p`;
|
|
41
|
+
await raw(["-y", "-loop", "1", "-framerate", "2", "-i", image, "-t", "1", "-vf", vf, "-c:v", "libx264", "-preset", "veryfast", "-crf", "20", "-an", out]);
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
// One frame of the transparent video the service returns, as a PNG with alpha: VP9 with alpha needs the libvpx decoder, so it is asked for by name.
|
|
45
|
+
// Throws when the PNG has no transparent pixel at all (the alpha did not survive), and returns what the subject covers and the box that holds it.
|
|
46
|
+
export async function frameWithAlpha(webm: string, png: string): Promise<{ coverage: number; box?: Rect }> {
|
|
47
|
+
await raw(["-y", "-c:v", "libvpx-vp9", "-i", webm, "-frames:v", "1", "-pix_fmt", "rgba", png]);
|
|
48
|
+
const a = await alphaGrid(png);
|
|
49
|
+
if (!a.some((v) => v < 250)) throw new Error("The cut-out came back without any transparent pixel, so there is no subject layer to make. The picture stays flat.");
|
|
50
|
+
return { coverage: coverageOf(a), box: subjectBoxOf(a, GRID, GRID) };
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
export const tempDir = (): { dir: string; done: () => void } => {
|
|
54
|
+
const dir = mkdtempSync(join(tmpdir(), "rk-layers-"));
|
|
55
|
+
return { dir, done: () => rmSync(dir, { recursive: true, force: true }) };
|
|
56
|
+
};
|
|
57
|
+
|
|
58
|
+
export type LayerEntry = { subject?: string; subjectBox?: Rect; textZone: TextZone; noSubject?: boolean; coverage?: number; cutoutId?: string };
|
|
59
|
+
// The record the manifest carries for a picture: its key as `back`, the subject's file and box when there is one, and where text can sit.
|
|
60
|
+
export const toImageLayers = (key: string, e: LayerEntry): ImageLayersRecord => ({ back: key, ...(e.subject && !e.noSubject ? { subject: e.subject, ...(e.subjectBox ? { subjectBox: e.subjectBox } : {}) } : {}), textZone: e.textZone });
|