reelkit-cli 0.3.1 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/package.json +1 -1
- package/skill/SKILL.md +11 -2
- package/skill/reference/beat-sync.md +45 -0
- package/skill/reference/captions.md +18 -2
- package/skill/reference/clips.md +1 -0
- package/skill/reference/continuity.md +99 -0
- package/skill/reference/kit.md +18 -1
- package/skill/reference/motion-design.md +9 -0
- package/skill/reference/references.md +43 -0
- package/skill/reference/remotion-composition.md +2 -1
- package/skill/reference/scene-treatments.md +1 -1
- package/skill/reference/scriptwriting.md +39 -5
- package/skill/reference/sound-design.md +11 -0
- package/skill/reference/styles.md +9 -0
- package/src/cli.ts +14 -2
- package/src/commands/assets.ts +38 -8
- package/src/commands/auth.ts +1 -1
- package/src/commands/build.ts +73 -8
- package/src/commands/plan.ts +26 -4
- package/src/commands/ref.ts +315 -0
- package/src/contract/index.ts +25 -1
- package/src/pipeline/beatsnap.ts +52 -0
- package/src/pipeline/review.ts +28 -1
- package/src/pipeline/schema.ts +14 -0
- package/src/project/beats.ts +130 -0
- package/src/project/chromakey.ts +14 -4
- package/src/project/manifest.ts +22 -1
- package/src/project/music.ts +25 -0
- package/src/project/project.ts +1 -1
- package/src/project/refmeasure.ts +192 -0
- package/src/remotion/Root.tsx +4 -2
- package/src/remotion/kit/Camera.tsx +22 -0
- package/src/remotion/kit/Captions.tsx +25 -8
- package/src/remotion/kit/Carry.tsx +38 -0
- package/src/remotion/kit/Music.tsx +19 -0
- package/src/remotion/kit/beat.ts +23 -0
- package/src/remotion/kit/caption-groups.ts +65 -0
- package/src/remotion/kit/docs.ts +18 -1
- package/src/remotion/kit/index.ts +7 -0
- package/src/remotion/kit/media.ts +15 -0
- package/src/remotion/kit/motion-math.ts +113 -0
- package/src/remotion/kit/music-math.ts +42 -0
- package/src/render/continuity.ts +87 -0
- package/src/render/validate.ts +27 -0
- package/src/testing/conformance.ts +107 -1
- package/src/testing/fake-api.ts +44 -6
package/src/contract/index.ts
CHANGED
|
@@ -67,6 +67,15 @@
|
|
|
67
67
|
// 20 seconds is refused by the request schema.
|
|
68
68
|
// - When the cutout quota is used up, `cutoutRun` is 429 `quota_exceeded` with the reset date. When the server has no GPU service configured,
|
|
69
69
|
// `cutoutStart` and `cutoutRun` are 400 `invalid_request` with the message "Background removal is not available on this server yet."
|
|
70
|
+
// - A transcript of a reference video's audio is a job with two calls. `transcribeStart` checks the request and answers with an id and an
|
|
71
|
+
// `uploadUrl` (charging nothing; the client sends the audio there exactly as for a library upload); `transcribeRun` checks that the audio
|
|
72
|
+
// arrived (else 400 `invalid_request`), reserves `ceil(durationSec)` seconds from the caller's monthly transcription quota, transcribes,
|
|
73
|
+
// deletes the audio and answers with the transcript. Only the audio is sent, never the video. The audio is deleted from the server when the
|
|
74
|
+
// call ends, whatever the result. The transcript is kept with the job, so calling `transcribeRun` again for the same id returns the same
|
|
75
|
+
// transcript and does not charge again. A failed run is not charged: the seconds are given back. It is charged in whole seconds of audio
|
|
76
|
+
// (`ceil(durationSec)`), and audio longer than 600 seconds is refused by the request schema. Segment times are in seconds from the start
|
|
77
|
+
// of the audio. A transcribe id belongs to its caller: another user's id, or an unknown one, is `not_found`.
|
|
78
|
+
// - When the transcription quota is used up, `transcribeRun` is 429 `quota_exceeded` with the reset date.
|
|
70
79
|
// - `durationSec` is the decoded audio length; `words` are in seconds from the start of that audio.
|
|
71
80
|
// - Search returns published items only, and leaves out items that are not matches at all; the default `limit` is 8.
|
|
72
81
|
import { z } from "zod";
|
|
@@ -101,6 +110,12 @@ export const CutoutTypeSchema = z.enum(["video/mp4", "video/quicktime", "video/w
|
|
|
101
110
|
export type CutoutType = z.infer<typeof CutoutTypeSchema>;
|
|
102
111
|
// The longest video a cutout takes, in seconds.
|
|
103
112
|
export const MAX_CUTOUT_SECONDS = 20;
|
|
113
|
+
// The audio types a transcription takes, the most bytes it takes (25 MB) and the longest audio in seconds.
|
|
114
|
+
export const TranscribeTypeSchema = z.enum(["audio/mpeg", "audio/mp4", "audio/wav"]);
|
|
115
|
+
export type TranscribeType = z.infer<typeof TranscribeTypeSchema>;
|
|
116
|
+
export const MAX_TRANSCRIBE_BYTES = 26_214_400;
|
|
117
|
+
export const MAX_TRANSCRIBE_SECONDS = 600;
|
|
118
|
+
export const TranscriptSegmentSchema = z.object({ text: z.string(), startSec: z.number(), endSec: z.number() });
|
|
104
119
|
|
|
105
120
|
export const ErrorCodeSchema = z.enum(["unauthenticated", "quota_exceeded", "not_found", "invalid_request", "server_error"]);
|
|
106
121
|
export type ErrorCode = z.infer<typeof ErrorCodeSchema>;
|
|
@@ -136,7 +151,7 @@ export type Voice = z.infer<typeof VoiceSchema>;
|
|
|
136
151
|
const Meter = z.object({ used: z.number(), limit: z.number() });
|
|
137
152
|
export const MeSchema = z.object({
|
|
138
153
|
userId: z.string(), handle: z.string(),
|
|
139
|
-
quota: z.object({ voiceoverChars: Meter, images: Meter, clips: Meter, cutoutSeconds: Meter, resetsAt: z.string() }),
|
|
154
|
+
quota: z.object({ voiceoverChars: Meter, images: Meter, clips: Meter, cutoutSeconds: Meter, transcribeSeconds: Meter, resetsAt: z.string() }),
|
|
140
155
|
// How many of the user's own items are published in the shared library. An item still in review, or sent back, is not counted.
|
|
141
156
|
contributions: z.number(),
|
|
142
157
|
});
|
|
@@ -207,6 +222,15 @@ export const routes = {
|
|
|
207
222
|
url: z.string().optional(), ext: z.enum(["webm"]).optional(), contentType: z.literal("video/webm").optional(),
|
|
208
223
|
}).refine((r) => (r.status === "done") === (r.url !== undefined && r.ext !== undefined && r.contentType !== undefined) && (r.status === "failed") === (r.message !== undefined),
|
|
209
224
|
"url, ext and contentType belong to done and message to failed")),
|
|
225
|
+
transcribeStart: route("POST", "/transcripts", true,
|
|
226
|
+
z.object({
|
|
227
|
+
filename: text(200), contentType: TranscribeTypeSchema,
|
|
228
|
+
bytes: z.number().int().min(1).max(MAX_TRANSCRIBE_BYTES), durationSec: z.number().positive().max(MAX_TRANSCRIBE_SECONDS),
|
|
229
|
+
}),
|
|
230
|
+
z.object({ id: z.string(), uploadUrl: z.string() })),
|
|
231
|
+
transcribeRun: route("POST", "/transcripts/run", true,
|
|
232
|
+
z.object({ id: ItemId }),
|
|
233
|
+
z.object({ id: z.string(), text: z.string(), language: z.string().optional(), segments: z.array(TranscriptSegmentSchema), durationSec: z.number().positive() })),
|
|
210
234
|
publicLibrary: route("GET", "/public/library", false,
|
|
211
235
|
z.object({ q: text(200).optional(), kind: LibraryKindSchema.optional(), page: z.coerce.number().int().min(1).max(10000).optional() }),
|
|
212
236
|
// With `q` the items are the best matches by meaning, best first, each with `match` (0..1), in one page (`hasMore` false). Under heavy
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
// Puts scene changes on the beat. A scene is held a little longer at its end so that the next one starts exactly on a beat.
|
|
2
|
+
// The voiceover and its word times never move: only a scene's length, and so where the later scenes start, changes.
|
|
3
|
+
|
|
4
|
+
// The last scene must keep at least this long after its last word, as the natural padding already does (0.4 s) plus a little.
|
|
5
|
+
export const MIN_TAIL_SEC = 0.5;
|
|
6
|
+
|
|
7
|
+
export type SnapInput = {
|
|
8
|
+
// Each scene's natural length in frames.
|
|
9
|
+
naturalFrames: number[];
|
|
10
|
+
// When the last word of the last scene ends, in frames from the start of that scene.
|
|
11
|
+
lastWordEndFrame: number;
|
|
12
|
+
// Beats as frame numbers, ascending. They should reach past the end of the video.
|
|
13
|
+
beatFrames: number[];
|
|
14
|
+
fps: number;
|
|
15
|
+
};
|
|
16
|
+
|
|
17
|
+
const firstAtOrAfter = (beats: number[], frame: number): number | undefined => beats.find((b) => b >= frame);
|
|
18
|
+
|
|
19
|
+
// The new length of every scene. A scene is only ever made longer, by less than one beat (the last by a few frames more, to keep its tail).
|
|
20
|
+
// A boundary that has no beat close enough keeps its natural place, and the scenes after it follow from there.
|
|
21
|
+
export function snapToBeats(input: SnapInput): number[] {
|
|
22
|
+
const { naturalFrames, beatFrames, fps } = input;
|
|
23
|
+
if (beatFrames.length < 2) return [...naturalFrames];
|
|
24
|
+
const gaps = beatFrames.slice(1).map((b, i) => b - beatFrames[i]!).sort((a, b) => a - b);
|
|
25
|
+
const period = Math.max(1, gaps[Math.floor(gaps.length / 2)]!);
|
|
26
|
+
let cursor = 0;
|
|
27
|
+
return naturalFrames.map((natural, i) => {
|
|
28
|
+
const last = i === naturalFrames.length - 1;
|
|
29
|
+
let end = cursor + natural;
|
|
30
|
+
let limit = period;
|
|
31
|
+
if (last) {
|
|
32
|
+
const tail = cursor + input.lastWordEndFrame + Math.ceil(MIN_TAIL_SEC * fps);
|
|
33
|
+
if (tail > end) { limit += tail - end; end = tail; }
|
|
34
|
+
}
|
|
35
|
+
const beat = firstAtOrAfter(beatFrames, end);
|
|
36
|
+
const target = beat !== undefined && beat - (cursor + natural) <= limit ? beat : end;
|
|
37
|
+
const frames = target - cursor;
|
|
38
|
+
cursor = target;
|
|
39
|
+
return frames;
|
|
40
|
+
});
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
// Beats in seconds, continued at the tempo's own spacing until `untilSec`, for a video longer than the track (the track loops).
|
|
44
|
+
export function extendBeats(beats: number[], bpm: number, untilSec: number): number[] {
|
|
45
|
+
if (!beats.length) return [];
|
|
46
|
+
const period = 60 / bpm;
|
|
47
|
+
const out = [...beats];
|
|
48
|
+
while (out[out.length - 1]! < untilSec) out.push(Math.round((out[out.length - 1]! + period) * 1000) / 1000);
|
|
49
|
+
return out;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
export const toBeatFrames = (beatsSec: number[], fps: number): number[] => [...new Set(beatsSec.map((t) => Math.round(t * fps)))].sort((a, b) => a - b);
|
package/src/pipeline/review.ts
CHANGED
|
@@ -8,6 +8,30 @@ export function estimateLength(plan: ScenePlan): { words: number; seconds: numbe
|
|
|
8
8
|
return { words: total, seconds: total / WORDS_PER_SEC };
|
|
9
9
|
}
|
|
10
10
|
|
|
11
|
+
// Each scene's estimated length in seconds, from its narration at the same pace as the whole-video estimate above.
|
|
12
|
+
export function sceneSeconds(plan: ScenePlan): { id: string; seconds: number }[] {
|
|
13
|
+
return plan.scenes.map((s) => ({ id: s.id, seconds: words(s.narration) / WORDS_PER_SEC }));
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
// How uneven the scene lengths are: the standard deviation over the mean (0 when they are all the same).
|
|
17
|
+
export function lengthVariation(seconds: number[]): number {
|
|
18
|
+
const mean = seconds.reduce((a, b) => a + b, 0) / (seconds.length || 1);
|
|
19
|
+
if (!(mean > 0)) return 0;
|
|
20
|
+
return Math.sqrt(seconds.reduce((a, b) => a + (b - mean) ** 2, 0) / seconds.length) / mean;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
// A video whose scenes all last about as long feels like a slideshow. With four scenes or more, the lengths should differ a lot.
|
|
24
|
+
export function rhythmNote(plan: ScenePlan): string | undefined {
|
|
25
|
+
if (plan.scenes.length < 4) return undefined;
|
|
26
|
+
const lengths = sceneSeconds(plan);
|
|
27
|
+
const values = lengths.map((l) => l.seconds);
|
|
28
|
+
const mean = values.reduce((a, b) => a + b, 0) / values.length;
|
|
29
|
+
const cv = lengthVariation(values);
|
|
30
|
+
if (cv >= 0.25 && values.some((v) => v < mean / 2)) return undefined;
|
|
31
|
+
const shortest = lengths.reduce((a, b) => (b.seconds < a.seconds ? b : a)), longest = lengths.reduce((a, b) => (b.seconds > a.seconds ? b : a));
|
|
32
|
+
return `The scenes are too even in length (the shortest, ${shortest.id}, is about ${shortest.seconds.toFixed(1)}s and the longest, ${longest.id}, about ${longest.seconds.toFixed(1)}s). Make one or two scenes much shorter (a hit of a few words) and let one run long.`;
|
|
33
|
+
}
|
|
34
|
+
|
|
11
35
|
// Soft quality checks on a plan that already passes the hard rules in validatePlan.
|
|
12
36
|
// These are things worth improving in the script; they never fail a video on their own.
|
|
13
37
|
export function reviewPlan(plan: ScenePlan, ctx: { footage?: AssetRecord }): string[] {
|
|
@@ -26,10 +50,13 @@ export function reviewPlan(plan: ScenePlan, ctx: { footage?: AssetRecord }): str
|
|
|
26
50
|
|
|
27
51
|
for (const s of plan.scenes) {
|
|
28
52
|
const n = words(s.narration);
|
|
29
|
-
|
|
53
|
+
// A hit of a few words is wanted in the rhythm of a video, so only a scene with almost nothing to say is flagged.
|
|
54
|
+
if (!ctx.footage && n < 3) issues.push(`Scene ${s.id} has only ${n} word${n === 1 ? "" : "s"} of narration; give it at least a short phrase.`);
|
|
30
55
|
if (n > 45) issues.push(`Scene ${s.id} has ${n} words of narration; split it or cut it to under 45.`);
|
|
31
56
|
if (s.onScreenText.length > 3) issues.push(`Scene ${s.id} has ${s.onScreenText.length} on-screen text items; use at most 3.`);
|
|
32
57
|
for (const t of s.onScreenText) if (words(t) > 7) issues.push(`Scene ${s.id}: on-screen text "${t}" is too long; keep each item to 7 words or fewer.`);
|
|
33
58
|
}
|
|
59
|
+
const rhythm = rhythmNote(plan);
|
|
60
|
+
if (rhythm) issues.push(rhythm);
|
|
34
61
|
return issues;
|
|
35
62
|
}
|
package/src/pipeline/schema.ts
CHANGED
|
@@ -28,6 +28,14 @@ export const ScenePlanSchema = z.object({
|
|
|
28
28
|
// Optional only so plans stored before voices existed still load; a new plan must choose one.
|
|
29
29
|
voiceId: z.string().optional().describe("id of the narration voice, one of the ids from `reelkit assets voices`, chosen to suit the idea, audience and language"),
|
|
30
30
|
pace: z.enum(["slow", "normal", "fast"]).optional().describe("speaking pace: slow for calm or emotional, normal by default, fast for high-energy"),
|
|
31
|
+
// Whether the video has captions and of which kind: none, one word at a time, or a phrase at a time. Absent means "phrase".
|
|
32
|
+
captions: z.enum(["none", "word", "phrase"]).optional().describe("captions: none, one word at a time (word), or a phrase at a time (phrase, the default), as the user chose"),
|
|
33
|
+
// A video the new one is made "like": what is taken from it is how it feels (structure, pacing, motion), never its footage, music or words.
|
|
34
|
+
// Optional, so plans stored before references existed still load.
|
|
35
|
+
reference: z.object({
|
|
36
|
+
id: z.string().describe("the id of a reference in this project, from `reelkit ref list`"),
|
|
37
|
+
take: z.array(z.string()).min(1).max(6).describe("short notes on what is taken from it, e.g. \"fast cuts every ~1.2s\", \"big type on colour fields\""),
|
|
38
|
+
}).optional(),
|
|
31
39
|
scenes: z.array(SceneSchema).min(3).max(8),
|
|
32
40
|
});
|
|
33
41
|
|
|
@@ -77,6 +85,12 @@ export const AssetManifestSchema = z.object({
|
|
|
77
85
|
height: z.number(),
|
|
78
86
|
totalFrames: z.number(),
|
|
79
87
|
footageKey: z.string().optional(),
|
|
88
|
+
// The video's music track, when one was pulled with `reelkit assets pull <id> --music`. beatFrames are the beats that fall inside the video.
|
|
89
|
+
music: z.object({ key: z.string(), bpm: z.number().optional(), beatFrames: z.array(z.number()) }).optional(),
|
|
90
|
+
// The plan's caption choice, with the default filled in, so the composition can pass it to <Captions group={...}>.
|
|
91
|
+
captions: z.enum(["none", "word", "phrase"]).optional(),
|
|
92
|
+
// Linear gain per sound file path that brings every pulled sound to a common level; the kit's Sfx and Music apply it.
|
|
93
|
+
soundGain: z.record(z.string(), z.number()).optional(),
|
|
80
94
|
scenes: z.array(ManifestSceneSchema),
|
|
81
95
|
});
|
|
82
96
|
export type AssetManifest = z.infer<typeof AssetManifestSchema>;
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
// Finds the tempo and the beat times of a piece of audio, on this machine, with no libraries. The audio is given as mono samples.
|
|
2
|
+
//
|
|
3
|
+
// The idea: loud moments (a drum hit, a click, a word) make the energy of the sound jump. The list of those jumps over time is the onset curve.
|
|
4
|
+
// If the music has a pulse, the curve repeats itself after one beat, so the tempo is the delay at which the curve best matches a copy of itself.
|
|
5
|
+
|
|
6
|
+
export type BeatAnalysis = {
|
|
7
|
+
// How sure the tempo is, from 0 (nothing repeats) to 1 (a perfect pulse). Real music is usually 0.3 to 0.7.
|
|
8
|
+
confidence: number;
|
|
9
|
+
// Present only when the pulse is clear enough to trust; a made-up tempo is worse than none.
|
|
10
|
+
tempoBpm?: number;
|
|
11
|
+
// Beat times in seconds from the start of the samples, on the grid of `tempoBpm`. Present with `tempoBpm`.
|
|
12
|
+
beats?: number[];
|
|
13
|
+
};
|
|
14
|
+
|
|
15
|
+
// Below this the autocorrelation peak is what noise produces by chance, so no tempo is reported.
|
|
16
|
+
export const MIN_BEAT_CONFIDENCE = 0.2;
|
|
17
|
+
|
|
18
|
+
const WINDOW = 512;
|
|
19
|
+
const HOP = 256;
|
|
20
|
+
const MIN_BPM = 60;
|
|
21
|
+
const MAX_BPM = 200;
|
|
22
|
+
// The range people actually move to; when a tempo outside it scores nearly as well, it is the half or double of one inside it.
|
|
23
|
+
const PREFERRED_LOW = 80;
|
|
24
|
+
const PREFERRED_HIGH = 160;
|
|
25
|
+
const BPM_STEP = 0.5;
|
|
26
|
+
// How many multiples of the beat length are compared: a real pulse matches at two, three and four beats too, noise does not.
|
|
27
|
+
const HARMONICS = 4;
|
|
28
|
+
|
|
29
|
+
// Energy per window, as a curve that jumps up where the sound gets louder. Quiet onsets count as much as loud ones (log scale), and the
|
|
30
|
+
// curve is smoothed a little so a hit that falls between two windows still shows up in both.
|
|
31
|
+
function onsetCurve(samples: Float32Array): Float64Array {
|
|
32
|
+
const frames = Math.max(0, Math.floor((samples.length - WINDOW) / HOP) + 1);
|
|
33
|
+
const energy = new Float64Array(frames);
|
|
34
|
+
for (let f = 0; f < frames; f++) {
|
|
35
|
+
let sum = 0;
|
|
36
|
+
const start = f * HOP;
|
|
37
|
+
for (let i = 0; i < WINDOW; i++) { const s = samples[start + i]!; sum += s * s; }
|
|
38
|
+
energy[f] = Math.log1p(100 * Math.sqrt(sum / WINDOW));
|
|
39
|
+
}
|
|
40
|
+
const rise = new Float64Array(frames);
|
|
41
|
+
for (let f = 1; f < frames; f++) rise[f] = Math.max(0, energy[f]! - energy[f - 1]!);
|
|
42
|
+
const smooth = new Float64Array(frames);
|
|
43
|
+
for (let f = 0; f < frames; f++) smooth[f] = 0.25 * (rise[f - 1] ?? 0) + 0.5 * rise[f]! + 0.25 * (rise[f + 1] ?? 0);
|
|
44
|
+
return smooth;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
// The value of the curve at a position between two windows.
|
|
48
|
+
function at(curve: Float64Array, pos: number): number {
|
|
49
|
+
if (pos < 0 || pos > curve.length - 1) return 0;
|
|
50
|
+
const i = Math.floor(pos), frac = pos - i;
|
|
51
|
+
return curve[i]! * (1 - frac) + (curve[i + 1] ?? curve[i]!) * frac;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
// How much the curve looks like itself `lag` windows later, from 0 to about 1.
|
|
55
|
+
function selfSimilarity(curve: Float64Array, lag: number): number {
|
|
56
|
+
const n = Math.floor(curve.length - lag);
|
|
57
|
+
if (n < 8) return 0;
|
|
58
|
+
let cross = 0, own = 0, shifted = 0;
|
|
59
|
+
for (let i = 0; i < n; i++) {
|
|
60
|
+
const a = curve[i]!, b = at(curve, i + lag);
|
|
61
|
+
cross += a * b; own += a * a; shifted += b * b;
|
|
62
|
+
}
|
|
63
|
+
return own > 0 && shifted > 0 ? cross / Math.sqrt(own * shifted) : 0;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
export function analyzeBeats(samples: Float32Array, sampleRate: number): BeatAnalysis {
|
|
67
|
+
const curve = onsetCurve(samples);
|
|
68
|
+
if (curve.length < 16) return { confidence: 0 };
|
|
69
|
+
const mean = curve.reduce((s, v) => s + v, 0) / curve.length;
|
|
70
|
+
// The mean is taken out so that a steady loudness (which has no beat) does not look like a match.
|
|
71
|
+
for (let i = 0; i < curve.length; i++) curve[i] = curve[i]! - mean;
|
|
72
|
+
let energy = 0;
|
|
73
|
+
for (const v of curve) energy += v * v;
|
|
74
|
+
if (energy < 1e-9) return { confidence: 0 };
|
|
75
|
+
|
|
76
|
+
const framesPerSec = sampleRate / HOP;
|
|
77
|
+
const candidates: { bpm: number; score: number }[] = [];
|
|
78
|
+
for (let bpm = MIN_BPM; bpm <= MAX_BPM + 1e-9; bpm += BPM_STEP) {
|
|
79
|
+
const period = (60 / bpm) * framesPerSec;
|
|
80
|
+
let total = 0, used = 0;
|
|
81
|
+
for (let m = 1; m <= HARMONICS; m++) {
|
|
82
|
+
if (m * period > curve.length / 2) break;
|
|
83
|
+
total += selfSimilarity(curve, m * period);
|
|
84
|
+
used++;
|
|
85
|
+
}
|
|
86
|
+
candidates.push({ bpm, score: used ? total / used : 0 });
|
|
87
|
+
}
|
|
88
|
+
const isPeak = (i: number) => candidates[i]!.score >= (candidates[i - 1]?.score ?? -Infinity) && candidates[i]!.score >= (candidates[i + 1]?.score ?? -Infinity);
|
|
89
|
+
let best = 0;
|
|
90
|
+
for (let i = 1; i < candidates.length; i++) if (candidates[i]!.score > candidates[best]!.score) best = i;
|
|
91
|
+
const inPreferred = (i: number) => candidates[i]!.bpm >= PREFERRED_LOW && candidates[i]!.bpm <= PREFERRED_HIGH;
|
|
92
|
+
if (!inPreferred(best)) {
|
|
93
|
+
let alt = -1;
|
|
94
|
+
for (let i = 0; i < candidates.length; i++) if (inPreferred(i) && isPeak(i) && (alt < 0 || candidates[i]!.score > candidates[alt]!.score)) alt = i;
|
|
95
|
+
if (alt >= 0 && candidates[alt]!.score >= 0.8 * candidates[best]!.score) best = alt;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
const confidence = Math.max(0, Math.min(1, candidates[best]!.score));
|
|
99
|
+
if (confidence < MIN_BEAT_CONFIDENCE) return { confidence: round(confidence, 2) };
|
|
100
|
+
|
|
101
|
+
// The peak lies between two of the tested tempos: a parabola through the three scores finds where it really is.
|
|
102
|
+
let bpm = candidates[best]!.bpm;
|
|
103
|
+
const l = candidates[best - 1]?.score, c = candidates[best]!.score, r = candidates[best + 1]?.score;
|
|
104
|
+
if (l !== undefined && r !== undefined && l - 2 * c + r < 0) bpm += (BPM_STEP * 0.5 * (l - r)) / (l - 2 * c + r);
|
|
105
|
+
|
|
106
|
+
// Where the beats fall: the shift of the grid that lands on the most onset.
|
|
107
|
+
const periodSec = 60 / bpm;
|
|
108
|
+
// Frame f describes the window that starts at f * HOP; a rise there means the sound arrived in the later part of that window.
|
|
109
|
+
const posOf = (t: number) => (t * sampleRate - 0.75 * WINDOW) / HOP;
|
|
110
|
+
const duration = samples.length / sampleRate;
|
|
111
|
+
let bestPhase = 0, bestHit = -Infinity;
|
|
112
|
+
for (let phase = 0; phase < periodSec; phase += HOP / 4 / sampleRate) {
|
|
113
|
+
let hit = 0;
|
|
114
|
+
for (let t = phase; t < duration; t += periodSec) hit += at(curve, posOf(t));
|
|
115
|
+
if (hit > bestHit) { bestHit = hit; bestPhase = phase; }
|
|
116
|
+
}
|
|
117
|
+
const beats: number[] = [];
|
|
118
|
+
// A beat that lands just before the first sample (a track that starts on a hit) is the first beat, at 0.
|
|
119
|
+
for (let t = bestPhase - periodSec; t < duration; t += periodSec) if (t >= -0.03) beats.push(round(Math.max(0, t), 3));
|
|
120
|
+
return { confidence: round(confidence, 2), tempoBpm: round(bpm, 1), beats };
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
// The share of cuts (0..1) that fall within `toleranceSec` of a beat.
|
|
124
|
+
export function cutsOnBeat(cuts: number[], beats: number[], toleranceSec = 0.08): number {
|
|
125
|
+
if (!cuts.length || !beats.length) return 0;
|
|
126
|
+
const near = cuts.filter((c) => beats.some((b) => Math.abs(b - c) <= toleranceSec)).length;
|
|
127
|
+
return near / cuts.length;
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
const round = (n: number, digits: number) => Math.round(n * 10 ** digits) / 10 ** digits;
|
package/src/project/chromakey.ts
CHANGED
|
@@ -12,12 +12,19 @@ export type KeyOptions = {
|
|
|
12
12
|
requireGreen?: boolean;
|
|
13
13
|
};
|
|
14
14
|
|
|
15
|
+
// Thrown when the background is green but not a saturated chroma green, or (for a clip that was asked to be on green) not green at
|
|
16
|
+
// all. Video models rarely produce a true chroma green: they give a soft sage or olive that sits too close to skin and grey to be
|
|
17
|
+
// keyed without taking the subject with it. The caller should cut the subject out on the server instead.
|
|
18
|
+
export class GreenTooDullError extends Error {
|
|
19
|
+
constructor() { super("The green background is too dull to key cleanly: it is not a saturated chroma green. Use --cutout instead, which removes any background."); this.name = "GreenTooDullError"; }
|
|
20
|
+
}
|
|
21
|
+
|
|
15
22
|
const NOT_GREEN = "This video does not have a green background in its corners, so there is nothing to key out. Film the subject in front of an evenly lit green background that fills the frame.";
|
|
16
23
|
|
|
17
24
|
// The colour to key on, read from the first frame's four corners. A real green screen, filmed or generated, is never exactly
|
|
18
25
|
// #00FF00: it is whatever green the light and the encoder made of it, and keying on the wrong green leaves a haze. Undefined when
|
|
19
26
|
// the corners are not green.
|
|
20
|
-
async function cornerGreen(input: string): Promise<string | undefined> {
|
|
27
|
+
async function cornerGreen(input: string): Promise<{ colour: string; saturated: boolean } | undefined> {
|
|
21
28
|
const corner = (x: string, y: string) => `crop=24:24:${x}:${y},scale=1:1:flags=area`;
|
|
22
29
|
const picks = ["0:0", "iw-24:0", "0:ih-24", "iw-24:ih-24"].map((c) => { const [x, y] = c.split(":") as [string, string]; return corner(x, y); });
|
|
23
30
|
const colours: [number, number, number][] = [];
|
|
@@ -29,7 +36,9 @@ async function cornerGreen(input: string): Promise<string | undefined> {
|
|
|
29
36
|
// Most corners must agree: a subject may reach one corner, but not three.
|
|
30
37
|
if (green.length < 3) return undefined;
|
|
31
38
|
const avg = (i: 0 | 1 | 2) => Math.round(green.reduce((sum, c) => sum + c[i], 0) / green.length);
|
|
32
|
-
|
|
39
|
+
const [r, g, b] = [avg(0), avg(1), avg(2)];
|
|
40
|
+
// How far the green stands above the other two channels. Studio and pure greens are far above 60; a sage green is around 30.
|
|
41
|
+
return { colour: `0x${[r, g, b].map((n) => n.toString(16).padStart(2, "0")).join("")}`, saturated: g - Math.max(r, b) >= 60 };
|
|
33
42
|
}
|
|
34
43
|
|
|
35
44
|
// Turns the green of a green-screen clip into transparency: VP9 with an alpha channel, which Remotion plays with `transparent`.
|
|
@@ -39,12 +48,13 @@ export async function keyGreen(input: string, output: string, opts: KeyOptions =
|
|
|
39
48
|
try {
|
|
40
49
|
const found = await cornerGreen(input);
|
|
41
50
|
if (!found && opts.requireGreen) throw new Error(NOT_GREEN);
|
|
42
|
-
|
|
51
|
+
if (!found || !found.saturated) throw new GreenTooDullError();
|
|
52
|
+
const filter = `chromakey=color=${found.colour ?? PURE_GREEN}:similarity=0.16:blend=0.08,despill=type=green:mix=0.5:expand=0,format=yuva420p`;
|
|
43
53
|
const audio = opts.keepAudio ? ["-c:a", "libopus", "-b:a", "96k"] : ["-an"];
|
|
44
54
|
await run("ffmpeg", ["-v", "error", "-y", "-i", input, "-vf", filter, "-c:v", "libvpx-vp9", "-pix_fmt", "yuva420p", "-b:v", "0", "-crf", "30", ...audio, output]);
|
|
45
55
|
} catch (e) {
|
|
46
56
|
rmSync(output, { force: true });
|
|
47
|
-
if (e instanceof Error && e.message === NOT_GREEN) throw e;
|
|
57
|
+
if (e instanceof GreenTooDullError || (e instanceof Error && e.message === NOT_GREEN)) throw e;
|
|
48
58
|
if ((e as NodeJS.ErrnoException)?.code === "ENOENT") throw new Error("ffmpeg was not found. Install ffmpeg (macOS: `brew install ffmpeg`) and run the command again.");
|
|
49
59
|
throw new Error("ffmpeg could not key the green out of the clip. The original clip is saved; run the command again to retry.");
|
|
50
60
|
}
|
package/src/project/manifest.ts
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import { PACE_SPEED, type AssetManifest, type ScenePlan, type WordTiming } from "../pipeline/schema";
|
|
2
|
+
import { extendBeats, snapToBeats, toBeatFrames } from "../pipeline/beatsnap";
|
|
2
3
|
import { dimensionsFor, FPS, layoutScenes } from "../pipeline/timing";
|
|
4
|
+
import type { MusicRecord } from "./music";
|
|
3
5
|
import { FILES, type Project } from "./project";
|
|
4
6
|
|
|
5
7
|
// What a scene was recorded from is stored with it, so a changed script or voice is noticed.
|
|
@@ -24,7 +26,22 @@ export function buildManifest(project: Project, plan: ScenePlan): AssetManifest
|
|
|
24
26
|
const footage = project.footage();
|
|
25
27
|
const byId = new Map(project.assets().map((a) => [a.id, a]));
|
|
26
28
|
|
|
27
|
-
|
|
29
|
+
let layout = layoutScenes(plan.scenes.map((s) => voiceovers[s.id].durationSec));
|
|
30
|
+
const music = project.readJsonOr<MusicRecord | undefined>(FILES.music, undefined);
|
|
31
|
+
let beatFrames: number[] = [];
|
|
32
|
+
if (music) {
|
|
33
|
+
beatFrames = toBeatFrames(extendBeats(music.beats, music.bpm ?? 120, layout.totalFrames / FPS + 10), FPS);
|
|
34
|
+
// Scene changes land on beats only when the track has a trusted tempo. Footage has a fixed length that a longer scene must not overrun.
|
|
35
|
+
if (music.bpm && music.beats.length) {
|
|
36
|
+
const last = voiceovers[plan.scenes[plan.scenes.length - 1].id];
|
|
37
|
+
const lastWords = last.words.length ? Math.max(...last.words.map((w) => w.endSec)) : last.durationSec;
|
|
38
|
+
const lengths = snapToBeats({ naturalFrames: layout.scenes.map((s) => s.durationFrames), lastWordEndFrame: Math.ceil(lastWords * FPS), beatFrames, fps: FPS });
|
|
39
|
+
let cursor = 0;
|
|
40
|
+
const snapped = { scenes: lengths.map((durationFrames) => { const r = { startFrame: cursor, durationFrames }; cursor += durationFrames; return r; }), totalFrames: 0 };
|
|
41
|
+
snapped.totalFrames = cursor;
|
|
42
|
+
if (!footage || cursor <= Math.floor((footage.durationSec ?? 0) * FPS)) layout = snapped;
|
|
43
|
+
}
|
|
44
|
+
}
|
|
28
45
|
let totalFrames = layout.totalFrames;
|
|
29
46
|
if (footage) {
|
|
30
47
|
const footageFrames = Math.floor((footage.durationSec ?? 0) * FPS);
|
|
@@ -38,7 +55,9 @@ export function buildManifest(project: Project, plan: ScenePlan): AssetManifest
|
|
|
38
55
|
fps: FPS,
|
|
39
56
|
...dimensionsFor(plan.aspect, footage),
|
|
40
57
|
totalFrames,
|
|
58
|
+
captions: plan.captions ?? "phrase",
|
|
41
59
|
...(footage ? { footageKey: footage.key } : {}),
|
|
60
|
+
...(music ? { music: { key: music.key, ...(music.bpm ? { bpm: music.bpm } : {}), beatFrames: music.bpm ? beatFrames.filter((f) => f < totalFrames) : [] } } : {}),
|
|
42
61
|
scenes: plan.scenes.map((scene, i) => ({
|
|
43
62
|
id: scene.id,
|
|
44
63
|
startFrame: layout.scenes[i].startFrame,
|
|
@@ -64,6 +83,8 @@ export function missingAssets(project: Project, plan: ScenePlan): string[] {
|
|
|
64
83
|
const images = project.readJsonOr<Record<string, string>>(FILES.images, {});
|
|
65
84
|
const clips = project.readJsonOr<Record<string, ClipRecord>>(FILES.clips, {});
|
|
66
85
|
const problems: string[] = [];
|
|
86
|
+
const music = project.readJsonOr<MusicRecord | undefined>(FILES.music, undefined);
|
|
87
|
+
if (music && !project.exists(music.key)) problems.push(`Missing file: ${music.key}. Pull the track again with \`reelkit assets pull ${music.id} --music\`.`);
|
|
67
88
|
for (const s of plan.scenes) {
|
|
68
89
|
if (!voiceovers[s.id]) problems.push(`Scene ${s.id} has no voiceover.`);
|
|
69
90
|
else if (voiceoverStale(voiceovers[s.id], s, plan)) problems.push(`Scene ${s.id}'s voiceover is out of date. Run \`reelkit assets voiceover --all\`.`);
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import type { AssetManifest } from "../pipeline/schema";
|
|
2
|
+
import { analyzeBeats } from "./beats";
|
|
3
|
+
import { audioDuration, decodeMono } from "./refmeasure";
|
|
4
|
+
|
|
5
|
+
// The video's one music track, stored in assets/music.json. `bpm` and `beats` are present only when the pulse was clear enough to trust.
|
|
6
|
+
export type MusicRecord = { key: string; id: string; title: string; durationSec: number; bpm?: number; beatConfidence: number; beats: number[]; gainDb: number };
|
|
7
|
+
|
|
8
|
+
const RATE = 11025;
|
|
9
|
+
|
|
10
|
+
// Measures a pulled track on this machine: its length, its tempo and where its beats fall.
|
|
11
|
+
export async function measureMusic(path: string): Promise<Pick<MusicRecord, "durationSec" | "bpm" | "beatConfidence" | "beats">> {
|
|
12
|
+
const durationSec = Math.round((await audioDuration(path)) * 1000) / 1000;
|
|
13
|
+
const r = analyzeBeats(await decodeMono(path, RATE), RATE);
|
|
14
|
+
return { durationSec, beatConfidence: r.confidence, beats: r.tempoBpm !== undefined && r.beats ? r.beats : [], ...(r.tempoBpm !== undefined ? { bpm: r.tempoBpm } : {}) };
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
// How many scene changes fall on a beat, for the preview.
|
|
18
|
+
export function beatReport(manifest: AssetManifest): { bpm: number; boundariesOnBeat: number; boundaries: number; line: string } | undefined {
|
|
19
|
+
const m = manifest.music;
|
|
20
|
+
if (!m?.bpm || manifest.scenes.length < 2) return undefined;
|
|
21
|
+
const beats = new Set(m.beatFrames);
|
|
22
|
+
const starts = manifest.scenes.slice(1).map((s) => s.startFrame);
|
|
23
|
+
const on = starts.filter((f) => beats.has(f)).length;
|
|
24
|
+
return { bpm: m.bpm, boundariesOnBeat: on, boundaries: starts.length, line: `${on} of ${starts.length} scene changes land on the beat at ${Math.round(m.bpm)} BPM` };
|
|
25
|
+
}
|
package/src/project/project.ts
CHANGED
|
@@ -5,7 +5,7 @@ import { AspectSchema, type AssetRecord } from "../pipeline/schema";
|
|
|
5
5
|
|
|
6
6
|
export const FILES = {
|
|
7
7
|
config: "reelkit.json", plan: "plan.json", manifest: "manifest.json",
|
|
8
|
-
assetIndex: "assets/index.json", voiceovers: "assets/voiceovers.json", images: "assets/images.json", clips: "assets/clips.json", library: "assets/library.json",
|
|
8
|
+
assetIndex: "assets/index.json", voiceovers: "assets/voiceovers.json", images: "assets/images.json", clips: "assets/clips.json", library: "assets/library.json", music: "assets/music.json",
|
|
9
9
|
} as const;
|
|
10
10
|
|
|
11
11
|
// "path: message", or just the message for a problem at the root of the file.
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
// Measuring a reference video with ffmpeg and ffprobe, all on this machine. Every failure is a one-line Error for the person who gave the video.
|
|
2
|
+
import { execFile } from "node:child_process";
|
|
3
|
+
import { mkdirSync } from "node:fs";
|
|
4
|
+
import { dirname } from "node:path";
|
|
5
|
+
import { promisify } from "node:util";
|
|
6
|
+
|
|
7
|
+
const exec = promisify(execFile);
|
|
8
|
+
|
|
9
|
+
export const REF_ASPECTS = ["9:16", "16:9", "1:1", "4:5"] as const;
|
|
10
|
+
export type RefAspect = (typeof REF_ASPECTS)[number];
|
|
11
|
+
|
|
12
|
+
async function tool(cmd: "ffmpeg" | "ffprobe", args: string[]): Promise<{ stdout: Buffer; stderr: string }> {
|
|
13
|
+
try {
|
|
14
|
+
const r = await exec(cmd, args, { maxBuffer: 256 * 1024 * 1024, encoding: "buffer" });
|
|
15
|
+
return { stdout: r.stdout, stderr: r.stderr.toString("utf8") };
|
|
16
|
+
} catch (e) {
|
|
17
|
+
if ((e as NodeJS.ErrnoException)?.code === "ENOENT") throw new Error(`${cmd} was not found. Install ffmpeg (macOS: \`brew install ffmpeg\`) and try again.`);
|
|
18
|
+
throw e;
|
|
19
|
+
}
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export type VideoInfo = { durationSec: number; width: number; height: number; fps: number; hasAudio: boolean };
|
|
23
|
+
|
|
24
|
+
const rate = (r: string | undefined): number => {
|
|
25
|
+
const [a, b] = (r ?? "").split("/").map(Number);
|
|
26
|
+
return a && b ? a / b : a || 0;
|
|
27
|
+
};
|
|
28
|
+
|
|
29
|
+
export async function probeVideo(path: string, name: string): Promise<VideoInfo> {
|
|
30
|
+
const unreadable = new Error(`${name} could not be read as a video. It may be damaged or not a video: try another file or link.`);
|
|
31
|
+
let info: { streams?: { codec_type?: string; width?: number; height?: number; avg_frame_rate?: string; r_frame_rate?: string; duration?: string }[]; format?: { duration?: string } };
|
|
32
|
+
try {
|
|
33
|
+
info = JSON.parse((await tool("ffprobe", ["-v", "error", "-print_format", "json", "-show_format", "-show_streams", path])).stdout.toString("utf8"));
|
|
34
|
+
} catch (e) {
|
|
35
|
+
if (e instanceof Error && e.message.startsWith("ffprobe was not found")) throw e;
|
|
36
|
+
throw unreadable;
|
|
37
|
+
}
|
|
38
|
+
const v = info.streams?.find((s) => s.codec_type === "video");
|
|
39
|
+
const duration = Number(info.format?.duration ?? v?.duration);
|
|
40
|
+
if (!v || !(v.width! > 0) || !(v.height! > 0) || !(duration > 0)) throw unreadable;
|
|
41
|
+
const fps = rate(v.avg_frame_rate) || rate(v.r_frame_rate);
|
|
42
|
+
return { durationSec: Math.round(duration * 100) / 100, width: v.width!, height: v.height!, fps: Math.round(fps * 100) / 100, hasAudio: Boolean(info.streams?.some((s) => s.codec_type === "audio")) };
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
// The nearest of the four shapes, by how far the ratios are apart on a log scale (so 2:1 is as far from 1:1 as 1:2).
|
|
46
|
+
export function nearestAspect(width: number, height: number): RefAspect {
|
|
47
|
+
const ratio = width / height;
|
|
48
|
+
const value: Record<RefAspect, number> = { "9:16": 9 / 16, "16:9": 16 / 9, "1:1": 1, "4:5": 4 / 5 };
|
|
49
|
+
return REF_ASPECTS.reduce((best, a) => (Math.abs(Math.log(ratio / value[a])) < Math.abs(Math.log(ratio / value[best])) ? a : best));
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
const MIN_SHOT_SEC = 0.25;
|
|
53
|
+
|
|
54
|
+
// The times where the picture changes completely. Cuts closer than a quarter of a second to each other, to the start or to the end are one
|
|
55
|
+
// flash or a fade, not a shot of their own.
|
|
56
|
+
export async function detectCuts(video: string, durationSec: number, threshold = 0.3): Promise<number[]> {
|
|
57
|
+
const { stderr } = await tool("ffmpeg", ["-hide_banner", "-nostats", "-i", video, "-an", "-vf", `select='gt(scene,${threshold})',showinfo`, "-f", "null", "-"]);
|
|
58
|
+
const times = [...stderr.matchAll(/pts_time:\s*([0-9.]+)/g)].map((m) => Number(m[1])).sort((a, b) => a - b);
|
|
59
|
+
const cuts: number[] = [];
|
|
60
|
+
for (const t of times) {
|
|
61
|
+
if (t < MIN_SHOT_SEC || durationSec - t < MIN_SHOT_SEC) continue;
|
|
62
|
+
if (cuts.length && t - cuts[cuts.length - 1]! < MIN_SHOT_SEC) continue;
|
|
63
|
+
cuts.push(Math.round(t * 1000) / 1000);
|
|
64
|
+
}
|
|
65
|
+
return cuts;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
// One JPEG of the video at `atSec`, at most 540 pixels on the long side.
|
|
69
|
+
export async function extractFrame(video: string, atSec: number, dest: string): Promise<void> {
|
|
70
|
+
mkdirSync(dirname(dest), { recursive: true });
|
|
71
|
+
await tool("ffmpeg", ["-v", "error", "-y", "-ss", atSec.toFixed(3), "-i", video, "-frames:v", "1",
|
|
72
|
+
"-vf", "scale=w='if(gt(iw,ih),min(540,iw),-2)':h='if(gt(iw,ih),-2,min(540,ih))'", "-q:v", "4", dest]);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
type Pixel = [number, number, number];
|
|
76
|
+
|
|
77
|
+
// Averages of the main colours: each pixel goes in one of 64 buckets (four levels per channel), and the fullest buckets, skipping one whose
|
|
78
|
+
// colour is close to a colour already taken, are named by the average of their pixels.
|
|
79
|
+
export function dominantColors(pixels: Pixel[], max: number): string[] {
|
|
80
|
+
const buckets = new Map<number, { n: number; sum: Pixel }>();
|
|
81
|
+
for (const [r, g, b] of pixels) {
|
|
82
|
+
const key = (r >> 6) * 16 + (g >> 6) * 4 + (b >> 6);
|
|
83
|
+
const e = buckets.get(key) ?? { n: 0, sum: [0, 0, 0] as Pixel };
|
|
84
|
+
e.n++; e.sum[0] += r; e.sum[1] += g; e.sum[2] += b;
|
|
85
|
+
buckets.set(key, e);
|
|
86
|
+
}
|
|
87
|
+
const chosen: Pixel[] = [];
|
|
88
|
+
for (const e of [...buckets.values()].sort((a, b) => b.n - a.n)) {
|
|
89
|
+
const c = e.sum.map((v) => Math.round(v / e.n)) as Pixel;
|
|
90
|
+
if (chosen.some((o) => Math.hypot(o[0] - c[0], o[1] - c[1], o[2] - c[2]) < 48)) continue;
|
|
91
|
+
chosen.push(c);
|
|
92
|
+
if (chosen.length === max) break;
|
|
93
|
+
}
|
|
94
|
+
return chosen.map(([r, g, b]) => `#${[r, g, b].map((v) => v.toString(16).padStart(2, "0")).join("")}`);
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
// Every picture is shrunk to 4 by 4 pixels, which averages away detail and leaves the colour of each part of the frame.
|
|
98
|
+
async function tinyPixels(inputArgs: string[], filter: string): Promise<Pixel[]> {
|
|
99
|
+
const { stdout } = await tool("ffmpeg", ["-v", "error", ...inputArgs, "-an", "-vf", `${filter}scale=4:4:flags=area`, "-f", "rawvideo", "-pix_fmt", "rgb24", "-"]);
|
|
100
|
+
const out: Pixel[] = [];
|
|
101
|
+
for (let i = 0; i + 2 < stdout.length; i += 3) out.push([stdout[i]!, stdout[i + 1]!, stdout[i + 2]!]);
|
|
102
|
+
return out;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
export async function paletteOfVideo(video: string, durationSec: number, max = 5): Promise<string[]> {
|
|
106
|
+
// At most about 120 pictures however long the video is.
|
|
107
|
+
const fps = Math.max(0.15, Math.min(2, 120 / durationSec));
|
|
108
|
+
return dominantColors(await tinyPixels(["-i", video], `fps=${fps.toFixed(3)},`), max);
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
export async function paletteOfFrames(frames: string[], max = 3): Promise<string[]> {
|
|
112
|
+
const pixels: Pixel[] = [];
|
|
113
|
+
for (const f of frames) pixels.push(...(await tinyPixels(["-i", f], "")));
|
|
114
|
+
return dominantColors(pixels, max);
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
// The integrated loudness in LUFS, or undefined when there is none (silence is minus infinity).
|
|
118
|
+
export async function loudnessLufs(video: string): Promise<number | undefined> {
|
|
119
|
+
const { stderr } = await tool("ffmpeg", ["-hide_banner", "-nostats", "-i", video, "-vn", "-af", "ebur128=framelog=quiet", "-f", "null", "-"]);
|
|
120
|
+
const all = [...stderr.matchAll(/\bI:\s+(-?[0-9.]+)\s+LUFS/g)];
|
|
121
|
+
const value = Number(all.at(-1)?.[1]);
|
|
122
|
+
return Number.isFinite(value) ? Math.round(value * 10) / 10 : undefined;
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
// The audio as mono samples between -1 and 1, for the beat analysis.
|
|
126
|
+
export async function decodeMono(video: string, sampleRate: number): Promise<Float32Array> {
|
|
127
|
+
const { stdout } = await tool("ffmpeg", ["-v", "error", "-i", video, "-vn", "-ac", "1", "-ar", String(sampleRate), "-f", "s16le", "-"]);
|
|
128
|
+
const n = Math.floor(stdout.length / 2);
|
|
129
|
+
const out = new Float32Array(n);
|
|
130
|
+
for (let i = 0; i < n; i++) out[i] = stdout.readInt16LE(i * 2) / 32768;
|
|
131
|
+
return out;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
// Speech needs little: mono, 16 kHz and 64 kbit/s keep a ten-minute reference under 5 MB.
|
|
135
|
+
export async function extractMp3(video: string, dest: string): Promise<void> {
|
|
136
|
+
mkdirSync(dirname(dest), { recursive: true });
|
|
137
|
+
await tool("ffmpeg", ["-v", "error", "-y", "-i", video, "-vn", "-ac", "1", "-ar", "16000", "-b:a", "64k", "-c:a", "libmp3lame", dest]);
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
export async function toMp4(source: string, dest: string): Promise<void> {
|
|
141
|
+
mkdirSync(dirname(dest), { recursive: true });
|
|
142
|
+
await tool("ffmpeg", ["-v", "error", "-y", "-i", source, "-c:v", "libx264", "-pix_fmt", "yuv420p", "-c:a", "aac", "-movflags", "+faststart", dest]);
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
export async function audioDuration(path: string): Promise<number> {
|
|
146
|
+
const { stdout } = await tool("ffprobe", ["-v", "error", "-show_entries", "format=duration", "-of", "csv=p=0", path]);
|
|
147
|
+
return Number(stdout.toString("utf8").trim());
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
// The mean absolute difference, from 0 to 255, between each frame of a run of grey frames and the next. frameSize is the pixels in one frame.
|
|
151
|
+
export function meanAbsDiffs(frames: Uint8Array, frameSize: number): number[] {
|
|
152
|
+
const count = Math.floor(frames.length / frameSize);
|
|
153
|
+
const out: number[] = [];
|
|
154
|
+
for (let f = 0; f + 1 < count; f++) {
|
|
155
|
+
let sum = 0;
|
|
156
|
+
for (let i = 0, a = f * frameSize, b = a + frameSize; i < frameSize; i++) sum += Math.abs(frames[a + i]! - frames[b + i]!);
|
|
157
|
+
out.push(sum / frameSize);
|
|
158
|
+
}
|
|
159
|
+
return out;
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
// Two frames whose pictures differ by less than this, on average over 255 grey levels, are the same picture for a viewer: video
|
|
163
|
+
// compression noise on a held shot stays well under it, and a slow push still stays above it.
|
|
164
|
+
const STILL_BELOW = 1;
|
|
165
|
+
|
|
166
|
+
// The share of the video's time in which the picture barely changes from one sampled frame to the next, and the longest unbroken stretch
|
|
167
|
+
// of it. diffs has one entry for each pair of consecutive samples, taken `rate` times a second.
|
|
168
|
+
export function stillness(diffs: number[], rate: number): { share: number; longestSec: number } {
|
|
169
|
+
if (!diffs.length) return { share: 0, longestSec: 0 };
|
|
170
|
+
let still = 0, run = 0, longest = 0;
|
|
171
|
+
for (const d of diffs) {
|
|
172
|
+
if (d < STILL_BELOW) { still++; longest = Math.max(longest, ++run); } else run = 0;
|
|
173
|
+
}
|
|
174
|
+
const round = (n: number) => Math.round(n * 1000) / 1000;
|
|
175
|
+
return { share: round(still / diffs.length), longestSec: round(longest / rate) };
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
// Sampled at 10 frames a second and 64 pixels wide, a ten-minute video is about 44 MB of grey pixels and is measured in a second or two.
|
|
179
|
+
export const STILL_SAMPLE = { rate: 10, width: 64 };
|
|
180
|
+
|
|
181
|
+
export async function stillnessOfVideo(video: string, width: number, height: number): Promise<{ share: number; longestSec: number }> {
|
|
182
|
+
const w = STILL_SAMPLE.width, h = Math.max(2, Math.round((w * height) / width / 2) * 2);
|
|
183
|
+
const { stdout } = await tool("ffmpeg", ["-v", "error", "-i", video, "-an", "-vf", `fps=${STILL_SAMPLE.rate},scale=${w}:${h}:flags=area,format=gray`, "-f", "rawvideo", "-pix_fmt", "gray", "-"]);
|
|
184
|
+
return stillness(meanAbsDiffs(new Uint8Array(stdout.buffer, stdout.byteOffset, stdout.length), w * h), STILL_SAMPLE.rate);
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
// The standard deviation of the values over their mean: 0 when they are all equal.
|
|
188
|
+
export function coefficientOfVariation(values: number[]): number {
|
|
189
|
+
const mean = values.reduce((a, b) => a + b, 0) / (values.length || 1);
|
|
190
|
+
if (!(mean > 0)) return 0;
|
|
191
|
+
return Math.sqrt(values.reduce((a, b) => a + (b - mean) ** 2, 0) / values.length) / mean;
|
|
192
|
+
}
|