reelkit-cli 0.10.5 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -1
- package/README.md +12 -2
- package/package.json +2 -1
- package/skill/SKILL.md +40 -5
- package/skill/reference/delivery.md +39 -0
- package/skill/reference/hebrew-rtl.md +10 -1
- package/skill/reference/reference-recreation.md +33 -0
- package/skill/reference/revisions.md +36 -0
- package/skill/reference/rights.md +27 -0
- package/skill/reference/voice-fixes.md +33 -0
- package/skill/reference/voice-sync.md +4 -0
- package/src/cli.ts +31 -2
- package/src/clock/clock.ts +76 -0
- package/src/commands/build.ts +82 -7
- package/src/commands/clock.ts +37 -0
- package/src/commands/diff.ts +81 -0
- package/src/commands/export.ts +51 -0
- package/src/commands/lint.ts +94 -0
- package/src/diff/framediff.ts +121 -0
- package/src/export/presets.ts +147 -0
- package/src/lint/pixel-rules.ts +149 -0
- package/src/lint/source-rules.ts +186 -0
- package/src/media/ffmpeg.ts +84 -0
- package/src/render/chunked.ts +139 -0
- package/src/render/render.ts +19 -2
- package/src/render/worker.ts +98 -54
- package/tools/voice/README.md +26 -0
- package/tools/voice/_audio.py +118 -0
- package/tools/voice/ab_video.py +93 -0
- package/tools/voice/credits.py +35 -0
- package/tools/voice/onsets.py +70 -0
- package/tools/voice/phonemes.py +85 -0
- package/tools/voice/pitch_check.py +51 -0
- package/tools/voice/place_line.py +67 -0
- package/tools/voice/splice.py +65 -0
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
// Rendering a long film in pieces that survive a crash. Each piece is a silent H.264 file of a range of frames, written whole or not at all; a piece
|
|
2
|
+
// that is already on disk, reads back with the right number of frames and has its audio sidecar is not rendered again. The pieces are joined
|
|
3
|
+
// without re-encoding. The sound is mixed once from those sidecars and laid under them. The planning here is pure; the rendering is handed in.
|
|
4
|
+
import { createHash } from "node:crypto";
|
|
5
|
+
import { existsSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, statSync, writeFileSync } from "node:fs";
|
|
6
|
+
import { join } from "node:path";
|
|
7
|
+
import { syncFile } from "../media/ffmpeg";
|
|
8
|
+
import { tool } from "../project/refmeasure";
|
|
9
|
+
|
|
10
|
+
export const DEFAULT_CHUNK_FRAMES = 300;
|
|
11
|
+
export const MAX_ATTEMPTS = 3;
|
|
12
|
+
|
|
13
|
+
export type Chunk = { index: number; from: number; to: number; file: string };
|
|
14
|
+
|
|
15
|
+
// Frame ranges (inclusive) that cover the film.
|
|
16
|
+
export function planChunks(totalFrames: number, chunkFrames = DEFAULT_CHUNK_FRAMES): Chunk[] {
|
|
17
|
+
const size = Math.max(1, Math.floor(chunkFrames));
|
|
18
|
+
const out: Chunk[] = [];
|
|
19
|
+
for (let from = 0, index = 0; from < totalFrames; from += size, index++) {
|
|
20
|
+
const to = Math.min(totalFrames, from + size) - 1;
|
|
21
|
+
out.push({ index, from, to, file: `chunk-${String(from).padStart(6, "0")}-${String(to).padStart(6, "0")}.mp4` });
|
|
22
|
+
}
|
|
23
|
+
return out;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
// What the pieces were rendered from. A piece made from other source, other media or another size is not reused.
|
|
27
|
+
export function renderKey(parts: { sources: Record<string, string>; manifest: unknown; media: Record<string, number>; chunkFrames: number }): string {
|
|
28
|
+
const h = createHash("sha256");
|
|
29
|
+
for (const name of Object.keys(parts.sources).sort()) h.update(name).update("\0").update(parts.sources[name]!).update("\0");
|
|
30
|
+
h.update(JSON.stringify(parts.manifest)).update("\0");
|
|
31
|
+
for (const name of Object.keys(parts.media).sort()) h.update(`${name}:${parts.media[name]}\0`);
|
|
32
|
+
h.update(String(parts.chunkFrames));
|
|
33
|
+
return h.digest("hex").slice(0, 16);
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
// A failure of the browser rather than of the composition: worth another attempt.
|
|
37
|
+
export const isBrowserFailure = (e: unknown): boolean => /browser|Target closed|ProtocolError|Protocol error|Navigation|WebSocket|ECONNRESET|ECONNREFUSED|Timed out|timeout|crash|disconnected|Session closed/i.test(e instanceof Error ? e.message : String(e));
|
|
38
|
+
|
|
39
|
+
// Runs fn, and again after a browser failure, up to `attempts` times in all. An error of the composition itself is thrown at once.
|
|
40
|
+
export async function withRetry<T>(fn: (attempt: number) => Promise<T>, opts: { attempts?: number; onRetry?: (attempt: number, error: unknown) => void; wait?: (ms: number) => Promise<void> } = {}): Promise<T> {
|
|
41
|
+
const attempts = opts.attempts ?? MAX_ATTEMPTS;
|
|
42
|
+
for (let attempt = 1; ; attempt++) {
|
|
43
|
+
try { return await fn(attempt); }
|
|
44
|
+
catch (e) {
|
|
45
|
+
if (attempt >= attempts || !isBrowserFailure(e)) throw e;
|
|
46
|
+
opts.onRetry?.(attempt, e);
|
|
47
|
+
await (opts.wait ?? ((ms) => new Promise((r) => setTimeout(r, ms))))(1500 * attempt);
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
export type ChunkDeps = {
|
|
53
|
+
// Renders frames from..to (inclusive), silent, to `out`. May also write `${out}.audio.json`.
|
|
54
|
+
renderRange: (from: number, to: number, out: string) => Promise<void>;
|
|
55
|
+
// Renders the whole film's sound to `out` (a wav or m4a); resolves false when the film has none.
|
|
56
|
+
renderSound: (out: string) => Promise<boolean>;
|
|
57
|
+
// How many frames a finished piece holds; undefined when it cannot be read.
|
|
58
|
+
countFrames: (file: string) => Promise<number | undefined>;
|
|
59
|
+
log: (line: string) => void;
|
|
60
|
+
concat?: (list: string, sound: string | undefined, out: string) => Promise<void>;
|
|
61
|
+
// Extra reason to render a piece again (for example its audio sidecar is missing).
|
|
62
|
+
reuseExtra?: (file: string) => boolean;
|
|
63
|
+
};
|
|
64
|
+
|
|
65
|
+
export type CapturedAudio = {
|
|
66
|
+
id: string; src: string; start: number; end: number; mediaStart: number; layer: number; volume: number;
|
|
67
|
+
volumeKeyframes: { time: number; volume: number }[]; type: "audio" | "video"; loop: boolean;
|
|
68
|
+
};
|
|
69
|
+
|
|
70
|
+
// Pieces of one film are captured in separate browser sessions, so the same clip can come back under different ids.
|
|
71
|
+
// Merge by where it sits, and keep every volume sample.
|
|
72
|
+
export function mergeAudioRecords(lists: CapturedAudio[][]): CapturedAudio[] {
|
|
73
|
+
const byKey = new Map<string, CapturedAudio>();
|
|
74
|
+
for (const list of lists) for (const record of list) {
|
|
75
|
+
const key = `${record.type}|${record.src}|${record.start}|${record.mediaStart}|${record.loop}`;
|
|
76
|
+
const prev = byKey.get(key);
|
|
77
|
+
if (!prev) { byKey.set(key, { ...record, volumeKeyframes: [...record.volumeKeyframes] }); continue; }
|
|
78
|
+
prev.end = Math.max(prev.end, record.end);
|
|
79
|
+
prev.volumeKeyframes.push(...record.volumeKeyframes);
|
|
80
|
+
}
|
|
81
|
+
for (const record of byKey.values()) {
|
|
82
|
+
const seen = new Map<number, { time: number; volume: number }>();
|
|
83
|
+
for (const key of record.volumeKeyframes) seen.set(key.time, key);
|
|
84
|
+
record.volumeKeyframes = [...seen.values()].sort((a, b) => a.time - b.time);
|
|
85
|
+
}
|
|
86
|
+
return [...byKey.values()];
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
export async function countFrames(file: string): Promise<number | undefined> {
|
|
90
|
+
try {
|
|
91
|
+
const { stdout } = await tool("ffprobe", ["-v", "error", "-select_streams", "v:0", "-count_packets", "-show_entries", "stream=nb_read_packets", "-of", "csv=p=0", file]);
|
|
92
|
+
const n = Number(stdout.toString("utf8").trim().split(/[,\n]/)[0]);
|
|
93
|
+
return Number.isFinite(n) ? n : undefined;
|
|
94
|
+
} catch { return undefined; }
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
// Joins the pieces without touching their pictures and lays the sound under them.
|
|
98
|
+
export async function concatChunks(list: string, sound: string | undefined, out: string): Promise<void> {
|
|
99
|
+
await tool("ffmpeg", ["-nostdin", "-y", "-v", "error", "-f", "concat", "-safe", "0", "-i", list, ...(sound ? ["-i", sound] : []), "-map", "0:v:0", ...(sound ? ["-map", "1:a:0", "-c:a", "aac", "-b:a", "320k", "-ar", "48000"] : []), "-c:v", "copy", "-movflags", "+faststart", out]);
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
export type ChunkedResult = { chunks: number; reused: number; rendered: number; retries: number };
|
|
103
|
+
|
|
104
|
+
// Renders whatever pieces are missing into `dir`, then assembles `outPath`. Safe to run again after any interruption: finished pieces are kept.
|
|
105
|
+
export async function renderChunked(dir: string, key: string, totalFrames: number, chunkFrames: number, outPath: string, deps: ChunkDeps): Promise<ChunkedResult> {
|
|
106
|
+
mkdirSync(dir, { recursive: true });
|
|
107
|
+
const keyFile = join(dir, "key");
|
|
108
|
+
// Pieces of another version of the film are of no use: start clean.
|
|
109
|
+
if (existsSync(keyFile) && readFileSync(keyFile, "utf8").trim() !== key) { for (const f of readdirSync(dir)) rmSync(join(dir, f), { force: true, recursive: true }); }
|
|
110
|
+
writeFileSync(keyFile, key);
|
|
111
|
+
const chunks = planChunks(totalFrames, chunkFrames);
|
|
112
|
+
let reused = 0, rendered = 0, retries = 0;
|
|
113
|
+
for (const c of chunks) {
|
|
114
|
+
const file = join(dir, c.file);
|
|
115
|
+
const want = c.to - c.from + 1;
|
|
116
|
+
if (existsSync(file) && statSync(file).size > 0 && (await deps.countFrames(file)) === want && (deps.reuseExtra?.(file) ?? true)) { reused++; continue; }
|
|
117
|
+
const part = `${file}.part.mp4`;
|
|
118
|
+
const partAudio = `${part}.audio.json`;
|
|
119
|
+
await withRetry(async () => { rmSync(part, { force: true }); rmSync(partAudio, { force: true }); await deps.renderRange(c.from, c.to, part); }, { onRetry: (n, e) => { retries++; deps.log(`The browser failed on frames ${c.from}-${c.to} (${(e instanceof Error ? e.message : String(e)).split("\n")[0]}); trying again (${n + 1} of ${MAX_ATTEMPTS}).`); } });
|
|
120
|
+
const got = await deps.countFrames(part);
|
|
121
|
+
if (got !== want) throw new Error(`frames ${c.from}-${c.to} rendered as ${got ?? "an unreadable file"} instead of ${want} frames`);
|
|
122
|
+
syncFile(part);
|
|
123
|
+
renameSync(part, file);
|
|
124
|
+
if (existsSync(partAudio)) { syncFile(partAudio); renameSync(partAudio, `${file}.audio.json`); }
|
|
125
|
+
rendered++;
|
|
126
|
+
deps.log(`Frames ${c.from}-${c.to} done (${c.index + 1} of ${chunks.length}).`);
|
|
127
|
+
}
|
|
128
|
+
const soundFile = join(dir, "sound.m4a");
|
|
129
|
+
let hasSound = existsSync(soundFile) && statSync(soundFile).size > 44;
|
|
130
|
+
if (!hasSound) {
|
|
131
|
+
const part = join(dir, "sound.part.m4a");
|
|
132
|
+
hasSound = await withRetry(() => deps.renderSound(part), { onRetry: () => { retries++; } });
|
|
133
|
+
if (hasSound) { syncFile(part); renameSync(part, soundFile); } else rmSync(part, { force: true });
|
|
134
|
+
}
|
|
135
|
+
const list = join(dir, "list.txt");
|
|
136
|
+
writeFileSync(list, chunks.map((c) => `file '${join(dir, c.file).replace(/'/g, "'\\''")}'`).join("\n") + "\n");
|
|
137
|
+
await (deps.concat ?? concatChunks)(list, hasSound ? soundFile : undefined, outPath);
|
|
138
|
+
return { chunks: chunks.length, reused, rendered, retries };
|
|
139
|
+
}
|
package/src/render/render.ts
CHANGED
|
@@ -45,6 +45,7 @@ export async function disposeBundle(dir: string): Promise<void> {
|
|
|
45
45
|
export type EnsureBrowser = (onDownload: () => void) => Promise<void>;
|
|
46
46
|
export const GL_BACKEND = "software" as const;
|
|
47
47
|
export type RenderOptions = { uses3D?: boolean; silent?: boolean; crf?: number };
|
|
48
|
+
export type FrameRange = { from: number; to: number };
|
|
48
49
|
export function browserPath(): string | undefined {
|
|
49
50
|
if (process.env.REELKIT_CHROME_PATH) return process.env.REELKIT_CHROME_PATH;
|
|
50
51
|
const paths = process.platform === "darwin" ? ["/Applications/Google Chrome.app/Contents/MacOS/Google Chrome", "/Applications/Chromium.app/Contents/MacOS/Chromium"] : process.platform === "linux" ? ["/usr/bin/chromium", "/usr/bin/chromium-browser", "/usr/bin/google-chrome"] : [];
|
|
@@ -63,12 +64,12 @@ export const ensureRenderBrowser: EnsureBrowser = async (onDownload) => {
|
|
|
63
64
|
const browser = await install({ browser: Browser.CHROMEHEADLESSSHELL, buildId: await resolveBuildId(Browser.CHROMEHEADLESSSHELL, platform, "stable"), cacheDir: join(PKG_ROOT, ".browser"), platform });
|
|
64
65
|
process.env.REELKIT_CHROME_PATH = browser.executablePath;
|
|
65
66
|
};
|
|
66
|
-
async function render(dir: string, props: VideoProps, mode: "stills" | "video", output: string, frames: number[], options?: RenderOptions) {
|
|
67
|
+
async function render(dir: string, props: VideoProps, mode: "stills" | "video", output: string, frames: number[], options?: RenderOptions, extra?: { range?: FrameRange; audioSidecar?: string }) {
|
|
67
68
|
await ensureRenderBrowser(() => {});
|
|
68
69
|
await writeFile(join(dir, "props.js"), `window.__reelkitProps=${JSON.stringify(props).replace(/</g, "\\u003c")};`);
|
|
69
70
|
// Isolate engine diagnostics from the CLI's JSON stdout.
|
|
70
71
|
const request = join(dir, "render-request.json");
|
|
71
|
-
await writeFile(request, JSON.stringify({ dir, props, mode, output, frames, options, chromePath: browserPath() }));
|
|
72
|
+
await writeFile(request, JSON.stringify({ dir, props, mode, output, frames, options, chromePath: browserPath(), ...extra }));
|
|
72
73
|
await run(process.execPath, ["--import", join(packageDir("tsx"), "dist/loader.mjs"), join(PKG_ROOT, "src/render/worker.ts"), request], { maxBuffer: 16 * 1024 * 1024 });
|
|
73
74
|
}
|
|
74
75
|
export async function renderStills(dir: string, props: VideoProps, frames: number[], outDir: string, options?: RenderOptions) {
|
|
@@ -77,3 +78,19 @@ export async function renderStills(dir: string, props: VideoProps, frames: numbe
|
|
|
77
78
|
export async function renderVideo(dir: string, props: VideoProps, outPath: string, options?: RenderOptions) {
|
|
78
79
|
await mkdir(dirname(outPath), { recursive: true }); await render(dir, props, "video", outPath, [], options);
|
|
79
80
|
}
|
|
81
|
+
// One inclusive range of frames as a silent file, plus the audio records those frames registered (`${outPath}.audio.json`).
|
|
82
|
+
export async function renderVideoRange(dir: string, props: VideoProps, outPath: string, from: number, to: number, options?: RenderOptions) {
|
|
83
|
+
await mkdir(dirname(outPath), { recursive: true });
|
|
84
|
+
await render(dir, props, "video", outPath, [], { ...options, silent: true }, { range: { from, to }, audioSidecar: `${outPath}.audio.json` });
|
|
85
|
+
}
|
|
86
|
+
// Mix audio records already collected from the pieces. No picture is drawn.
|
|
87
|
+
export async function mixCapturedAudio(records: unknown[], outPath: string, durationSec: number) {
|
|
88
|
+
const dir = await mkdtemp(join(tmpdir(), "reelkit-mix-"));
|
|
89
|
+
try {
|
|
90
|
+
const request = join(dir, "render-request.json");
|
|
91
|
+
await writeFile(request, JSON.stringify({ mode: "mix", output: outPath, records, durationSec, frames: [] }));
|
|
92
|
+
await run(process.execPath, ["--import", join(packageDir("tsx"), "dist/loader.mjs"), join(PKG_ROOT, "src/render/worker.ts"), request], { maxBuffer: 16 * 1024 * 1024 });
|
|
93
|
+
} finally {
|
|
94
|
+
await rm(dir, { recursive: true, force: true });
|
|
95
|
+
}
|
|
96
|
+
}
|
package/src/render/worker.ts
CHANGED
|
@@ -2,70 +2,114 @@
|
|
|
2
2
|
import { execFile } from "node:child_process";
|
|
3
3
|
import { cp, mkdir, mkdtemp, rm, writeFile, readFile } from "node:fs/promises";
|
|
4
4
|
import { tmpdir } from "node:os";
|
|
5
|
-
import { join, extname } from "node:path";
|
|
5
|
+
import { dirname, join, extname } from "node:path";
|
|
6
6
|
import { promisify } from "node:util";
|
|
7
7
|
import { createCaptureSession, initializeSession, captureFrame, closeCaptureSession, encodeFramesFromDir, processCompositionAudio, muxVideoWithAudio, applyFaststart, type AudioElement } from "@hyperframes/engine";
|
|
8
8
|
import { serveComposition } from "./serve";
|
|
9
9
|
import type { VideoProps } from "../hyperframes/types";
|
|
10
10
|
import type { AudioRecord } from "../hyperframes/frame";
|
|
11
|
-
import type { RenderOptions } from "./render";
|
|
11
|
+
import type { FrameRange, RenderOptions } from "./render";
|
|
12
12
|
|
|
13
13
|
const run = promisify(execFile);
|
|
14
|
-
const input = JSON.parse(await readFile(process.argv[2]!, "utf8")) as {
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
14
|
+
const input = JSON.parse(await readFile(process.argv[2]!, "utf8")) as {
|
|
15
|
+
dir?: string; props?: VideoProps; mode: "stills" | "video" | "mix"; output: string; frames: number[];
|
|
16
|
+
options?: RenderOptions; chromePath?: string; range?: FrameRange; audioSidecar?: string; records?: AudioRecord[]; durationSec?: number;
|
|
17
|
+
};
|
|
18
|
+
|
|
19
|
+
async function mixRecords(records: AudioRecord[], output: string, durationSec: number, work: string): Promise<boolean> {
|
|
20
|
+
if (!records.length) return false;
|
|
21
|
+
const audio: AudioElement[] = [];
|
|
22
|
+
const localSources = new Map<string, string>();
|
|
23
|
+
for (const record of records) {
|
|
24
|
+
let src = localSources.get(record.src);
|
|
25
|
+
if (!src) {
|
|
26
|
+
const response = await fetch(record.src); if (!response.ok) throw new Error(`Audio source returned ${response.status}.`);
|
|
27
|
+
src = join(work, `source-${localSources.size}${extname(new URL(record.src).pathname) || ".media"}`);
|
|
28
|
+
await writeFile(src, new Uint8Array(await response.arrayBuffer())); localSources.set(record.src, src);
|
|
29
|
+
}
|
|
30
|
+
const info = JSON.parse((await run("ffprobe", ["-v", "error", "-show_format", "-of", "json", src])).stdout) as { format: { duration: string } };
|
|
31
|
+
let end = Math.min(record.end, record.start + Number(info.format.duration) - record.mediaStart);
|
|
32
|
+
if (record.loop) {
|
|
33
|
+
const looped = join(work, `loop-${audio.length}.wav`);
|
|
34
|
+
await run("ffmpeg", ["-nostdin", "-y", "-v", "error", "-stream_loop", "-1", "-i", src, "-t", String(record.end - record.start + record.mediaStart), "-vn", looped]);
|
|
35
|
+
src = looped; end = record.end;
|
|
36
|
+
}
|
|
37
|
+
if (end <= record.start) continue;
|
|
38
|
+
const keys = [...new Map(record.volumeKeyframes.map((k) => [k.time, k])).values()].sort((a, b) => a.time - b.time);
|
|
39
|
+
audio.push({ ...record, src, end, volumeKeyframes: keys.length ? keys : undefined });
|
|
40
|
+
}
|
|
41
|
+
if (!audio.length) return false;
|
|
42
|
+
const mixed = output.endsWith(".m4a") ? output : join(work, "audio.m4a");
|
|
43
|
+
const result = await processCompositionAudio(audio, work, join(work, "audio-work"), mixed, durationSec);
|
|
44
|
+
if (!result.success) throw new Error(result.error ?? "HyperFrames audio mixing failed.");
|
|
45
|
+
if (mixed !== output) await cp(mixed, output);
|
|
46
|
+
return true;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
if (input.mode === "mix") {
|
|
50
|
+
const work = await mkdtemp(join(tmpdir(), "reelkit-mix-"));
|
|
51
|
+
try {
|
|
52
|
+
await mkdir(dirname(input.output), { recursive: true });
|
|
53
|
+
const wrote = await mixRecords(input.records ?? [], input.output, input.durationSec ?? 0, work);
|
|
54
|
+
if (!wrote) throw new Error("This film has no sound to mix.");
|
|
55
|
+
} finally {
|
|
56
|
+
await rm(work, { recursive: true, force: true });
|
|
30
57
|
}
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
58
|
+
} else {
|
|
59
|
+
const props = input.props!;
|
|
60
|
+
const m = props.manifest, fps = { num: m.fps, den: 1 };
|
|
61
|
+
const work = await mkdtemp(join(tmpdir(), "reelkit-capture-"));
|
|
62
|
+
const server = await serveComposition(input.dir!);
|
|
63
|
+
let session: Awaited<ReturnType<typeof createCaptureSession>> | undefined;
|
|
64
|
+
try {
|
|
65
|
+
session = await createCaptureSession(server.url.replace(/\/$/, ""), work, { width: m.width, height: m.height, fps, format: "png", compositionDurationSeconds: m.totalFrames / m.fps }, async (page) => {
|
|
66
|
+
await page.evaluate(async () => { await (window as unknown as { __reelkitReady: Promise<void> }).__reelkitReady; });
|
|
67
|
+
}, { chromePath: input.chromePath, forceScreenshot: true, useDrawElement: false, staticFrameDedup: false, enableBrowserPool: false, browserGpuMode: input.options?.uses3D ? "software" : "auto" });
|
|
68
|
+
await initializeSession(session);
|
|
69
|
+
await session.page.evaluate(async () => { await (window as unknown as { __reelkitReady: Promise<void> }).__reelkitReady; });
|
|
70
|
+
const from = input.range?.from ?? 0, to = input.range?.to ?? m.totalFrames - 1;
|
|
71
|
+
const frames = input.mode === "stills" ? input.frames : Array.from({ length: to - from + 1 }, (_, i) => from + i);
|
|
72
|
+
for (const frame of frames) {
|
|
73
|
+
if (!Number.isInteger(frame) || frame < 0 || frame >= m.totalFrames) throw new Error(`Frame ${frame} is outside the composition.`);
|
|
74
|
+
const captured = await captureFrame(session, frame, frame / m.fps);
|
|
75
|
+
if (input.mode === "stills") await cp(captured.path, join(input.output, `still-${frame}.png`));
|
|
76
|
+
}
|
|
77
|
+
if (input.mode === "video") {
|
|
78
|
+
// The encoder reads frame_000000.png onward. A piece that starts later is renamed into its own folder.
|
|
79
|
+
let framesDir = work, pattern = "frame_%06d.png";
|
|
80
|
+
if (from !== 0) {
|
|
81
|
+
framesDir = join(work, "seq"); await mkdir(framesDir);
|
|
82
|
+
for (let i = 0; i < frames.length; i++) await cp(join(work, `frame_${String(from + i).padStart(6, "0")}.png`), join(framesDir, `frame_${String(i).padStart(6, "0")}.png`));
|
|
44
83
|
}
|
|
45
|
-
const
|
|
46
|
-
|
|
47
|
-
if (
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
84
|
+
const silent = join(work, "silent.mp4");
|
|
85
|
+
const encoded = await encodeFramesFromDir(framesDir, pattern, silent, { fps, width: m.width, height: m.height, codec: "h264", pixelFormat: "yuv420p", quality: input.options?.crf ?? 18, preset: "fast" });
|
|
86
|
+
if (!encoded.success) throw new Error(encoded.error ?? "HyperFrames encoding failed.");
|
|
87
|
+
const collect = Boolean(input.audioSidecar) || !input.options?.silent;
|
|
88
|
+
const records: AudioRecord[] = collect ? await session.page.evaluate(() => [...(window as unknown as { __reelkitAudio: Map<string, AudioRecord> }).__reelkitAudio.values()]) : [];
|
|
89
|
+
if (input.audioSidecar) await writeFile(input.audioSidecar, JSON.stringify(records));
|
|
90
|
+
if (input.options?.silent || input.range) {
|
|
91
|
+
await mkdir(join(work, "faststart"));
|
|
92
|
+
const result = await applyFaststart(silent, input.output);
|
|
93
|
+
if (!result.success) throw new Error(result.error ?? "Faststart failed.");
|
|
94
|
+
} else if (records.length) {
|
|
95
|
+
const mixed = join(work, "audio.m4a");
|
|
96
|
+
const wrote = await mixRecords(records, mixed, m.totalFrames / m.fps, work);
|
|
97
|
+
if (!wrote) {
|
|
98
|
+
await mkdir(join(work, "faststart"));
|
|
99
|
+
const result = await applyFaststart(silent, input.output);
|
|
100
|
+
if (!result.success) throw new Error(result.error ?? "Faststart failed.");
|
|
101
|
+
} else {
|
|
102
|
+
const muxed = await muxVideoWithAudio(silent, mixed, input.output, undefined, { audioCodec: "aac" });
|
|
103
|
+
if (!muxed.success) throw new Error(muxed.error ?? "HyperFrames audio mux failed.");
|
|
104
|
+
}
|
|
105
|
+
} else {
|
|
106
|
+
await mkdir(join(work, "faststart"));
|
|
107
|
+
const result = await applyFaststart(silent, input.output);
|
|
108
|
+
if (!result.success) throw new Error(result.error ?? "Faststart failed.");
|
|
51
109
|
}
|
|
52
|
-
if (end <= record.start) continue;
|
|
53
|
-
const keys = [...new Map(record.volumeKeyframes.map((k) => [k.time, k])).values()].sort((a, b) => a.time - b.time);
|
|
54
|
-
audio.push({ ...record, src, end, volumeKeyframes: keys.length ? keys : undefined });
|
|
55
|
-
}
|
|
56
|
-
if (audio.length) {
|
|
57
|
-
const mixed = join(work, "audio.m4a");
|
|
58
|
-
const result = await processCompositionAudio(audio, work, join(work, "audio-work"), mixed, m.totalFrames / m.fps);
|
|
59
|
-
if (!result.success) throw new Error(result.error ?? "HyperFrames audio mixing failed.");
|
|
60
|
-
const muxed = await muxVideoWithAudio(silent, mixed, input.output, undefined, { audioCodec: "aac" });
|
|
61
|
-
if (!muxed.success) throw new Error(muxed.error ?? "HyperFrames audio mux failed.");
|
|
62
|
-
} else {
|
|
63
|
-
await mkdir(join(work, "faststart"));
|
|
64
|
-
const result = await applyFaststart(silent, input.output);
|
|
65
|
-
if (!result.success) throw new Error(result.error ?? "Faststart failed.");
|
|
66
110
|
}
|
|
111
|
+
} finally {
|
|
112
|
+
if (session) await closeCaptureSession(session);
|
|
113
|
+
await server.close(); await rm(work, { recursive: true, force: true });
|
|
67
114
|
}
|
|
68
|
-
} finally {
|
|
69
|
-
if (session) await closeCaptureSession(session);
|
|
70
|
-
await server.close(); await rm(work, { recursive: true, force: true });
|
|
71
115
|
}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# Voice tools
|
|
2
|
+
|
|
3
|
+
Scripts for getting one narration line right and into a film without disturbing anything else. They are plain Python (numpy, ffmpeg; Pillow for the
|
|
4
|
+
options video; soundfile for `place_line.py replace`). None of them generates speech; only `credits.py --check` talks to a network, and only to read a balance.
|
|
5
|
+
|
|
6
|
+
| script | what it answers |
|
|
7
|
+
|---|---|
|
|
8
|
+
| `credits.py --takes N --chars C [--check]` | What will this batch cost, and can the account pay for it? Run before generating. |
|
|
9
|
+
| `phonemes.py take.wav --want "m e a f j e n i m"` | What does the take actually say? Speech-to-text spells the expected word even when it was mispronounced; phonemes do not. Exit 2 (not a failure) when the model's packages are not installed. |
|
|
10
|
+
| `pitch_check.py take.wav` | Pitch range, largest step, and the pitch over the last 0.6 s. Verdicts on octave jumps and question-like endings need `praat-parselmouth`. |
|
|
11
|
+
| `ab_video.py --title "..." --out options.mp4 a.mp3:"current" b.mp3:"slower"` | One phone-safe video that plays every take under a big number, with a key file, so the user can answer "3". |
|
|
12
|
+
| `splice.py head.wav tail.wav --head-end 2.68 --tail-start 0.05 --out j.wav` | Join two takes at the quietest point near the cut, at a zero crossing, with a 10 ms crossfade; reports levels, pitch either side and the exact length. |
|
|
13
|
+
| `place_line.py locate track.wav line.mp3 --near 5.9` | Where is this line inside the assembled voice track? Often in several pieces at different offsets. |
|
|
14
|
+
| `place_line.py replace track.wav new.wav --at 8.485 --until 10.45 --from 2.705 --match-old --out t.wav` | Overwrite only that stretch; every other sample stays identical. Reports the margin to the next line. |
|
|
15
|
+
| `onsets.py voice.wav --from 8.4 --to 9.3` | Where does the word really start (rise out of a pause, fricative onset, release after a stop)? For placing a caption within a frame. |
|
|
16
|
+
|
|
17
|
+
## The order that worked
|
|
18
|
+
|
|
19
|
+
1. `credits.py` before any generation. Stop and say so if the balance is short.
|
|
20
|
+
2. Generate two or three spellings, two takes each (pointed, unpointed, hyphenated: pointing a whole sentence can change the neighbouring words).
|
|
21
|
+
3. `phonemes.py` and `pitch_check.py` on every take, on the clean voice. Drop the ones that fail; note duration (a stretched or pitched take is too long for its phonemes).
|
|
22
|
+
4. `ab_video.py` with the take in use first. Send the video and the key; wait for a number.
|
|
23
|
+
5. `splice.py` if only part of a take is wanted. Cut inside a pause, a stop closure or just before a fricative.
|
|
24
|
+
6. `place_line.py locate` then `replace`: change only the stretch that changed. Check the margin to the next line; if it is negative, trim a pause inside the take.
|
|
25
|
+
7. `onsets.py` on the new track for every caption near the change; move a caption only if its word's onset moved, and by the same number of frames.
|
|
26
|
+
8. `reelkit diff old.mp4 new.mp4 --allow <frames> --allow-audio <seconds>s-<seconds>s` to prove nothing else moved.
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""Shared audio helpers for the voice tools. Needs ffmpeg and numpy; nothing here talks to a network."""
|
|
2
|
+
import subprocess
|
|
3
|
+
import sys
|
|
4
|
+
try:
|
|
5
|
+
import numpy as np
|
|
6
|
+
except ImportError:
|
|
7
|
+
print("These voice tools need numpy. In a virtual environment: pip install numpy", file=sys.stderr)
|
|
8
|
+
raise SystemExit(2)
|
|
9
|
+
|
|
10
|
+
SR = 48000
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def load(path, sr=SR):
|
|
14
|
+
"""Any audio file as mono float32 at `sr`."""
|
|
15
|
+
raw = subprocess.run(["ffmpeg", "-nostdin", "-v", "error", "-i", path, "-vn", "-ac", "1", "-ar", str(sr), "-f", "f32le", "-"], capture_output=True, check=True).stdout
|
|
16
|
+
return np.frombuffer(raw, np.float32).astype(np.float64)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def save(path, x, sr=SR):
|
|
20
|
+
"""Mono float samples to a 16-bit wav (or anything ffmpeg writes, by extension)."""
|
|
21
|
+
pcm = np.clip(np.round(np.asarray(x) * 32767), -32768, 32767).astype("<i2").tobytes()
|
|
22
|
+
extra = ["-c:a", "libmp3lame", "-b:a", "192k"] if path.lower().endswith(".mp3") else []
|
|
23
|
+
subprocess.run(["ffmpeg", "-nostdin", "-y", "-v", "error", "-f", "s16le", "-ar", str(sr), "-ac", "1", "-i", "-", *extra, path], input=pcm, check=True)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def rms(x):
|
|
27
|
+
x = np.asarray(x, dtype=np.float64)
|
|
28
|
+
return float(np.sqrt(np.mean(x * x))) if len(x) else 0.0
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def db(v):
|
|
32
|
+
return 20 * np.log10(max(v, 1e-9))
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def envelope_db(x, sr=SR, win=0.01):
|
|
36
|
+
"""Level of each `win`-second window in dB."""
|
|
37
|
+
n = max(1, int(sr * win))
|
|
38
|
+
k = len(x) // n
|
|
39
|
+
if k == 0:
|
|
40
|
+
return np.array([])
|
|
41
|
+
frames = np.asarray(x[: k * n]).reshape(k, n)
|
|
42
|
+
return 20 * np.log10(np.sqrt((frames ** 2).mean(axis=1)) + 1e-9)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def nearest_zero_crossing(x, at, sr=SR, within=0.004):
|
|
46
|
+
"""The sample nearest `at` (a sample index) where the signal crosses zero, searched `within` seconds either side."""
|
|
47
|
+
r = int(sr * within)
|
|
48
|
+
lo, hi = max(1, at - r), min(len(x) - 1, at + r)
|
|
49
|
+
seg = x[lo - 1: hi + 1]
|
|
50
|
+
idx = np.nonzero(np.signbit(seg[:-1]) != np.signbit(seg[1:]))[0]
|
|
51
|
+
if len(idx) == 0:
|
|
52
|
+
return at
|
|
53
|
+
cand = idx + lo
|
|
54
|
+
return int(cand[np.argmin(np.abs(cand - at))])
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def quietest(x, a, b, sr=SR, win=0.006):
|
|
58
|
+
"""The centre (sample index) of the quietest `win` seconds between samples a and b: where a pause or a stop closure is deepest."""
|
|
59
|
+
n = max(1, int(sr * win))
|
|
60
|
+
best, where = None, (a + b) // 2
|
|
61
|
+
for s in range(a, max(a + 1, b - n), max(1, n // 4)):
|
|
62
|
+
v = rms(x[s: s + n])
|
|
63
|
+
if best is None or v < best:
|
|
64
|
+
best, where = v, s + n // 2
|
|
65
|
+
return where
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def speech_rms(x, sr=SR, win=0.01):
|
|
69
|
+
"""RMS over the parts that are speech: the 10 ms windows within 20 dB of the loudest. Pauses inside the stretch do not pull the level down."""
|
|
70
|
+
n = max(1, int(sr * win))
|
|
71
|
+
k = len(x) // n
|
|
72
|
+
if k == 0:
|
|
73
|
+
return rms(x)
|
|
74
|
+
power = (np.asarray(x[: k * n]).reshape(k, n) ** 2).mean(axis=1)
|
|
75
|
+
keep = power >= power.max() * 0.01
|
|
76
|
+
return float(np.sqrt(power[keep].mean())) if keep.any() else 0.0
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def f0_track(x, sr=SR, lo=70.0, hi=400.0):
|
|
80
|
+
"""Pitch every 10 ms: (times, Hz) of the voiced frames. Uses Praat (the `praat-parselmouth` package) when it is installed, which is the
|
|
81
|
+
reliable way. Without it, a plain autocorrelation over 40 ms frames is used: the peak is looked for only after the autocorrelation has first
|
|
82
|
+
fallen below zero, hissing frames (many zero crossings) are skipped, and single-frame outliers are removed by a median of five."""
|
|
83
|
+
try:
|
|
84
|
+
import parselmouth
|
|
85
|
+
p = parselmouth.Sound(np.asarray(x, dtype=np.float64), sr).to_pitch(time_step=0.01, pitch_floor=lo, pitch_ceiling=hi)
|
|
86
|
+
f = p.selected_array["frequency"]
|
|
87
|
+
m = f > 0
|
|
88
|
+
return p.xs()[m], f[m]
|
|
89
|
+
except ImportError:
|
|
90
|
+
pass
|
|
91
|
+
n, hop = int(sr * 0.04), int(sr * 0.01)
|
|
92
|
+
ts, fs = [], []
|
|
93
|
+
peak = float(np.abs(x).max()) if len(x) else 0.0
|
|
94
|
+
for s in range(0, len(x) - n, hop):
|
|
95
|
+
f = np.asarray(x[s: s + n])
|
|
96
|
+
if rms(f) < max(0.004, peak * 0.05):
|
|
97
|
+
continue
|
|
98
|
+
if np.count_nonzero(np.signbit(f[:-1]) != np.signbit(f[1:])) / 0.04 > 3000:
|
|
99
|
+
continue
|
|
100
|
+
f = (f - f.mean()) * np.hanning(n)
|
|
101
|
+
ac = np.correlate(f, f, "full")[n - 1:]
|
|
102
|
+
neg = np.nonzero(ac < 0)[0]
|
|
103
|
+
a, b = max(int(sr / hi), int(neg[0]) if len(neg) else n), min(int(sr / lo), n - 1)
|
|
104
|
+
if ac[0] <= 0 or b <= a + 2:
|
|
105
|
+
continue
|
|
106
|
+
k = a + int(np.argmax(ac[a:b]))
|
|
107
|
+
if k > a and ac[k] > 0.55 * ac[0]:
|
|
108
|
+
ts.append(s / sr); fs.append(sr / k)
|
|
109
|
+
fs = np.array(fs)
|
|
110
|
+
if len(fs) >= 5:
|
|
111
|
+
fs = np.array([np.median(fs[max(0, i - 2): i + 3]) for i in range(len(fs))])
|
|
112
|
+
return np.array(ts), fs
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def f0_median(x, sr=SR, lo=70.0, hi=400.0):
|
|
116
|
+
"""Median pitch of a stretch; None when it is not voiced. Good enough to compare the two sides of a join."""
|
|
117
|
+
_, f = f0_track(x, sr, lo, hi)
|
|
118
|
+
return round(float(np.median(f)), 1) if len(f) >= 3 else None
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Build one video that plays several takes of a line one after another, each under a big number, so that someone can listen on a phone and answer "3".
|
|
3
|
+
|
|
4
|
+
ab_video.py --title "the line as written" --out options.mp4 take1.mp3:"current take" take2.mp3:"slower" take3.mp3:"new spelling"
|
|
5
|
+
|
|
6
|
+
Each take plays over a card with its number, the title and its description; 0.7 s of black and silence separates them. A key file is written beside the
|
|
7
|
+
video (options.key.json) listing number, file, length and description, so the answer "3" maps back to a file without guessing. Put the take in use
|
|
8
|
+
FIRST and say so in its description: the listener needs the reference.
|
|
9
|
+
The video is encoded so that a phone plays it: 1280x720 H.264 High level 4.0, TV-range yuv420p tagged BT.709, AAC 48 kHz, index at the front.
|
|
10
|
+
Needs ffmpeg and Pillow. A title in a right-to-left script needs Pillow built with libraqm; without it the title is drawn as given.
|
|
11
|
+
"""
|
|
12
|
+
import argparse, json, os, subprocess, sys, tempfile
|
|
13
|
+
try:
|
|
14
|
+
from PIL import Image, ImageDraw, ImageFont, features
|
|
15
|
+
except ImportError:
|
|
16
|
+
Image = ImageDraw = ImageFont = features = None
|
|
17
|
+
|
|
18
|
+
FONTS = ["/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf", "/System/Library/Fonts/Supplemental/Arial Bold.ttf", "/Library/Fonts/Arial Bold.ttf", "C:/Windows/Fonts/arialbd.ttf"]
|
|
19
|
+
ENC = ["-c:v", "libx264", "-preset", "medium", "-profile:v", "high", "-level:v", "4.0", "-pix_fmt", "yuv420p", "-r", "30", "-color_range", "tv", "-colorspace", "bt709", "-color_trc", "bt709", "-color_primaries", "bt709",
|
|
20
|
+
"-x264-params", "colorprim=bt709:transfer=bt709:colormatrix=bt709:range=tv", "-c:a", "aac", "-b:a", "160k", "-ar", "48000", "-ac", "2"]
|
|
21
|
+
TAG = "format=yuv420p,setparams=range=tv:color_primaries=bt709:color_trc=bt709:colorspace=bt709"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def ff(*args):
|
|
25
|
+
subprocess.run(["ffmpeg", "-nostdin", "-y", "-v", "error", *args], check=True)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def duration(path):
|
|
29
|
+
return float(subprocess.check_output(["ffprobe", "-v", "error", "-show_entries", "format=duration", "-of", "csv=p=0", path]))
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def font(path, size):
|
|
33
|
+
for f in ([path] if path else []) + FONTS:
|
|
34
|
+
if f and os.path.exists(f):
|
|
35
|
+
return ImageFont.truetype(f, size)
|
|
36
|
+
return ImageFont.load_default()
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def card(path, number, title, sub, font_path, rtl):
|
|
40
|
+
im = Image.new("RGB", (1280, 720), "#101010")
|
|
41
|
+
d = ImageDraw.Draw(im)
|
|
42
|
+
if number is not None:
|
|
43
|
+
d.text((640, 290), str(number), font=font(font_path, 330), fill="#F2E9D8", anchor="mm")
|
|
44
|
+
kw = {"direction": "rtl"} if rtl and features.check("raqm") else {}
|
|
45
|
+
d.text((640, 540), title, font=font(font_path, 46), fill="#E23B2E", anchor="mm", **kw)
|
|
46
|
+
d.text((640, 620), sub[:90], font=font(font_path, 28), fill="#9a9a9a", anchor="mm")
|
|
47
|
+
else:
|
|
48
|
+
d.rectangle([0, 0, 1280, 720], fill="#000000")
|
|
49
|
+
im.save(path)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def build(title, items, out, gap=0.7, font_path=None, rtl=False):
|
|
53
|
+
tmp = tempfile.mkdtemp(prefix="ab_")
|
|
54
|
+
card(f"{tmp}/gap.png", None, "", "", font_path, rtl)
|
|
55
|
+
gap_file = f"{tmp}/gap.mp4"
|
|
56
|
+
ff("-loop", "1", "-framerate", "30", "-i", f"{tmp}/gap.png", "-f", "lavfi", "-i", "anullsrc=r=48000:cl=stereo", "-t", str(gap), "-vf", TAG, *ENC, gap_file)
|
|
57
|
+
parts, key = [], []
|
|
58
|
+
for n, (path, desc) in enumerate(items, 1):
|
|
59
|
+
d = duration(path)
|
|
60
|
+
card(f"{tmp}/c{n}.png", n, title, desc, font_path, rtl)
|
|
61
|
+
seg = f"{tmp}/s{n}.mp4"
|
|
62
|
+
ff("-loop", "1", "-framerate", "30", "-i", f"{tmp}/c{n}.png", "-i", path, "-af", "apad=pad_dur=0.15", "-t", f"{d + 0.15:.3f}", "-vf", TAG, *ENC, seg)
|
|
63
|
+
parts += [gap_file, seg]
|
|
64
|
+
key.append({"n": n, "file": os.path.abspath(path), "dur": round(d, 2), "desc": desc})
|
|
65
|
+
parts.append(gap_file)
|
|
66
|
+
with open(f"{tmp}/list.txt", "w") as f:
|
|
67
|
+
f.write("".join(f"file '{p}'\n" for p in parts))
|
|
68
|
+
# The pieces share one encoder setting, so they are joined once more through the encoder only to get one clean timeline; the range is not converted again.
|
|
69
|
+
ff("-f", "concat", "-safe", "0", "-i", f"{tmp}/list.txt", "-vf", TAG, *ENC, "-crf", "22", "-movflags", "+faststart", out)
|
|
70
|
+
key_path = os.path.splitext(out)[0] + ".key.json"
|
|
71
|
+
with open(key_path, "w", encoding="utf8") as f:
|
|
72
|
+
json.dump(key, f, ensure_ascii=False, indent=1)
|
|
73
|
+
os.sync()
|
|
74
|
+
errors = subprocess.run(["ffmpeg", "-nostdin", "-v", "error", "-i", out, "-f", "null", "-"], capture_output=True, text=True).stderr.strip()
|
|
75
|
+
return {"video": out, "key": key_path, "seconds": round(duration(out), 2), "takes": len(items), "decodes_cleanly": errors == ""}
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
if __name__ == "__main__":
|
|
79
|
+
p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
80
|
+
p.add_argument("takes", nargs="+", help='file:"description", in the order they should play; the take in use first')
|
|
81
|
+
p.add_argument("--title", required=True); p.add_argument("--out", required=True); p.add_argument("--gap", type=float, default=0.7)
|
|
82
|
+
p.add_argument("--font", help="a TrueType font that has the title's script"); p.add_argument("--rtl", action="store_true", help="the title is right to left")
|
|
83
|
+
a = p.parse_args()
|
|
84
|
+
if Image is None:
|
|
85
|
+
print("options video needs Pillow. In a virtual environment: pip install pillow", file=sys.stderr)
|
|
86
|
+
sys.exit(2)
|
|
87
|
+
items = []
|
|
88
|
+
for t in a.takes:
|
|
89
|
+
path, _, desc = t.partition(":")
|
|
90
|
+
if not os.path.exists(path):
|
|
91
|
+
sys.exit(f"{path} does not exist")
|
|
92
|
+
items.append((path, desc or os.path.basename(path)))
|
|
93
|
+
json.dump(build(a.title, items, a.out, a.gap, a.font, a.rtl), sys.stdout, indent=1); print()
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Before generating speech: what it will cost, and whether the account can pay for it.
|
|
3
|
+
|
|
4
|
+
credits.py --takes 6 --chars 28 # an estimate only; nothing is sent anywhere
|
|
5
|
+
credits.py --takes 6 --chars 28 --check # also asks the provider how many credits are left
|
|
6
|
+
|
|
7
|
+
The estimate is takes x characters x --rate (credits per character; 2 fits the short expressive takes of this project, about 57 credits for a
|
|
8
|
+
28-character line). With --check the remaining credits are read from ElevenLabs (GET /v1/user/subscription) using the key in the environment variable
|
|
9
|
+
ELEVENLABS_API_KEY. The key is never printed, logged or written. Exit code 1 when the estimate is more than what is left: stop and tell the user
|
|
10
|
+
rather than generating half a set.
|
|
11
|
+
"""
|
|
12
|
+
import argparse, json, os, sys, urllib.request
|
|
13
|
+
|
|
14
|
+
if __name__ == "__main__":
|
|
15
|
+
p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
16
|
+
p.add_argument("--takes", type=int, required=True); p.add_argument("--chars", type=int, required=True); p.add_argument("--rate", type=float, default=2.0)
|
|
17
|
+
p.add_argument("--check", action="store_true")
|
|
18
|
+
a = p.parse_args()
|
|
19
|
+
need = round(a.takes * a.chars * a.rate)
|
|
20
|
+
out = {"takes": a.takes, "chars_each": a.chars, "estimated_credits": need}
|
|
21
|
+
if a.check:
|
|
22
|
+
key = os.environ.get("ELEVENLABS_API_KEY")
|
|
23
|
+
if not key:
|
|
24
|
+
out["remaining"] = None; out["note"] = "ELEVENLABS_API_KEY is not set: the balance was not checked"
|
|
25
|
+
else:
|
|
26
|
+
req = urllib.request.Request("https://api.elevenlabs.io/v1/user/subscription", headers={"xi-api-key": key})
|
|
27
|
+
try:
|
|
28
|
+
with urllib.request.urlopen(req, timeout=20) as r:
|
|
29
|
+
sub = json.load(r)
|
|
30
|
+
left = int(sub.get("character_limit", 0)) - int(sub.get("character_count", 0))
|
|
31
|
+
out["remaining"] = left; out["enough"] = left >= need
|
|
32
|
+
except Exception as e: # the message of a failed request must never carry the key
|
|
33
|
+
out["remaining"] = None; out["note"] = f"the balance could not be read ({type(e).__name__})"
|
|
34
|
+
print(json.dumps(out))
|
|
35
|
+
sys.exit(1 if out.get("enough") is False else 0)
|