reelkit-cli 0.10.6 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,70 +2,114 @@
2
2
  import { execFile } from "node:child_process";
3
3
  import { cp, mkdir, mkdtemp, rm, writeFile, readFile } from "node:fs/promises";
4
4
  import { tmpdir } from "node:os";
5
- import { join, extname } from "node:path";
5
+ import { dirname, join, extname } from "node:path";
6
6
  import { promisify } from "node:util";
7
7
  import { createCaptureSession, initializeSession, captureFrame, closeCaptureSession, encodeFramesFromDir, processCompositionAudio, muxVideoWithAudio, applyFaststart, type AudioElement } from "@hyperframes/engine";
8
8
  import { serveComposition } from "./serve";
9
9
  import type { VideoProps } from "../hyperframes/types";
10
10
  import type { AudioRecord } from "../hyperframes/frame";
11
- import type { RenderOptions } from "./render";
11
+ import type { FrameRange, RenderOptions } from "./render";
12
12
 
13
13
  const run = promisify(execFile);
14
- const input = JSON.parse(await readFile(process.argv[2]!, "utf8")) as { dir: string; props: VideoProps; mode: "stills" | "video"; output: string; frames: number[]; options?: RenderOptions; chromePath?: string };
15
- const m = input.props.manifest, fps = { num: m.fps, den: 1 };
16
- const work = await mkdtemp(join(tmpdir(), "reelkit-capture-"));
17
- const server = await serveComposition(input.dir);
18
- let session: Awaited<ReturnType<typeof createCaptureSession>> | undefined;
19
- try {
20
- session = await createCaptureSession(server.url.replace(/\/$/, ""), work, { width: m.width, height: m.height, fps, format: "png", compositionDurationSeconds: m.totalFrames / m.fps }, async (page) => {
21
- await page.evaluate(async () => { await (window as unknown as { __reelkitReady: Promise<void> }).__reelkitReady; });
22
- }, { chromePath: input.chromePath, forceScreenshot: true, useDrawElement: false, staticFrameDedup: false, enableBrowserPool: false, browserGpuMode: input.options?.uses3D ? "software" : "auto" });
23
- await initializeSession(session);
24
- await session.page.evaluate(async () => { await (window as unknown as { __reelkitReady: Promise<void> }).__reelkitReady; });
25
- const frames = input.mode === "stills" ? input.frames : Array.from({ length: m.totalFrames }, (_, i) => i);
26
- for (const frame of frames) {
27
- if (!Number.isInteger(frame) || frame < 0 || frame >= m.totalFrames) throw new Error(`Frame ${frame} is outside the composition.`);
28
- const captured = await captureFrame(session, frame, frame / m.fps);
29
- if (input.mode === "stills") await cp(captured.path, join(input.output, `still-${frame}.png`));
14
+ const input = JSON.parse(await readFile(process.argv[2]!, "utf8")) as {
15
+ dir?: string; props?: VideoProps; mode: "stills" | "video" | "mix"; output: string; frames: number[];
16
+ options?: RenderOptions; chromePath?: string; range?: FrameRange; audioSidecar?: string; records?: AudioRecord[]; durationSec?: number;
17
+ };
18
+
19
+ async function mixRecords(records: AudioRecord[], output: string, durationSec: number, work: string): Promise<boolean> {
20
+ if (!records.length) return false;
21
+ const audio: AudioElement[] = [];
22
+ const localSources = new Map<string, string>();
23
+ for (const record of records) {
24
+ let src = localSources.get(record.src);
25
+ if (!src) {
26
+ const response = await fetch(record.src); if (!response.ok) throw new Error(`Audio source returned ${response.status}.`);
27
+ src = join(work, `source-${localSources.size}${extname(new URL(record.src).pathname) || ".media"}`);
28
+ await writeFile(src, new Uint8Array(await response.arrayBuffer())); localSources.set(record.src, src);
29
+ }
30
+ const info = JSON.parse((await run("ffprobe", ["-v", "error", "-show_format", "-of", "json", src])).stdout) as { format: { duration: string } };
31
+ let end = Math.min(record.end, record.start + Number(info.format.duration) - record.mediaStart);
32
+ if (record.loop) {
33
+ const looped = join(work, `loop-${audio.length}.wav`);
34
+ await run("ffmpeg", ["-nostdin", "-y", "-v", "error", "-stream_loop", "-1", "-i", src, "-t", String(record.end - record.start + record.mediaStart), "-vn", looped]);
35
+ src = looped; end = record.end;
36
+ }
37
+ if (end <= record.start) continue;
38
+ const keys = [...new Map(record.volumeKeyframes.map((k) => [k.time, k])).values()].sort((a, b) => a.time - b.time);
39
+ audio.push({ ...record, src, end, volumeKeyframes: keys.length ? keys : undefined });
40
+ }
41
+ if (!audio.length) return false;
42
+ const mixed = output.endsWith(".m4a") ? output : join(work, "audio.m4a");
43
+ const result = await processCompositionAudio(audio, work, join(work, "audio-work"), mixed, durationSec);
44
+ if (!result.success) throw new Error(result.error ?? "HyperFrames audio mixing failed.");
45
+ if (mixed !== output) await cp(mixed, output);
46
+ return true;
47
+ }
48
+
49
+ if (input.mode === "mix") {
50
+ const work = await mkdtemp(join(tmpdir(), "reelkit-mix-"));
51
+ try {
52
+ await mkdir(dirname(input.output), { recursive: true });
53
+ const wrote = await mixRecords(input.records ?? [], input.output, input.durationSec ?? 0, work);
54
+ if (!wrote) throw new Error("This film has no sound to mix.");
55
+ } finally {
56
+ await rm(work, { recursive: true, force: true });
30
57
  }
31
- if (input.mode === "video") {
32
- const silent = join(work, "silent.mp4");
33
- const encoded = await encodeFramesFromDir(work, "frame_%06d.png", silent, { fps, width: m.width, height: m.height, codec: "h264", pixelFormat: "yuv420p", quality: input.options?.crf ?? 18, preset: "fast" });
34
- if (!encoded.success) throw new Error(encoded.error ?? "HyperFrames encoding failed.");
35
- const records = input.options?.silent ? [] : await session.page.evaluate(() => [...(window as unknown as { __reelkitAudio: Map<string, AudioRecord> }).__reelkitAudio.values()]);
36
- const audio: AudioElement[] = [];
37
- const localSources = new Map<string, string>();
38
- for (const record of records) {
39
- let src = localSources.get(record.src);
40
- if (!src) {
41
- const response = await fetch(record.src); if (!response.ok) throw new Error(`Audio source returned ${response.status}.`);
42
- src = join(work, `source-${localSources.size}${extname(new URL(record.src).pathname) || ".media"}`);
43
- await writeFile(src, new Uint8Array(await response.arrayBuffer())); localSources.set(record.src, src);
58
+ } else {
59
+ const props = input.props!;
60
+ const m = props.manifest, fps = { num: m.fps, den: 1 };
61
+ const work = await mkdtemp(join(tmpdir(), "reelkit-capture-"));
62
+ const server = await serveComposition(input.dir!);
63
+ let session: Awaited<ReturnType<typeof createCaptureSession>> | undefined;
64
+ try {
65
+ session = await createCaptureSession(server.url.replace(/\/$/, ""), work, { width: m.width, height: m.height, fps, format: "png", compositionDurationSeconds: m.totalFrames / m.fps }, async (page) => {
66
+ await page.evaluate(async () => { await (window as unknown as { __reelkitReady: Promise<void> }).__reelkitReady; });
67
+ }, { chromePath: input.chromePath, forceScreenshot: true, useDrawElement: false, staticFrameDedup: false, enableBrowserPool: false, browserGpuMode: input.options?.uses3D ? "software" : "auto" });
68
+ await initializeSession(session);
69
+ await session.page.evaluate(async () => { await (window as unknown as { __reelkitReady: Promise<void> }).__reelkitReady; });
70
+ const from = input.range?.from ?? 0, to = input.range?.to ?? m.totalFrames - 1;
71
+ const frames = input.mode === "stills" ? input.frames : Array.from({ length: to - from + 1 }, (_, i) => from + i);
72
+ for (const frame of frames) {
73
+ if (!Number.isInteger(frame) || frame < 0 || frame >= m.totalFrames) throw new Error(`Frame ${frame} is outside the composition.`);
74
+ const captured = await captureFrame(session, frame, frame / m.fps);
75
+ if (input.mode === "stills") await cp(captured.path, join(input.output, `still-${frame}.png`));
76
+ }
77
+ if (input.mode === "video") {
78
+ // The encoder reads frame_000000.png onward. A piece that starts later is renamed into its own folder.
79
+ let framesDir = work, pattern = "frame_%06d.png";
80
+ if (from !== 0) {
81
+ framesDir = join(work, "seq"); await mkdir(framesDir);
82
+ for (let i = 0; i < frames.length; i++) await cp(join(work, `frame_${String(from + i).padStart(6, "0")}.png`), join(framesDir, `frame_${String(i).padStart(6, "0")}.png`));
44
83
  }
45
- const info = JSON.parse((await run("ffprobe", ["-v", "error", "-show_format", "-of", "json", src])).stdout) as { format: { duration: string } };
46
- let end = Math.min(record.end, record.start + Number(info.format.duration) - record.mediaStart);
47
- if (record.loop) {
48
- const looped = join(work, `loop-${audio.length}.wav`);
49
- await run("ffmpeg", ["-y", "-v", "error", "-stream_loop", "-1", "-i", src, "-t", String(record.end - record.start + record.mediaStart), "-vn", looped]);
50
- src = looped; end = record.end;
84
+ const silent = join(work, "silent.mp4");
85
+ const encoded = await encodeFramesFromDir(framesDir, pattern, silent, { fps, width: m.width, height: m.height, codec: "h264", pixelFormat: "yuv420p", quality: input.options?.crf ?? 18, preset: "fast" });
86
+ if (!encoded.success) throw new Error(encoded.error ?? "HyperFrames encoding failed.");
87
+ const collect = Boolean(input.audioSidecar) || !input.options?.silent;
88
+ const records: AudioRecord[] = collect ? await session.page.evaluate(() => [...(window as unknown as { __reelkitAudio: Map<string, AudioRecord> }).__reelkitAudio.values()]) : [];
89
+ if (input.audioSidecar) await writeFile(input.audioSidecar, JSON.stringify(records));
90
+ if (input.options?.silent || input.range) {
91
+ await mkdir(join(work, "faststart"));
92
+ const result = await applyFaststart(silent, input.output);
93
+ if (!result.success) throw new Error(result.error ?? "Faststart failed.");
94
+ } else if (records.length) {
95
+ const mixed = join(work, "audio.m4a");
96
+ const wrote = await mixRecords(records, mixed, m.totalFrames / m.fps, work);
97
+ if (!wrote) {
98
+ await mkdir(join(work, "faststart"));
99
+ const result = await applyFaststart(silent, input.output);
100
+ if (!result.success) throw new Error(result.error ?? "Faststart failed.");
101
+ } else {
102
+ const muxed = await muxVideoWithAudio(silent, mixed, input.output, undefined, { audioCodec: "aac" });
103
+ if (!muxed.success) throw new Error(muxed.error ?? "HyperFrames audio mux failed.");
104
+ }
105
+ } else {
106
+ await mkdir(join(work, "faststart"));
107
+ const result = await applyFaststart(silent, input.output);
108
+ if (!result.success) throw new Error(result.error ?? "Faststart failed.");
51
109
  }
52
- if (end <= record.start) continue;
53
- const keys = [...new Map(record.volumeKeyframes.map((k) => [k.time, k])).values()].sort((a, b) => a.time - b.time);
54
- audio.push({ ...record, src, end, volumeKeyframes: keys.length ? keys : undefined });
55
- }
56
- if (audio.length) {
57
- const mixed = join(work, "audio.m4a");
58
- const result = await processCompositionAudio(audio, work, join(work, "audio-work"), mixed, m.totalFrames / m.fps);
59
- if (!result.success) throw new Error(result.error ?? "HyperFrames audio mixing failed.");
60
- const muxed = await muxVideoWithAudio(silent, mixed, input.output, undefined, { audioCodec: "aac" });
61
- if (!muxed.success) throw new Error(muxed.error ?? "HyperFrames audio mux failed.");
62
- } else {
63
- await mkdir(join(work, "faststart"));
64
- const result = await applyFaststart(silent, input.output);
65
- if (!result.success) throw new Error(result.error ?? "Faststart failed.");
66
110
  }
111
+ } finally {
112
+ if (session) await closeCaptureSession(session);
113
+ await server.close(); await rm(work, { recursive: true, force: true });
67
114
  }
68
- } finally {
69
- if (session) await closeCaptureSession(session);
70
- await server.close(); await rm(work, { recursive: true, force: true });
71
115
  }
@@ -5,7 +5,7 @@ import { createServer } from "node:http";
5
5
  import type { AddressInfo } from "node:net";
6
6
  import { tmpdir } from "node:os";
7
7
  import { join } from "node:path";
8
- import { MAX_UPLOAD_BYTES, routes, type ErrorCode, type LibraryItem, type RouteName, type Voice } from "../contract";
8
+ import { MAX_TEMPLATES, MAX_UPLOAD_BYTES, TEMPLATE_BUNDLE_TYPE, routes, type ErrorCode, type LibraryItem, type RouteName, type Template, type TemplateFilm, type Voice } from "../contract";
9
9
  import { staticCheck } from "../render/validate";
10
10
  import { FAKE_SVG, PIXEL_PNG, silentWav } from "./fixtures";
11
11
 
@@ -124,6 +124,9 @@ export async function startFakeApi(opts: { voiceoverCharLimit?: number; imageLim
124
124
  const secret = () => randomBytes(16).toString("hex");
125
125
  let origin = "";
126
126
  // Every call makes a new URL for its own blob. The URL says nothing about the item behind it.
127
+ // A user's templates: the files as they were uploaded, and whether the template was committed.
128
+ type TemplateJob = { owner: string; record: Template; film: TemplateFilm; committed: boolean; bundle?: Uint8Array; video?: Uint8Array; sources?: Uint8Array; thumbs: (Uint8Array | undefined)[] };
129
+ const templates = new Map<string, TemplateJob>();
127
130
  const fileUrl = (blob: Blob) => { const t = secret(); blobs.set(t, blob); return `${origin}/files/${t}`; };
128
131
  const words = (s: string) => new Set(s.toLowerCase().match(/[a-z0-9]+/g) ?? []);
129
132
 
@@ -323,6 +326,51 @@ export async function startFakeApi(opts: { voiceoverCharLimit?: number; imageLim
323
326
  };
324
327
  return job.result;
325
328
  },
329
+ templateStart: (input, { userId }) => {
330
+ const owner = userId ?? "u-test";
331
+ if ([...templates.values()].filter((t) => t.owner === owner && t.committed).length >= MAX_TEMPLATES) throw new Fail(400, "invalid_request", `You already keep ${MAX_TEMPLATES} templates. Delete one with \`reelkit template delete <id>\` and push again.`);
332
+ const id = `tpl-${randomBytes(6).toString("hex")}`;
333
+ const film = input.film;
334
+ const record: Template = {
335
+ id, name: input.name, ...(input.title ? { title: input.title } : {}), createdAt: new Date().toISOString(), bundleBytes: input.bundleBytes,
336
+ ...(input.videoBytes ? { videoBytes: input.videoBytes } : {}), aspect: film.aspect, durationSec: Math.round((film.totalFrames / film.fps) * 100) / 100,
337
+ scenes: film.scenes.length, thumbs: input.thumbBytes.length, ...(input.from ? { from: input.from } : {}),
338
+ };
339
+ const job: TemplateJob = { owner, record, film, committed: false, thumbs: (input.thumbBytes as number[]).map(() => undefined) };
340
+ templates.set(id, job);
341
+ const put = (contentType: string, bytes: number, has: () => boolean, set: (b: Uint8Array) => void) => {
342
+ const token = secret();
343
+ uploads.set(token, { contentType, bytes, used: false, sink: { has, put: set } });
344
+ return `${origin}/upload/${token}`;
345
+ };
346
+ return {
347
+ id, bundleUrl: put(TEMPLATE_BUNDLE_TYPE, input.bundleBytes, () => Boolean(job.bundle), (b) => { job.bundle = b; }),
348
+ ...(input.videoBytes ? { videoUrl: put("video/mp4", input.videoBytes, () => Boolean(job.video), (b) => { job.video = b; }) } : {}),
349
+ ...(input.sourcesBytes ? { sourcesUrl: put("application/json", input.sourcesBytes, () => Boolean(job.sources), (b) => { job.sources = b; }) } : {}),
350
+ thumbUrls: (input.thumbBytes as number[]).map((n, i) => put("image/jpeg", n, () => Boolean(job.thumbs[i]), (b) => { job.thumbs[i] = b; })),
351
+ };
352
+ },
353
+ templateCommit: ({ id }, { userId }) => {
354
+ const job = templates.get(id);
355
+ if (!job || job.owner !== (userId ?? "u-test")) throw new Fail(404, "not_found", "No such template.");
356
+ if (!job.bundle || (job.record.videoBytes && !job.video)) throw new Fail(400, "invalid_request", "The template's files have not been uploaded. Run `reelkit template push` again.");
357
+ job.committed = true;
358
+ return { template: job.record };
359
+ },
360
+ templateList: (_input, { userId }) => ({
361
+ templates: [...templates.values()].filter((t) => t.owner === (userId ?? "u-test") && t.committed).map((t) => t.record).sort((a, b) => b.createdAt.localeCompare(a.createdAt)),
362
+ }),
363
+ templateGet: ({ id }, { userId }) => {
364
+ const job = templates.get(id);
365
+ if (!job || !job.committed || job.owner !== (userId ?? "u-test")) throw new Fail(404, "not_found", "No such template.");
366
+ return { template: job.record, film: job.film, bundleUrl: fileUrl({ filename: "bundle.tgz", contentType: TEMPLATE_BUNDLE_TYPE, bytes: job.bundle! }) };
367
+ },
368
+ templateDelete: ({ id }, { userId }) => {
369
+ const job = templates.get(id);
370
+ if (!job || !job.committed || job.owner !== (userId ?? "u-test")) throw new Fail(404, "not_found", "No such template.");
371
+ templates.delete(id);
372
+ return { deleted: true as const };
373
+ },
326
374
  publicLibrary: ({ q, kind, page, sort }) => {
327
375
  // Published items only. Newest first, or oldest first when asked; with a query, best match first and equal matches newest first.
328
376
  const want = q ? words(q) : undefined;
@@ -0,0 +1,26 @@
1
+ # Voice tools
2
+
3
+ Scripts for getting one narration line right and into a film without disturbing anything else. They are plain Python (numpy, ffmpeg; Pillow for the
4
+ options video; soundfile for `place_line.py replace`). None of them generates speech; only `credits.py --check` talks to a network, and only to read a balance.
5
+
6
+ | script | what it answers |
7
+ |---|---|
8
+ | `credits.py --takes N --chars C [--check]` | What will this batch cost, and can the account pay for it? Run before generating. |
9
+ | `phonemes.py take.wav --want "m e a f j e n i m"` | What does the take actually say? Speech-to-text spells the expected word even when it was mispronounced; phonemes do not. Exit 2 (not a failure) when the model's packages are not installed. |
10
+ | `pitch_check.py take.wav` | Pitch range, largest step, and the pitch over the last 0.6 s. Verdicts on octave jumps and question-like endings need `praat-parselmouth`. |
11
+ | `ab_video.py --title "..." --out options.mp4 a.mp3:"current" b.mp3:"slower"` | One phone-safe video that plays every take under a big number, with a key file, so the user can answer "3". |
12
+ | `splice.py head.wav tail.wav --head-end 2.68 --tail-start 0.05 --out j.wav` | Join two takes at the quietest point near the cut, at a zero crossing, with a 10 ms crossfade; reports levels, pitch either side and the exact length. |
13
+ | `place_line.py locate track.wav line.mp3 --near 5.9` | Where is this line inside the assembled voice track? Often in several pieces at different offsets. |
14
+ | `place_line.py replace track.wav new.wav --at 8.485 --until 10.45 --from 2.705 --match-old --out t.wav` | Overwrite only that stretch; every other sample stays identical. Reports the margin to the next line. |
15
+ | `onsets.py voice.wav --from 8.4 --to 9.3` | Where does the word really start (rise out of a pause, fricative onset, release after a stop)? For placing a caption within a frame. |
16
+
17
+ ## The order that worked
18
+
19
+ 1. `credits.py` before any generation. Stop and say so if the balance is short.
20
+ 2. Generate two or three spellings, two takes each (pointed, unpointed, hyphenated: pointing a whole sentence can change the neighbouring words).
21
+ 3. `phonemes.py` and `pitch_check.py` on every take, on the clean voice. Drop the ones that fail; note duration (a stretched or pitched take is too long for its phonemes).
22
+ 4. `ab_video.py` with the take in use first. Send the video and the key; wait for a number.
23
+ 5. `splice.py` if only part of a take is wanted. Cut inside a pause, a stop closure or just before a fricative.
24
+ 6. `place_line.py locate` then `replace`: change only the stretch that changed. Check the margin to the next line; if it is negative, trim a pause inside the take.
25
+ 7. `onsets.py` on the new track for every caption near the change; move a caption only if its word's onset moved, and by the same number of frames.
26
+ 8. `reelkit diff old.mp4 new.mp4 --allow <frames> --allow-audio <seconds>s-<seconds>s` to prove nothing else moved.
@@ -0,0 +1,118 @@
1
+ """Shared audio helpers for the voice tools. Needs ffmpeg and numpy; nothing here talks to a network."""
2
+ import subprocess
3
+ import sys
4
+ try:
5
+ import numpy as np
6
+ except ImportError:
7
+ print("These voice tools need numpy. In a virtual environment: pip install numpy", file=sys.stderr)
8
+ raise SystemExit(2)
9
+
10
+ SR = 48000
11
+
12
+
13
+ def load(path, sr=SR):
14
+ """Any audio file as mono float32 at `sr`."""
15
+ raw = subprocess.run(["ffmpeg", "-nostdin", "-v", "error", "-i", path, "-vn", "-ac", "1", "-ar", str(sr), "-f", "f32le", "-"], capture_output=True, check=True).stdout
16
+ return np.frombuffer(raw, np.float32).astype(np.float64)
17
+
18
+
19
+ def save(path, x, sr=SR):
20
+ """Mono float samples to a 16-bit wav (or anything ffmpeg writes, by extension)."""
21
+ pcm = np.clip(np.round(np.asarray(x) * 32767), -32768, 32767).astype("<i2").tobytes()
22
+ extra = ["-c:a", "libmp3lame", "-b:a", "192k"] if path.lower().endswith(".mp3") else []
23
+ subprocess.run(["ffmpeg", "-nostdin", "-y", "-v", "error", "-f", "s16le", "-ar", str(sr), "-ac", "1", "-i", "-", *extra, path], input=pcm, check=True)
24
+
25
+
26
+ def rms(x):
27
+ x = np.asarray(x, dtype=np.float64)
28
+ return float(np.sqrt(np.mean(x * x))) if len(x) else 0.0
29
+
30
+
31
+ def db(v):
32
+ return 20 * np.log10(max(v, 1e-9))
33
+
34
+
35
+ def envelope_db(x, sr=SR, win=0.01):
36
+ """Level of each `win`-second window in dB."""
37
+ n = max(1, int(sr * win))
38
+ k = len(x) // n
39
+ if k == 0:
40
+ return np.array([])
41
+ frames = np.asarray(x[: k * n]).reshape(k, n)
42
+ return 20 * np.log10(np.sqrt((frames ** 2).mean(axis=1)) + 1e-9)
43
+
44
+
45
+ def nearest_zero_crossing(x, at, sr=SR, within=0.004):
46
+ """The sample nearest `at` (a sample index) where the signal crosses zero, searched `within` seconds either side."""
47
+ r = int(sr * within)
48
+ lo, hi = max(1, at - r), min(len(x) - 1, at + r)
49
+ seg = x[lo - 1: hi + 1]
50
+ idx = np.nonzero(np.signbit(seg[:-1]) != np.signbit(seg[1:]))[0]
51
+ if len(idx) == 0:
52
+ return at
53
+ cand = idx + lo
54
+ return int(cand[np.argmin(np.abs(cand - at))])
55
+
56
+
57
+ def quietest(x, a, b, sr=SR, win=0.006):
58
+ """The centre (sample index) of the quietest `win` seconds between samples a and b: where a pause or a stop closure is deepest."""
59
+ n = max(1, int(sr * win))
60
+ best, where = None, (a + b) // 2
61
+ for s in range(a, max(a + 1, b - n), max(1, n // 4)):
62
+ v = rms(x[s: s + n])
63
+ if best is None or v < best:
64
+ best, where = v, s + n // 2
65
+ return where
66
+
67
+
68
+ def speech_rms(x, sr=SR, win=0.01):
69
+ """RMS over the parts that are speech: the 10 ms windows within 20 dB of the loudest. Pauses inside the stretch do not pull the level down."""
70
+ n = max(1, int(sr * win))
71
+ k = len(x) // n
72
+ if k == 0:
73
+ return rms(x)
74
+ power = (np.asarray(x[: k * n]).reshape(k, n) ** 2).mean(axis=1)
75
+ keep = power >= power.max() * 0.01
76
+ return float(np.sqrt(power[keep].mean())) if keep.any() else 0.0
77
+
78
+
79
+ def f0_track(x, sr=SR, lo=70.0, hi=400.0):
80
+ """Pitch every 10 ms: (times, Hz) of the voiced frames. Uses Praat (the `praat-parselmouth` package) when it is installed, which is the
81
+ reliable way. Without it, a plain autocorrelation over 40 ms frames is used: the peak is looked for only after the autocorrelation has first
82
+ fallen below zero, hissing frames (many zero crossings) are skipped, and single-frame outliers are removed by a median of five."""
83
+ try:
84
+ import parselmouth
85
+ p = parselmouth.Sound(np.asarray(x, dtype=np.float64), sr).to_pitch(time_step=0.01, pitch_floor=lo, pitch_ceiling=hi)
86
+ f = p.selected_array["frequency"]
87
+ m = f > 0
88
+ return p.xs()[m], f[m]
89
+ except ImportError:
90
+ pass
91
+ n, hop = int(sr * 0.04), int(sr * 0.01)
92
+ ts, fs = [], []
93
+ peak = float(np.abs(x).max()) if len(x) else 0.0
94
+ for s in range(0, len(x) - n, hop):
95
+ f = np.asarray(x[s: s + n])
96
+ if rms(f) < max(0.004, peak * 0.05):
97
+ continue
98
+ if np.count_nonzero(np.signbit(f[:-1]) != np.signbit(f[1:])) / 0.04 > 3000:
99
+ continue
100
+ f = (f - f.mean()) * np.hanning(n)
101
+ ac = np.correlate(f, f, "full")[n - 1:]
102
+ neg = np.nonzero(ac < 0)[0]
103
+ a, b = max(int(sr / hi), int(neg[0]) if len(neg) else n), min(int(sr / lo), n - 1)
104
+ if ac[0] <= 0 or b <= a + 2:
105
+ continue
106
+ k = a + int(np.argmax(ac[a:b]))
107
+ if k > a and ac[k] > 0.55 * ac[0]:
108
+ ts.append(s / sr); fs.append(sr / k)
109
+ fs = np.array(fs)
110
+ if len(fs) >= 5:
111
+ fs = np.array([np.median(fs[max(0, i - 2): i + 3]) for i in range(len(fs))])
112
+ return np.array(ts), fs
113
+
114
+
115
+ def f0_median(x, sr=SR, lo=70.0, hi=400.0):
116
+ """Median pitch of a stretch; None when it is not voiced. Good enough to compare the two sides of a join."""
117
+ _, f = f0_track(x, sr, lo, hi)
118
+ return round(float(np.median(f)), 1) if len(f) >= 3 else None
@@ -0,0 +1,93 @@
1
+ #!/usr/bin/env python3
2
+ """Build one video that plays several takes of a line one after another, each under a big number, so that someone can listen on a phone and answer "3".
3
+
4
+ ab_video.py --title "the line as written" --out options.mp4 take1.mp3:"current take" take2.mp3:"slower" take3.mp3:"new spelling"
5
+
6
+ Each take plays over a card with its number, the title and its description; 0.7 s of black and silence separates them. A key file is written beside the
7
+ video (options.key.json) listing number, file, length and description, so the answer "3" maps back to a file without guessing. Put the take in use
8
+ FIRST and say so in its description: the listener needs the reference.
9
+ The video is encoded so that a phone plays it: 1280x720 H.264 High level 4.0, TV-range yuv420p tagged BT.709, AAC 48 kHz, index at the front.
10
+ Needs ffmpeg and Pillow. A title in a right-to-left script needs Pillow built with libraqm; without it the title is drawn as given.
11
+ """
12
+ import argparse, json, os, subprocess, sys, tempfile
13
+ try:
14
+ from PIL import Image, ImageDraw, ImageFont, features
15
+ except ImportError:
16
+ Image = ImageDraw = ImageFont = features = None
17
+
18
+ FONTS = ["/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf", "/System/Library/Fonts/Supplemental/Arial Bold.ttf", "/Library/Fonts/Arial Bold.ttf", "C:/Windows/Fonts/arialbd.ttf"]
19
+ ENC = ["-c:v", "libx264", "-preset", "medium", "-profile:v", "high", "-level:v", "4.0", "-pix_fmt", "yuv420p", "-r", "30", "-color_range", "tv", "-colorspace", "bt709", "-color_trc", "bt709", "-color_primaries", "bt709",
20
+ "-x264-params", "colorprim=bt709:transfer=bt709:colormatrix=bt709:range=tv", "-c:a", "aac", "-b:a", "160k", "-ar", "48000", "-ac", "2"]
21
+ TAG = "format=yuv420p,setparams=range=tv:color_primaries=bt709:color_trc=bt709:colorspace=bt709"
22
+
23
+
24
+ def ff(*args):
25
+ subprocess.run(["ffmpeg", "-nostdin", "-y", "-v", "error", *args], check=True)
26
+
27
+
28
+ def duration(path):
29
+ return float(subprocess.check_output(["ffprobe", "-v", "error", "-show_entries", "format=duration", "-of", "csv=p=0", path]))
30
+
31
+
32
+ def font(path, size):
33
+ for f in ([path] if path else []) + FONTS:
34
+ if f and os.path.exists(f):
35
+ return ImageFont.truetype(f, size)
36
+ return ImageFont.load_default()
37
+
38
+
39
+ def card(path, number, title, sub, font_path, rtl):
40
+ im = Image.new("RGB", (1280, 720), "#101010")
41
+ d = ImageDraw.Draw(im)
42
+ if number is not None:
43
+ d.text((640, 290), str(number), font=font(font_path, 330), fill="#F2E9D8", anchor="mm")
44
+ kw = {"direction": "rtl"} if rtl and features.check("raqm") else {}
45
+ d.text((640, 540), title, font=font(font_path, 46), fill="#E23B2E", anchor="mm", **kw)
46
+ d.text((640, 620), sub[:90], font=font(font_path, 28), fill="#9a9a9a", anchor="mm")
47
+ else:
48
+ d.rectangle([0, 0, 1280, 720], fill="#000000")
49
+ im.save(path)
50
+
51
+
52
+ def build(title, items, out, gap=0.7, font_path=None, rtl=False):
53
+ tmp = tempfile.mkdtemp(prefix="ab_")
54
+ card(f"{tmp}/gap.png", None, "", "", font_path, rtl)
55
+ gap_file = f"{tmp}/gap.mp4"
56
+ ff("-loop", "1", "-framerate", "30", "-i", f"{tmp}/gap.png", "-f", "lavfi", "-i", "anullsrc=r=48000:cl=stereo", "-t", str(gap), "-vf", TAG, *ENC, gap_file)
57
+ parts, key = [], []
58
+ for n, (path, desc) in enumerate(items, 1):
59
+ d = duration(path)
60
+ card(f"{tmp}/c{n}.png", n, title, desc, font_path, rtl)
61
+ seg = f"{tmp}/s{n}.mp4"
62
+ ff("-loop", "1", "-framerate", "30", "-i", f"{tmp}/c{n}.png", "-i", path, "-af", "apad=pad_dur=0.15", "-t", f"{d + 0.15:.3f}", "-vf", TAG, *ENC, seg)
63
+ parts += [gap_file, seg]
64
+ key.append({"n": n, "file": os.path.abspath(path), "dur": round(d, 2), "desc": desc})
65
+ parts.append(gap_file)
66
+ with open(f"{tmp}/list.txt", "w") as f:
67
+ f.write("".join(f"file '{p}'\n" for p in parts))
68
+ # The pieces share one encoder setting, so they are joined once more through the encoder only to get one clean timeline; the range is not converted again.
69
+ ff("-f", "concat", "-safe", "0", "-i", f"{tmp}/list.txt", "-vf", TAG, *ENC, "-crf", "22", "-movflags", "+faststart", out)
70
+ key_path = os.path.splitext(out)[0] + ".key.json"
71
+ with open(key_path, "w", encoding="utf8") as f:
72
+ json.dump(key, f, ensure_ascii=False, indent=1)
73
+ os.sync()
74
+ errors = subprocess.run(["ffmpeg", "-nostdin", "-v", "error", "-i", out, "-f", "null", "-"], capture_output=True, text=True).stderr.strip()
75
+ return {"video": out, "key": key_path, "seconds": round(duration(out), 2), "takes": len(items), "decodes_cleanly": errors == ""}
76
+
77
+
78
+ if __name__ == "__main__":
79
+ p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
80
+ p.add_argument("takes", nargs="+", help='file:"description", in the order they should play; the take in use first')
81
+ p.add_argument("--title", required=True); p.add_argument("--out", required=True); p.add_argument("--gap", type=float, default=0.7)
82
+ p.add_argument("--font", help="a TrueType font that has the title's script"); p.add_argument("--rtl", action="store_true", help="the title is right to left")
83
+ a = p.parse_args()
84
+ if Image is None:
85
+ print("options video needs Pillow. In a virtual environment: pip install pillow", file=sys.stderr)
86
+ sys.exit(2)
87
+ items = []
88
+ for t in a.takes:
89
+ path, _, desc = t.partition(":")
90
+ if not os.path.exists(path):
91
+ sys.exit(f"{path} does not exist")
92
+ items.append((path, desc or os.path.basename(path)))
93
+ json.dump(build(a.title, items, a.out, a.gap, a.font, a.rtl), sys.stdout, indent=1); print()
@@ -0,0 +1,35 @@
1
+ #!/usr/bin/env python3
2
+ """Before generating speech: what it will cost, and whether the account can pay for it.
3
+
4
+ credits.py --takes 6 --chars 28 # an estimate only; nothing is sent anywhere
5
+ credits.py --takes 6 --chars 28 --check # also asks the provider how many credits are left
6
+
7
+ The estimate is takes x characters x --rate (credits per character; 2 fits the short expressive takes of this project, about 57 credits for a
8
+ 28-character line). With --check the remaining credits are read from ElevenLabs (GET /v1/user/subscription) using the key in the environment variable
9
+ ELEVENLABS_API_KEY. The key is never printed, logged or written. Exit code 1 when the estimate is more than what is left: stop and tell the user
10
+ rather than generating half a set.
11
+ """
12
+ import argparse, json, os, sys, urllib.request
13
+
14
+ if __name__ == "__main__":
15
+ p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
16
+ p.add_argument("--takes", type=int, required=True); p.add_argument("--chars", type=int, required=True); p.add_argument("--rate", type=float, default=2.0)
17
+ p.add_argument("--check", action="store_true")
18
+ a = p.parse_args()
19
+ need = round(a.takes * a.chars * a.rate)
20
+ out = {"takes": a.takes, "chars_each": a.chars, "estimated_credits": need}
21
+ if a.check:
22
+ key = os.environ.get("ELEVENLABS_API_KEY")
23
+ if not key:
24
+ out["remaining"] = None; out["note"] = "ELEVENLABS_API_KEY is not set: the balance was not checked"
25
+ else:
26
+ req = urllib.request.Request("https://api.elevenlabs.io/v1/user/subscription", headers={"xi-api-key": key})
27
+ try:
28
+ with urllib.request.urlopen(req, timeout=20) as r:
29
+ sub = json.load(r)
30
+ left = int(sub.get("character_limit", 0)) - int(sub.get("character_count", 0))
31
+ out["remaining"] = left; out["enough"] = left >= need
32
+ except Exception as e: # the message of a failed request must never carry the key
33
+ out["remaining"] = None; out["note"] = f"the balance could not be read ({type(e).__name__})"
34
+ print(json.dumps(out))
35
+ sys.exit(1 if out.get("enough") is False else 0)
@@ -0,0 +1,70 @@
1
+ #!/usr/bin/env python3
2
+ """Where a word really starts in a recording, to place its caption within a frame.
3
+
4
+ onsets.py voice.wav --from 8.4 --to 9.3 [--fps 30] [--hf 5500]
5
+
6
+ Prints the level of each 10 ms window between --from and --to in two bands (everything, and above --hf Hz), then the moments that look like onsets:
7
+ a rise out of a pause, the start of a fricative (s, sh, ts, f: high-band energy with little low-band energy), and the release after a stop closure
8
+ (a dip of 30 ms or more). Speech-to-text word times are good to about 60 ms and are thrown off by pauses; these are good to about 10 ms.
9
+ Place a caption one or two frames before the onset. Each time is given in seconds and as a frame at --fps.
10
+ """
11
+ import argparse, json, sys
12
+ import numpy as np
13
+ from _audio import SR, envelope_db, load
14
+
15
+
16
+ def bands(x, sr, hf):
17
+ spec = np.fft.rfft(x)
18
+ freqs = np.fft.rfftfreq(len(x), 1 / sr)
19
+ hi = np.fft.irfft(np.where(freqs >= hf, spec, 0), len(x))
20
+ lo = np.fft.irfft(np.where(freqs <= 1500, spec, 0), len(x))
21
+ return hi, lo
22
+
23
+
24
+ def onsets(x, sr=SR, t0=0.0, hf=5500.0, win=0.01):
25
+ hi, lo = bands(x, sr, hf)
26
+ full, e_hi, e_lo = envelope_db(x, sr, win), envelope_db(hi, sr, win), envelope_db(lo, sr, win)
27
+ peak = full.max() if len(full) else 0
28
+ found = []
29
+ for i in range(1, len(full)):
30
+ t = t0 + i * win
31
+ # out of a pause: at least 30 ms more than 30 dB under the peak, then a rise
32
+ if i >= 3 and (full[i - 3:i] < peak - 30).all() and full[i] > peak - 24:
33
+ found.append({"t": round(t, 3), "kind": "rise out of a pause"})
34
+ # a fricative: the high band comes up while the low band stays down
35
+ if e_hi[i] > e_hi.max() - 12 and e_hi[i - 1] < e_hi.max() - 18 and e_lo[i] < e_lo.max() - 12:
36
+ found.append({"t": round(t, 3), "kind": "fricative starts"})
37
+ # release after a stop closure: a dip of 30 ms or more inside speech
38
+ i = 1
39
+ while i < len(full) - 1:
40
+ if full[i] < peak - 26 and full[i - 1] >= peak - 26:
41
+ j = i
42
+ while j < len(full) and full[j] < peak - 26:
43
+ j += 1
44
+ if 3 <= j - i <= 12 and j < len(full):
45
+ found.append({"t": round(t0 + j * win, 3), "kind": f"release after a {int((j - i) * win * 1000)} ms closure"})
46
+ i = j
47
+ else:
48
+ i += 1
49
+ return full, e_hi, e_lo, sorted(found, key=lambda f: f["t"])
50
+
51
+
52
+ if __name__ == "__main__":
53
+ p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
54
+ p.add_argument("file"); p.add_argument("--from", dest="t0", type=float, default=0.0); p.add_argument("--to", dest="t1", type=float)
55
+ p.add_argument("--fps", type=float, default=30.0); p.add_argument("--hf", type=float, default=5500.0); p.add_argument("--json", action="store_true")
56
+ a = p.parse_args()
57
+ x = load(a.file)
58
+ x = x[int(a.t0 * SR): int(a.t1 * SR) if a.t1 else None]
59
+ full, hi, lo, found = onsets(x, SR, a.t0, a.hf)
60
+ for f in found:
61
+ f["frame"] = round(f["t"] * a.fps, 1)
62
+ if a.json:
63
+ json.dump({"onsets": found}, sys.stdout, indent=1); print()
64
+ else:
65
+ print("time all high low (dB per 10 ms)")
66
+ for i in range(len(full)):
67
+ print(f"{a.t0 + i * 0.01:6.2f} {full[i]:5.0f} {hi[i]:5.0f} {lo[i]:5.0f}")
68
+ print("\nonsets:")
69
+ for f in found:
70
+ print(f" {f['t']:.3f} s frame {f['frame']:.1f} {f['kind']}")