@gentbajko/slopify 0.5.1 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/README.md +36 -1
  2. package/SUBTITLES.md +21 -0
  3. package/dist/adapter-registry.js +18 -3
  4. package/dist/adapters/alignment/audio.js +45 -0
  5. package/dist/adapters/alignment/cache.js +91 -0
  6. package/dist/adapters/alignment/ctc.js +128 -0
  7. package/dist/adapters/alignment/index.js +34 -0
  8. package/dist/adapters/alignment/lock.js +58 -0
  9. package/dist/adapters/alignment/numbers.js +91 -0
  10. package/dist/adapters/alignment/protocol.js +23 -0
  11. package/dist/adapters/alignment/quality.js +17 -0
  12. package/dist/adapters/alignment/runner.js +76 -0
  13. package/dist/adapters/alignment/text.js +61 -0
  14. package/dist/adapters/alignment/vocabulary.js +69 -0
  15. package/dist/adapters/alignment/worker.js +163 -0
  16. package/dist/adapters/image/google.js +12 -15
  17. package/dist/adapters/image/models.js +72 -0
  18. package/dist/adapters/image/openai.js +3 -5
  19. package/dist/adapters/llm/catalogue-files.js +23 -0
  20. package/dist/adapters/llm/claude-code.js +9 -7
  21. package/dist/adapters/llm/codex-models.js +46 -0
  22. package/dist/adapters/llm/codex.js +7 -12
  23. package/dist/adapters/llm/gemini-models.js +85 -0
  24. package/dist/adapters/llm/gemini-workspace.js +70 -0
  25. package/dist/adapters/llm/gemini.js +138 -0
  26. package/dist/adapters/llm/openrouter.js +11 -2
  27. package/dist/adapters/llm/run-cli.js +6 -9
  28. package/dist/adapters/tts/cartesia.js +9 -1
  29. package/dist/adapters/tts/elevenlabs.js +33 -2
  30. package/dist/adapters/tts/inworld-async.js +150 -0
  31. package/dist/adapters/tts/inworld-text.js +32 -0
  32. package/dist/adapters/tts/inworld.js +170 -0
  33. package/dist/adapters/tts/openai.js +31 -4
  34. package/dist/assets/fonts/Barlow-Regular.ttf +0 -0
  35. package/dist/assets/fonts/OFL.txt +93 -0
  36. package/dist/assets/fonts/SOURCE.txt +7 -0
  37. package/dist/edge/cli.js +5 -0
  38. package/dist/edge/events/hub.js +9 -2
  39. package/dist/edge/events/preview-cache.js +40 -0
  40. package/dist/edge/http/actions.js +2 -0
  41. package/dist/edge/http/app.js +30 -1
  42. package/dist/edge/http/audio-preview.js +46 -0
  43. package/dist/edge/http/fonts.js +147 -0
  44. package/dist/edge/http/projects.js +23 -1
  45. package/dist/edge/http/providers.js +29 -0
  46. package/dist/edge/http/subtitles.js +92 -0
  47. package/dist/edge/http/update.js +57 -0
  48. package/dist/edge/update-worker.js +44 -0
  49. package/dist/kernel/audio-preview.js +188 -0
  50. package/dist/kernel/cli-command.js +45 -0
  51. package/dist/kernel/ports/subtitles.js +1 -0
  52. package/dist/kernel/runner/providers.js +75 -24
  53. package/dist/main.js +106 -25
  54. package/dist/model-catalog.js +37 -0
  55. package/dist/slices/admission/repo.js +2 -0
  56. package/dist/slices/admission/rules.js +6 -0
  57. package/dist/slices/control/index.js +1 -0
  58. package/dist/slices/control/providers.js +2 -0
  59. package/dist/slices/fonts/catalog.js +95 -0
  60. package/dist/slices/fonts/discovery.js +81 -0
  61. package/dist/slices/fonts/files.js +37 -0
  62. package/dist/slices/fonts/index.js +5 -0
  63. package/dist/slices/fonts/model.js +4 -0
  64. package/dist/slices/fonts/preview.js +14 -0
  65. package/dist/slices/fonts/sfnt.js +262 -0
  66. package/dist/slices/fonts/upload.js +40 -0
  67. package/dist/slices/narration/live.js +21 -0
  68. package/dist/slices/narration/run.js +9 -9
  69. package/dist/slices/research/run.js +3 -0
  70. package/dist/slices/settings/cli-paths.js +98 -0
  71. package/dist/slices/settings/cli-status.js +21 -12
  72. package/dist/slices/settings/model.js +11 -0
  73. package/dist/slices/settings/models.js +89 -0
  74. package/dist/slices/settings/readiness.js +13 -9
  75. package/dist/slices/storage/downloads.js +2 -0
  76. package/dist/slices/storage/layout.js +10 -0
  77. package/dist/slices/storage/model.js +5 -0
  78. package/dist/slices/storage/repo.js +1 -0
  79. package/dist/slices/subtitles/captions.js +91 -0
  80. package/dist/slices/subtitles/layout.js +19 -0
  81. package/dist/slices/subtitles/model.js +27 -0
  82. package/dist/slices/subtitles/prepare.js +167 -0
  83. package/dist/slices/subtitles/transcript.js +34 -0
  84. package/dist/slices/video/audio-export.js +3 -0
  85. package/dist/slices/video/ffmpeg.js +14 -7
  86. package/dist/slices/video/plan.js +4 -4
  87. package/dist/slices/video/run.js +5 -2
  88. package/dist/slices/video/write-export.js +85 -20
  89. package/dist/updater/candidate.js +28 -0
  90. package/dist/updater/forward.js +35 -0
  91. package/dist/updater/install-flow.js +26 -0
  92. package/dist/updater/install.js +50 -0
  93. package/dist/updater/model.js +24 -0
  94. package/dist/updater/plan.js +135 -0
  95. package/dist/updater/readiness.js +15 -0
  96. package/dist/updater/registry.js +19 -0
  97. package/dist/updater/service.js +128 -0
  98. package/dist/updater/worker.js +136 -0
  99. package/dist/web/assets/index-6QIQ-TzO.css +1 -0
  100. package/dist/web/assets/index-CNgWL-ct.js +130 -0
  101. package/dist/web/index.html +2 -2
  102. package/package.json +4 -2
  103. package/dist/web/assets/index-C1fsrIP5.css +0 -1
  104. package/dist/web/assets/index-DwJ_XTl3.js +0 -81
package/README.md CHANGED
@@ -1,4 +1,15 @@
1
- # Slopify
1
+ <h1 align="center">
2
+ <a href="https://slopify.stream"><img src="https://slopify.stream/assets/favicon.svg" width="40" height="40" align="middle" alt="" /></a>
3
+ Slopify
4
+ </h1>
5
+
6
+ <p align="center">
7
+ <a href="https://slopify.stream">slopify.stream</a>
8
+ &nbsp;·&nbsp;
9
+ <a href="https://www.patreon.com/cw/GentBajko"><img src="https://slopify.stream/assets/patreon-green.svg" width="14" height="14" align="middle" alt="" /> Patreon</a>
10
+ &nbsp;·&nbsp;
11
+ <a href="https://buymeacoffee.com/gentbajko"><img src="https://slopify.stream/assets/buymeacoffee-green.svg" width="14" height="14" align="middle" alt="" /> Buy Me a Coffee</a>
12
+ </p>
2
13
 
3
14
  A prompt and a few keywords in. A narrated slideshow video out. Your keys, your machine, free.
4
15
 
@@ -14,6 +25,23 @@ Six stages run as a graph: research, article, narration, images, thumbnail, vide
14
25
  Generate any of them, or provide the output yourself and that stage is skipped. The
15
26
  video is a slideshow with alternating zoom over the narration, rendered with ffmpeg.
16
27
 
28
+ ## Inworld narration
29
+
30
+ In Settings, add the **Base64 credentials** from Inworld's API Keys page, then add an
31
+ Inworld voice ID (for example `Dennis`, or a voice from your workspace). In Play,
32
+ choose **Realtime TTS-2** or **Realtime TTS-2 Flash** and that voice.
33
+
34
+ Short text streams immediately. TTS-2 text over 4,000 characters uses one async job,
35
+ up to 100,000 characters; Inworld caps On-Demand accounts at 10,000. Audio becomes
36
+ available once that job finishes. Flash uses streamed parts of at most 4,000 characters.
37
+ For longer articles, select paragraph chunking. Successful status checks keep long jobs
38
+ alive, and automatic polling/download retries reuse the accepted job. Pausing stops
39
+ local requests; Inworld may still finish and bill an accepted job. Resuming after a
40
+ pause or app restart starts a new request for unfinished narration.
41
+
42
+ See [Inworld's async API](https://docs.inworld.ai/api-reference/ttsAPI/texttospeech/synthesize-speech-async)
43
+ for account limits. Both model IDs are bundled; Inworld's LLM catalogue does not list TTS models.
44
+
17
45
  ## Options
18
46
 
19
47
  | Flag | Environment variable | Default |
@@ -67,6 +95,13 @@ Never your keys, prompts, keywords, titles, article text, filenames, or anything
67
95
  your machine. A notice says all of this the first time you run it, before the machine
68
96
  id exists, and the Usage screen shows you your own numbers at any time.
69
97
 
98
+ ## Supporting the project
99
+
100
+ Slopify is free and always will be. If it is worth something to you, there is
101
+ [Patreon](https://www.patreon.com/cw/GentBajko) and
102
+ [Buy Me a Coffee](https://buymeacoffee.com/gentbajko). The people who do are listed
103
+ in [SUPPORTERS.md](https://github.com/GentBajko/slopify/blob/main/SUPPORTERS.md).
104
+
70
105
  ## Licence
71
106
 
72
107
  MIT. The ffmpeg binary fetched at install time is a separate GPL-3.0-or-later program,
package/SUBTITLES.md ADDED
@@ -0,0 +1,21 @@
1
+ # Subtitles in Slopify
2
+
3
+ 1. In Play, enable narration or provide matching English audio and article text.
4
+ 2. In Subtitles, choose **Subtitle files (.srt + .vtt)** or **Burn into video + files**.
5
+ 3. For burned captions, select a bundled or system font, or upload a `.ttf` or `.otf` file (up to 32 MiB). Choose a size from 16 to 120 and a position: top, upper-middle, center, lower-middle or bottom. The side preview shows the selected font, size and position in a 16:9 or 9:16 frame.
6
+ 4. Start the run. Slopify times the article against the actual narration on your computer. The first use downloads an approximately 95 MB English speech model; later runs work offline with the cached model. Allow roughly 1 GB of available memory during alignment. No extra API key, Python, or compiler is needed.
7
+ 5. Download SRT/VTT beside the final export. Files mode adds an optional native caption track to the in-app video preview. Burned captions remain visible in the downloaded MP4.
8
+
9
+ For a completed project, open its final Video or Audio export section, set subtitles, and click **Save subtitles**. This rebuilds only the local export from saved narration and images. Changing font, size or position reuses word timing when the audio and spoken text are unchanged. A paused run saves these choices until Resume; pause an active run before editing. Failed alignment or rendering keeps the previous finished export.
10
+
11
+ Audio Off disables subtitles. Video Off produces WAV audio with separate subtitle files. Uploaded audio must match the article; substantial mismatches fail with a message to correct the transcript. Review timing and spelling before publishing. English is supported first; unusual pronunciations and non-English passages can fail alignment.
12
+
13
+ Fonts are copied into the completed project's caption assets, so an existing export can reuse its chosen font even after the original system font is removed. Installed fonts and uploaded fonts stay local. SRT/VTT are portable text/timing files and do not embed a font; the chosen font is used in burned video captions.
14
+
15
+ ## Local model and licenses
16
+
17
+ Speech inference uses MIT-licensed `onnxruntime-web@1.24.3` in a separate WASM process. The runtime is installed with the package (about 138 MB on disk); model weights are downloaded lazily into `<data-dir>/models/english-subtitles/`.
18
+
19
+ Model: [Xenova/wav2vec2-base-960h](https://huggingface.co/Xenova/wav2vec2-base-960h/tree/a19f851b3d42865797e410752b4c570c871e4825), an ONNX conversion of [facebook/wav2vec2-base-960h](https://huggingface.co/facebook/wav2vec2-base-960h), licensed Apache-2.0. Pinned revision: `a19f851b3d42865797e410752b4c570c871e4825`. The quantized model is 95,286,046 bytes, SHA256 `cd5040c147381580ed73258143dd8e0c28e800a09e74ee42ee2b3e8cb4d760a3`; every cached/downloaded model is verified before use.
20
+
21
+ The bundled Barlow font uses the SIL Open Font License; its license and source record ship in `dist/assets/fonts/`. Uploaded/system font licensing remains with its author.
@@ -4,10 +4,15 @@ import { openAiImage } from "./adapters/image/openai.js";
4
4
  import { replicateImage } from "./adapters/image/replicate.js";
5
5
  import { claudeCodeLlm } from "./adapters/llm/claude-code.js";
6
6
  import { codexLlm } from "./adapters/llm/codex.js";
7
+ import { nodeCodexModels } from "./adapters/llm/codex-models.js";
8
+ import { geminiLlm } from "./adapters/llm/gemini.js";
9
+ import { nodeGeminiModels } from "./adapters/llm/gemini-models.js";
7
10
  import { openRouterLlm } from "./adapters/llm/openrouter.js";
8
11
  import { cartesiaTts } from "./adapters/tts/cartesia.js";
9
12
  import { elevenLabsTts } from "./adapters/tts/elevenlabs.js";
13
+ import { inworldTts } from "./adapters/tts/inworld.js";
10
14
  import { openAiTts } from "./adapters/tts/openai.js";
15
+ import { cliBinary } from "./slices/settings/cli-paths.js";
11
16
  import { keyForAttempt } from "./slices/settings/keys.js";
12
17
  import { providerStatuses } from "./slices/settings/readiness.js";
13
18
  export function buildRegistry(deps) {
@@ -18,11 +23,20 @@ export function buildRegistry(deps) {
18
23
  const found = keyForAttempt({ db: deps.db, clock: deps.clock }, provider);
19
24
  return found.ok ? found.key : undefined;
20
25
  };
26
+ // Resolve at invocation time so saved path changes apply to the next attempt.
27
+ const cliFor = (provider) => (_binary, args, signal, ...options) => deps.spawn(cliBinary(deps.db, provider), args, signal, ...options);
21
28
  const llms = new Map([
22
29
  ["openrouter", openRouterLlm({ fetch: deps.fetch, key: keyOf("openrouter") })],
23
- // No key: both CLIs authenticate with their own login.
24
- ["claude-code", claudeCodeLlm({ run: deps.spawn })],
25
- ["codex", codexLlm({ run: deps.spawn })],
30
+ // Each CLI authenticates with its own login.
31
+ ["claude-code", claudeCodeLlm({ run: cliFor("claude-code") })],
32
+ ["codex", codexLlm({ run: cliFor("codex"), readModels: () => nodeCodexModels() })],
33
+ [
34
+ "gemini",
35
+ geminiLlm({
36
+ run: cliFor("gemini"),
37
+ readModels: () => nodeGeminiModels(cliBinary(deps.db, "gemini")),
38
+ }),
39
+ ],
26
40
  ]);
27
41
  // One key per provider, so each adapter is handed the reader for its own row and no other.
28
42
  // OpenAI keeps two rows because it ships an adapter in two families and a user may key one
@@ -31,6 +45,7 @@ export function buildRegistry(deps) {
31
45
  ["elevenlabs", elevenLabsTts({ fetch: deps.fetch, key: keyOf("elevenlabs") })],
32
46
  ["openai-tts", openAiTts({ fetch: deps.fetch, key: keyOf("openai-tts") })],
33
47
  ["cartesia", cartesiaTts({ fetch: deps.fetch, key: keyOf("cartesia") })],
48
+ ["inworld", inworldTts({ fetch: deps.fetch, key: keyOf("inworld"), clock: deps.clock })],
34
49
  ]);
35
50
  // Four image providers behind one port, each handed the reader for its own key row.
36
51
  // Replicate also takes the clock: `Prefer: wait` gives up after 60 s and the prediction has
@@ -0,0 +1,45 @@
1
+ import { spawn } from "node:child_process";
2
+ export async function decodeAudio(ffmpeg, source, output, signal) {
3
+ signal.throwIfAborted();
4
+ return new Promise((resolve, reject) => {
5
+ const child = spawn(ffmpeg, [
6
+ "-hide_banner",
7
+ "-nostdin",
8
+ "-loglevel",
9
+ "error",
10
+ "-y",
11
+ "-i",
12
+ source,
13
+ "-map",
14
+ "0:a:0",
15
+ "-vn",
16
+ "-ar",
17
+ "16000",
18
+ "-ac",
19
+ "1",
20
+ "-f",
21
+ "f32le",
22
+ output,
23
+ ], { stdio: ["ignore", "ignore", "pipe"], windowsHide: true });
24
+ let errorText = "";
25
+ const abort = () => {
26
+ child.kill("SIGKILL");
27
+ };
28
+ signal.addEventListener("abort", abort, { once: true });
29
+ child.stderr.on("data", (data) => {
30
+ errorText = (errorText + data.toString("utf8")).slice(-2000);
31
+ });
32
+ child.on("error", reject);
33
+ child.on("close", (code) => {
34
+ signal.removeEventListener("abort", abort);
35
+ if (signal.aborted)
36
+ reject(new Error("Subtitle audio preparation was canceled."));
37
+ else if (code !== 0)
38
+ reject(new Error(`Could not prepare the narration for subtitles. ${errorText}`));
39
+ else
40
+ resolve();
41
+ });
42
+ if (signal.aborted)
43
+ abort();
44
+ });
45
+ }
@@ -0,0 +1,91 @@
1
+ import { createHash } from "node:crypto";
2
+ import { createReadStream, createWriteStream } from "node:fs";
3
+ import { mkdir, mkdtemp, rename, rm, stat } from "node:fs/promises";
4
+ import { join } from "node:path";
5
+ import { Readable } from "node:stream";
6
+ import { pipeline } from "node:stream/promises";
7
+ const revision = "a19f851b3d42865797e410752b4c570c871e4825";
8
+ export const alignmentModel = {
9
+ filename: "wav2vec2-base-960h-cd5040c1.onnx",
10
+ url: `https://huggingface.co/Xenova/wav2vec2-base-960h/resolve/${revision}/onnx/model_quantized.onnx`,
11
+ bytes: 95_286_046,
12
+ sha256: "cd5040c147381580ed73258143dd8e0c28e800a09e74ee42ee2b3e8cb4d760a3",
13
+ };
14
+ export async function prepareModel(input, deps = { model: alignmentModel, fetch: globalThis.fetch }) {
15
+ input.signal.throwIfAborted();
16
+ await mkdir(input.cacheDir, { recursive: true, mode: 0o700 });
17
+ const target = join(input.cacheDir, deps.model.filename);
18
+ if (await verified(target, deps.model, input.signal))
19
+ return target;
20
+ const staging = await mkdtemp(join(input.cacheDir, ".alignment-download-"));
21
+ const part = join(staging, "model.part");
22
+ try {
23
+ const response = await deps.fetch(deps.model.url, {
24
+ signal: AbortSignal.any([input.signal, AbortSignal.timeout(300_000)]),
25
+ });
26
+ if (!response.ok || response.body === null)
27
+ throw new Error(`The free subtitle model could not be downloaded (HTTP ${response.status}). Check your connection and try again.`);
28
+ let current = 0;
29
+ const hash = createHash("sha256");
30
+ const reader = response.body.getReader();
31
+ async function* chunks() {
32
+ try {
33
+ for (;;) {
34
+ input.signal.throwIfAborted();
35
+ const result = await reader.read();
36
+ if (result.done)
37
+ break;
38
+ current += result.value.length;
39
+ if (current > deps.model.bytes)
40
+ throw new Error("The subtitle model failed download size verification.");
41
+ hash.update(result.value);
42
+ input.onProgress?.(current, deps.model.bytes);
43
+ yield result.value;
44
+ }
45
+ }
46
+ finally {
47
+ await reader.cancel();
48
+ }
49
+ }
50
+ await pipeline(Readable.from(chunks()), createWriteStream(part, { mode: 0o600 }), {
51
+ signal: input.signal,
52
+ });
53
+ input.signal.throwIfAborted();
54
+ if (current !== deps.model.bytes || hash.digest("hex") !== deps.model.sha256)
55
+ throw new Error("The subtitle model failed download verification. Please try again.");
56
+ // Each caller writes in its own directory. On Windows a completed concurrent download
57
+ // may already own the final filename; reuse it only after the same verification.
58
+ try {
59
+ await rename(part, target);
60
+ }
61
+ catch (error) {
62
+ if (await verified(target, deps.model, input.signal))
63
+ return target;
64
+ if (!isCode(error, "EEXIST") && !isCode(error, "EPERM"))
65
+ throw error;
66
+ await rm(target, { force: true });
67
+ await rename(part, target);
68
+ }
69
+ return target;
70
+ }
71
+ finally {
72
+ await rm(staging, { recursive: true, force: true });
73
+ }
74
+ }
75
+ async function verified(path, model, signal) {
76
+ try {
77
+ if ((await stat(path)).size !== model.bytes)
78
+ return false;
79
+ const hash = createHash("sha256");
80
+ await pipeline(createReadStream(path), hash, { signal });
81
+ return hash.digest("hex") === model.sha256;
82
+ }
83
+ catch (error) {
84
+ if (isCode(error, "ENOENT"))
85
+ return false;
86
+ throw error;
87
+ }
88
+ }
89
+ function isCode(error, code) {
90
+ return error instanceof Error && "code" in error && error.code === code;
91
+ }
@@ -0,0 +1,128 @@
1
+ import { vocabulary } from "./vocabulary.js";
2
+ export const frameSeconds = 0.02;
3
+ export const mismatch = "The audio does not closely match the English transcript. Check the article and audio, including any intro or outro, before generating subtitles.";
4
+ export function alignWindow(logits, frames, words, complete) {
5
+ if (frames < 1 || frames > 1100 || logits.length !== frames * 32)
6
+ throw new Error("The subtitle alignment window exceeds its safe size.");
7
+ const tokens = tokenize(words);
8
+ if (tokens.ids.length === 0 || tokens.ids.length > 1500)
9
+ throw new Error("The subtitle transcript window exceeds its safe size.");
10
+ const states = [0];
11
+ for (const id of tokens.ids)
12
+ states.push(id, 0);
13
+ const probabilities = logSoftmax(logits, frames);
14
+ const width = states.length;
15
+ const back = new Uint8Array(frames * width);
16
+ let previous = new Float32Array(width).fill(-Infinity);
17
+ previous[0] = 0;
18
+ for (let frame = 0; frame < frames; frame += 1) {
19
+ const current = new Float32Array(width).fill(-Infinity);
20
+ for (let state = 0; state < Math.min(width, frame * 2 + 3); state += 1) {
21
+ let score = previous[state] ?? -Infinity;
22
+ let step = 0;
23
+ if (state > 0 && (previous[state - 1] ?? -Infinity) > score) {
24
+ score = previous[state - 1] ?? -Infinity;
25
+ step = 1;
26
+ }
27
+ if (state > 1 &&
28
+ states[state] !== 0 &&
29
+ states[state] !== states[state - 2] &&
30
+ (previous[state - 2] ?? -Infinity) > score) {
31
+ score = previous[state - 2] ?? -Infinity;
32
+ step = 2;
33
+ }
34
+ current[state] = score + (probabilities[frame * 32 + (states[state] ?? 0)] ?? -Infinity);
35
+ back[frame * width + state] = step;
36
+ }
37
+ previous = current;
38
+ }
39
+ const state = endState(previous, tokens, complete);
40
+ if (!Number.isFinite(previous[state]))
41
+ throw new Error(mismatch);
42
+ const traced = trace(back, states, tokens, words, probabilities, frames, state);
43
+ if (traced.length === 0)
44
+ throw new Error(mismatch);
45
+ const confidence = traced.reduce((sum, word) => sum + (word.confidence ?? 0), 0) / traced.length;
46
+ const poor = traced.filter((word) => (word.confidence ?? 0) < 0.2).length / traced.length;
47
+ if (confidence < 0.48 || poor > 0.3)
48
+ throw new Error(mismatch);
49
+ return { words: traced, confidence };
50
+ }
51
+ function tokenize(words) {
52
+ const ids = [], owners = [], ends = [];
53
+ for (const [index, word] of words.entries()) {
54
+ if (index > 0) {
55
+ ids.push(4);
56
+ owners.push(-1);
57
+ }
58
+ for (const letter of word.spoken.replaceAll(" ", "|")) {
59
+ const id = vocabulary[letter];
60
+ if (id !== undefined) {
61
+ ids.push(id);
62
+ owners.push(index);
63
+ }
64
+ }
65
+ ends.push(ids.length * 2 - 1);
66
+ }
67
+ return { ids, owners, ends };
68
+ }
69
+ function endState(scores, tokens, complete) {
70
+ if (complete)
71
+ return (scores.at(-1) ?? -Infinity) > (scores.at(-2) ?? -Infinity)
72
+ ? scores.length - 1
73
+ : scores.length - 2;
74
+ let best = 0;
75
+ for (const end of tokens.ends) {
76
+ for (const state of [end, end + 1, end + 2]) {
77
+ if ((scores[state] ?? -Infinity) > (scores[best] ?? -Infinity))
78
+ best = state;
79
+ }
80
+ }
81
+ return best;
82
+ }
83
+ function trace(back, states, tokens, words, probabilities, frames, finalState) {
84
+ const found = words.map((word) => ({
85
+ text: word.text,
86
+ start: Infinity,
87
+ end: 0,
88
+ score: 0,
89
+ count: 0,
90
+ }));
91
+ let state = finalState;
92
+ for (let frame = frames - 1; frame >= 0; frame -= 1) {
93
+ if (state % 2 === 1) {
94
+ const owner = tokens.owners[(state - 1) / 2] ?? -1;
95
+ const word = found[owner];
96
+ if (word !== undefined) {
97
+ word.start = Math.min(word.start, frame * frameSeconds);
98
+ word.end = Math.max(word.end, (frame + 1) * frameSeconds);
99
+ word.score += Math.exp(probabilities[frame * 32 + (states[state] ?? 0)] ?? -Infinity);
100
+ word.count += 1;
101
+ }
102
+ }
103
+ state -= back[frame * states.length + state] ?? 0;
104
+ }
105
+ return found.flatMap((word, index) => {
106
+ if (word.count === 0 || (tokens.ends[index] ?? Infinity) > finalState)
107
+ return [];
108
+ return [
109
+ { text: word.text, start: word.start, end: word.end, confidence: word.score / word.count },
110
+ ];
111
+ });
112
+ }
113
+ function logSoftmax(logits, frames) {
114
+ const result = new Float32Array(logits.length);
115
+ for (let frame = 0; frame < frames; frame += 1) {
116
+ const offset = frame * 32;
117
+ let max = -Infinity;
118
+ for (let label = 0; label < 32; label += 1)
119
+ max = Math.max(max, logits[offset + label] ?? -Infinity);
120
+ let sum = 0;
121
+ for (let label = 0; label < 32; label += 1)
122
+ sum += Math.exp((logits[offset + label] ?? -Infinity) - max);
123
+ const divisor = max + Math.log(sum);
124
+ for (let label = 0; label < 32; label += 1)
125
+ result[offset + label] = (logits[offset + label] ?? -Infinity) - divisor;
126
+ }
127
+ return result;
128
+ }
@@ -0,0 +1,34 @@
1
+ import { mkdir, mkdtemp, rm } from "node:fs/promises";
2
+ import { join } from "node:path";
3
+ import { decodeAudio } from "./audio.js";
4
+ import { prepareModel } from "./cache.js";
5
+ import { claimWorker } from "./lock.js";
6
+ import { runAlignmentWorker } from "./runner.js";
7
+ import { speechWords } from "./text.js";
8
+ export const alignSubtitles = async (request) => {
9
+ request.signal.throwIfAborted();
10
+ speechWords(request.text);
11
+ await mkdir(request.cacheDir, { recursive: true, mode: 0o700 });
12
+ const release = await claimWorker(request.cacheDir, request.signal);
13
+ try {
14
+ const modelPath = await prepareModel({
15
+ cacheDir: request.cacheDir,
16
+ signal: request.signal,
17
+ onProgress: (current, total) => request.onProgress?.(Math.round((current / total) * 20), 100),
18
+ });
19
+ request.onProgress?.(20, 100);
20
+ const working = await mkdtemp(join(request.cacheDir, ".alignment-audio-"));
21
+ try {
22
+ const pcmPath = join(working, "audio.f32");
23
+ await decodeAudio(request.ffmpeg, request.audioPath, pcmPath, request.signal);
24
+ request.onProgress?.(25, 100);
25
+ return await runAlignmentWorker({ modelPath, pcmPath, text: request.text }, request.signal, (current, total) => request.onProgress?.(25 + Math.round((current / total) * 75), 100));
26
+ }
27
+ finally {
28
+ await rm(working, { recursive: true, force: true });
29
+ }
30
+ }
31
+ finally {
32
+ await release();
33
+ }
34
+ };
@@ -0,0 +1,58 @@
1
+ import { open, readFile, rm, stat } from "node:fs/promises";
2
+ import { join } from "node:path";
3
+ import { setTimeout } from "node:timers/promises";
4
+ // Models run one at a time per data directory, including requests from separate app
5
+ // processes. A canceled waiter never removes another request's lock.
6
+ export async function claimWorker(cacheDir, signal) {
7
+ const path = join(cacheDir, "alignment-worker.lock");
8
+ for (;;) {
9
+ signal.throwIfAborted();
10
+ try {
11
+ const file = await open(path, "wx", 0o600);
12
+ try {
13
+ try {
14
+ await file.writeFile(String(process.pid));
15
+ }
16
+ finally {
17
+ await file.close();
18
+ }
19
+ }
20
+ catch (error) {
21
+ await rm(path, { force: true });
22
+ throw error;
23
+ }
24
+ return async () => {
25
+ await rm(path, { force: true });
26
+ };
27
+ }
28
+ catch (error) {
29
+ if (!(error instanceof Error && "code" in error && error.code === "EEXIST"))
30
+ throw error;
31
+ }
32
+ try {
33
+ const owner = Number(await readFile(path, "utf8"));
34
+ if ((!Number.isInteger(owner) || owner <= 0) &&
35
+ Date.now() - (await stat(path)).mtimeMs > 10_000) {
36
+ await rm(path, { force: true });
37
+ continue;
38
+ }
39
+ if (Number.isInteger(owner) && owner > 0) {
40
+ try {
41
+ process.kill(owner, 0);
42
+ }
43
+ catch (error) {
44
+ if (error instanceof Error && "code" in error && error.code === "ESRCH") {
45
+ await rm(path, { force: true });
46
+ continue;
47
+ }
48
+ throw error;
49
+ }
50
+ }
51
+ }
52
+ catch (error) {
53
+ if (!(error instanceof Error && "code" in error && error.code === "ENOENT"))
54
+ throw error;
55
+ }
56
+ await setTimeout(250, undefined, { signal });
57
+ }
58
+ }
@@ -0,0 +1,91 @@
1
+ const small = [
2
+ "ZERO",
3
+ "ONE",
4
+ "TWO",
5
+ "THREE",
6
+ "FOUR",
7
+ "FIVE",
8
+ "SIX",
9
+ "SEVEN",
10
+ "EIGHT",
11
+ "NINE",
12
+ "TEN",
13
+ "ELEVEN",
14
+ "TWELVE",
15
+ "THIRTEEN",
16
+ "FOURTEEN",
17
+ "FIFTEEN",
18
+ "SIXTEEN",
19
+ "SEVENTEEN",
20
+ "EIGHTEEN",
21
+ "NINETEEN",
22
+ ];
23
+ const tens = [
24
+ "",
25
+ "",
26
+ "TWENTY",
27
+ "THIRTY",
28
+ "FORTY",
29
+ "FIFTY",
30
+ "SIXTY",
31
+ "SEVENTY",
32
+ "EIGHTY",
33
+ "NINETY",
34
+ ];
35
+ const ordinal = {
36
+ ONE: "FIRST",
37
+ TWO: "SECOND",
38
+ THREE: "THIRD",
39
+ FIVE: "FIFTH",
40
+ EIGHT: "EIGHTH",
41
+ NINE: "NINTH",
42
+ TWELVE: "TWELFTH",
43
+ };
44
+ export function cardinal(value) {
45
+ if (value < 20)
46
+ return small[value] ?? "";
47
+ if (value < 100)
48
+ return `${tens[Math.floor(value / 10)] ?? ""} ${value % 10 === 0 ? "" : cardinal(value % 10)}`.trim();
49
+ for (const [size, name] of [
50
+ [1_000_000_000, "BILLION"],
51
+ [1_000_000, "MILLION"],
52
+ [1000, "THOUSAND"],
53
+ [100, "HUNDRED"],
54
+ ]) {
55
+ if (value >= size)
56
+ return `${cardinal(Math.floor(value / size))} ${name}${value % size === 0 ? "" : ` ${cardinal(value % size)}`}`;
57
+ }
58
+ return "";
59
+ }
60
+ export function numberForms(raw) {
61
+ const match = /^(\d[\d,]*)(?:\.(\d+))?(s|st|nd|rd|th)?$/i.exec(raw);
62
+ if (match === null)
63
+ return [raw.replace(/\d/g, (digit) => ` ${small[Number(digit)] ?? ""} `)];
64
+ const digits = (match[1] ?? "").replaceAll(",", "");
65
+ const value = Number(digits);
66
+ if (!Number.isSafeInteger(value) || value >= 1_000_000_000_000)
67
+ return [
68
+ digits
69
+ .split("")
70
+ .map((digit) => small[Number(digit)])
71
+ .join(" "),
72
+ ];
73
+ const fraction = match[2];
74
+ const suffix = match[3]?.toLowerCase();
75
+ const decimal = fraction === undefined
76
+ ? ""
77
+ : ` POINT ${fraction
78
+ .split("")
79
+ .map((digit) => small[Number(digit)])
80
+ .join(" ")}`;
81
+ const ordinary = `${cardinal(value)}${decimal}`;
82
+ const year = value >= 1900 && value <= 2099 && value % 100 >= 10
83
+ ? `${cardinal(Math.floor(value / 100))} ${cardinal(value % 100)}`
84
+ : undefined;
85
+ const forms = year === undefined ? [ordinary] : [year, ordinary];
86
+ if (suffix === "s")
87
+ return forms.map((form) => `${form.replace(/Y$/, "IE")}S`);
88
+ if (suffix !== undefined)
89
+ return forms.map((form) => form.replace(/[A-Z]+$/, (word) => ordinal[word] ?? (word.endsWith("Y") ? `${word.slice(0, -1)}IETH` : `${word}TH`)));
90
+ return forms;
91
+ }
@@ -0,0 +1,23 @@
1
+ import { z } from "zod";
2
+ export const workerInput = z.object({
3
+ modelPath: z.string(),
4
+ pcmPath: z.string(),
5
+ text: z.string(),
6
+ });
7
+ export const workerMessage = z.discriminatedUnion("type", [
8
+ z.object({
9
+ type: z.literal("progress"),
10
+ current: z.number().finite().nonnegative(),
11
+ total: z.number().finite().positive(),
12
+ }),
13
+ z.object({
14
+ type: z.literal("done"),
15
+ words: z.array(z.object({
16
+ text: z.string(),
17
+ start: z.number().finite().nonnegative(),
18
+ end: z.number().finite().nonnegative(),
19
+ confidence: z.number().finite().min(0).max(1).optional(),
20
+ })),
21
+ }),
22
+ z.object({ type: z.literal("error"), message: z.string() }),
23
+ ]);
@@ -0,0 +1,17 @@
1
+ // Compare acoustic recognition with the already aligned transcript. This catches extra
2
+ // speech and missing phrases even when a few forced words individually score well.
3
+ export function agreesWithSpeech(expected, observed) {
4
+ const left = expected.toUpperCase().replace(/[^A-Z']/g, "");
5
+ const right = observed.toUpperCase().replace(/[^A-Z']/g, "");
6
+ if (left.length === 0 || right.length === 0)
7
+ return false;
8
+ let previous = Uint16Array.from({ length: right.length + 1 }, (_, index) => index);
9
+ for (let row = 1; row <= left.length; row += 1) {
10
+ const current = new Uint16Array(right.length + 1);
11
+ current[0] = row;
12
+ for (let column = 1; column <= right.length; column += 1)
13
+ current[column] = Math.min((previous[column] ?? 0) + 1, (current[column - 1] ?? 0) + 1, (previous[column - 1] ?? 0) + (left[row - 1] === right[column - 1] ? 0 : 1));
14
+ previous = current;
15
+ }
16
+ return (previous[right.length] ?? Infinity) / Math.max(left.length, right.length) <= 0.42;
17
+ }