@gentbajko/slopify 0.5.1 → 0.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/SUBTITLES.md +21 -0
- package/dist/adapter-registry.js +8 -3
- package/dist/adapters/alignment/audio.js +45 -0
- package/dist/adapters/alignment/cache.js +91 -0
- package/dist/adapters/alignment/ctc.js +128 -0
- package/dist/adapters/alignment/index.js +34 -0
- package/dist/adapters/alignment/lock.js +58 -0
- package/dist/adapters/alignment/numbers.js +91 -0
- package/dist/adapters/alignment/protocol.js +23 -0
- package/dist/adapters/alignment/quality.js +17 -0
- package/dist/adapters/alignment/runner.js +76 -0
- package/dist/adapters/alignment/text.js +61 -0
- package/dist/adapters/alignment/vocabulary.js +69 -0
- package/dist/adapters/alignment/worker.js +163 -0
- package/dist/adapters/llm/claude-code.js +6 -3
- package/dist/adapters/llm/codex.js +3 -2
- package/dist/adapters/llm/gemini-workspace.js +70 -0
- package/dist/adapters/llm/gemini.js +141 -0
- package/dist/adapters/llm/run-cli.js +6 -9
- package/dist/assets/fonts/Barlow-Regular.ttf +0 -0
- package/dist/assets/fonts/OFL.txt +93 -0
- package/dist/assets/fonts/SOURCE.txt +7 -0
- package/dist/edge/http/app.js +4 -0
- package/dist/edge/http/fonts.js +147 -0
- package/dist/edge/http/projects.js +23 -1
- package/dist/edge/http/providers.js +13 -0
- package/dist/edge/http/subtitles.js +92 -0
- package/dist/kernel/cli-command.js +45 -0
- package/dist/kernel/ports/subtitles.js +1 -0
- package/dist/kernel/runner/providers.js +3 -2
- package/dist/main.js +2 -1
- package/dist/slices/admission/repo.js +2 -0
- package/dist/slices/admission/rules.js +6 -0
- package/dist/slices/fonts/catalog.js +95 -0
- package/dist/slices/fonts/discovery.js +81 -0
- package/dist/slices/fonts/files.js +37 -0
- package/dist/slices/fonts/index.js +5 -0
- package/dist/slices/fonts/model.js +4 -0
- package/dist/slices/fonts/preview.js +14 -0
- package/dist/slices/fonts/sfnt.js +262 -0
- package/dist/slices/fonts/upload.js +40 -0
- package/dist/slices/settings/cli-paths.js +98 -0
- package/dist/slices/settings/cli-status.js +21 -12
- package/dist/slices/settings/model.js +9 -0
- package/dist/slices/settings/readiness.js +13 -9
- package/dist/slices/storage/downloads.js +2 -0
- package/dist/slices/storage/layout.js +10 -0
- package/dist/slices/storage/model.js +5 -0
- package/dist/slices/storage/repo.js +1 -0
- package/dist/slices/subtitles/captions.js +88 -0
- package/dist/slices/subtitles/model.js +18 -0
- package/dist/slices/subtitles/prepare.js +162 -0
- package/dist/slices/subtitles/transcript.js +34 -0
- package/dist/slices/video/audio-export.js +3 -0
- package/dist/slices/video/ffmpeg.js +14 -7
- package/dist/slices/video/plan.js +4 -4
- package/dist/slices/video/run.js +5 -2
- package/dist/slices/video/write-export.js +85 -20
- package/dist/web/assets/index-CNYs0noB.js +130 -0
- package/dist/web/assets/index-Mk0bvBg-.css +1 -0
- package/dist/web/index.html +2 -2
- package/package.json +4 -2
- package/dist/web/assets/index-C1fsrIP5.css +0 -1
- package/dist/web/assets/index-DwJ_XTl3.js +0 -81
package/SUBTITLES.md
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# Subtitles in Slopify 0.6
|
|
2
|
+
|
|
3
|
+
1. In Play, enable narration or provide matching English audio and article text.
|
|
4
|
+
2. In Subtitles, choose **Subtitle files (.srt + .vtt)** or **Burn into video + files**.
|
|
5
|
+
3. For burned captions, select a bundled or system font, or upload a `.ttf` or `.otf` file (up to 32 MiB). Preview the font and choose a size from 16 to 120.
|
|
6
|
+
4. Start the run. Slopify times the article against the actual narration on your computer. The first use downloads an approximately 95 MB English speech model; later runs work offline with the cached model. Allow roughly 1 GB of available memory during alignment. No extra API key, Python, or compiler is needed.
|
|
7
|
+
5. Download SRT/VTT beside the final export. Files mode adds an optional native caption track to the in-app video preview. Burned captions remain visible in the downloaded MP4.
|
|
8
|
+
|
|
9
|
+
For a completed project, open its final Video or Audio export section, set subtitles, and click **Save subtitles**. This rebuilds only the local export from saved narration and images. Changing font or size reuses word timing when the audio and spoken text are unchanged. A paused run saves these choices until Resume; pause an active run before editing. Failed alignment or rendering keeps the previous finished export.
|
|
10
|
+
|
|
11
|
+
Audio Off disables subtitles. Video Off produces WAV audio with separate subtitle files. Uploaded audio must match the article; substantial mismatches fail with a message to correct the transcript. Review timing and spelling before publishing. English is supported first; unusual pronunciations and non-English passages can fail alignment.
|
|
12
|
+
|
|
13
|
+
Fonts are copied into the completed project's caption assets, so an existing export can reuse its chosen font even after the original system font is removed. Installed fonts and uploaded fonts stay local. SRT/VTT are portable text/timing files and do not embed a font; the chosen font is used in burned video captions.
|
|
14
|
+
|
|
15
|
+
## Local model and licenses
|
|
16
|
+
|
|
17
|
+
Speech inference uses MIT-licensed `onnxruntime-web@1.24.3` in a separate WASM process. The runtime is installed with the package (about 138 MB on disk); model weights are downloaded lazily into `<data-dir>/models/english-subtitles/`.
|
|
18
|
+
|
|
19
|
+
Model: [Xenova/wav2vec2-base-960h](https://huggingface.co/Xenova/wav2vec2-base-960h/tree/a19f851b3d42865797e410752b4c570c871e4825), an ONNX conversion of [facebook/wav2vec2-base-960h](https://huggingface.co/facebook/wav2vec2-base-960h), licensed Apache-2.0. Pinned revision: `a19f851b3d42865797e410752b4c570c871e4825`. The quantized model is 95,286,046 bytes, SHA256 `cd5040c147381580ed73258143dd8e0c28e800a09e74ee42ee2b3e8cb4d760a3`; every cached/downloaded model is verified before use.
|
|
20
|
+
|
|
21
|
+
The bundled Barlow font uses the SIL Open Font License; its license and source record ship in `dist/assets/fonts/`. Uploaded/system font licensing remains with its author.
|
package/dist/adapter-registry.js
CHANGED
|
@@ -4,10 +4,12 @@ import { openAiImage } from "./adapters/image/openai.js";
|
|
|
4
4
|
import { replicateImage } from "./adapters/image/replicate.js";
|
|
5
5
|
import { claudeCodeLlm } from "./adapters/llm/claude-code.js";
|
|
6
6
|
import { codexLlm } from "./adapters/llm/codex.js";
|
|
7
|
+
import { geminiLlm } from "./adapters/llm/gemini.js";
|
|
7
8
|
import { openRouterLlm } from "./adapters/llm/openrouter.js";
|
|
8
9
|
import { cartesiaTts } from "./adapters/tts/cartesia.js";
|
|
9
10
|
import { elevenLabsTts } from "./adapters/tts/elevenlabs.js";
|
|
10
11
|
import { openAiTts } from "./adapters/tts/openai.js";
|
|
12
|
+
import { cliBinary } from "./slices/settings/cli-paths.js";
|
|
11
13
|
import { keyForAttempt } from "./slices/settings/keys.js";
|
|
12
14
|
import { providerStatuses } from "./slices/settings/readiness.js";
|
|
13
15
|
export function buildRegistry(deps) {
|
|
@@ -18,11 +20,14 @@ export function buildRegistry(deps) {
|
|
|
18
20
|
const found = keyForAttempt({ db: deps.db, clock: deps.clock }, provider);
|
|
19
21
|
return found.ok ? found.key : undefined;
|
|
20
22
|
};
|
|
23
|
+
// Resolve at invocation time so saved path changes apply to the next attempt.
|
|
24
|
+
const cliFor = (provider) => (_binary, args, signal, ...options) => deps.spawn(cliBinary(deps.db, provider), args, signal, ...options);
|
|
21
25
|
const llms = new Map([
|
|
22
26
|
["openrouter", openRouterLlm({ fetch: deps.fetch, key: keyOf("openrouter") })],
|
|
23
|
-
//
|
|
24
|
-
["claude-code", claudeCodeLlm({ run:
|
|
25
|
-
["codex", codexLlm({ run:
|
|
27
|
+
// Each CLI authenticates with its own login.
|
|
28
|
+
["claude-code", claudeCodeLlm({ run: cliFor("claude-code") })],
|
|
29
|
+
["codex", codexLlm({ run: cliFor("codex") })],
|
|
30
|
+
["gemini", geminiLlm({ run: cliFor("gemini") })],
|
|
26
31
|
]);
|
|
27
32
|
// One key per provider, so each adapter is handed the reader for its own row and no other.
|
|
28
33
|
// OpenAI keeps two rows because it ships an adapter in two families and a user may key one
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import { spawn } from "node:child_process";
|
|
2
|
+
export async function decodeAudio(ffmpeg, source, output, signal) {
|
|
3
|
+
signal.throwIfAborted();
|
|
4
|
+
return new Promise((resolve, reject) => {
|
|
5
|
+
const child = spawn(ffmpeg, [
|
|
6
|
+
"-hide_banner",
|
|
7
|
+
"-nostdin",
|
|
8
|
+
"-loglevel",
|
|
9
|
+
"error",
|
|
10
|
+
"-y",
|
|
11
|
+
"-i",
|
|
12
|
+
source,
|
|
13
|
+
"-map",
|
|
14
|
+
"0:a:0",
|
|
15
|
+
"-vn",
|
|
16
|
+
"-ar",
|
|
17
|
+
"16000",
|
|
18
|
+
"-ac",
|
|
19
|
+
"1",
|
|
20
|
+
"-f",
|
|
21
|
+
"f32le",
|
|
22
|
+
output,
|
|
23
|
+
], { stdio: ["ignore", "ignore", "pipe"], windowsHide: true });
|
|
24
|
+
let errorText = "";
|
|
25
|
+
const abort = () => {
|
|
26
|
+
child.kill("SIGKILL");
|
|
27
|
+
};
|
|
28
|
+
signal.addEventListener("abort", abort, { once: true });
|
|
29
|
+
child.stderr.on("data", (data) => {
|
|
30
|
+
errorText = (errorText + data.toString("utf8")).slice(-2000);
|
|
31
|
+
});
|
|
32
|
+
child.on("error", reject);
|
|
33
|
+
child.on("close", (code) => {
|
|
34
|
+
signal.removeEventListener("abort", abort);
|
|
35
|
+
if (signal.aborted)
|
|
36
|
+
reject(new Error("Subtitle audio preparation was canceled."));
|
|
37
|
+
else if (code !== 0)
|
|
38
|
+
reject(new Error(`Could not prepare the narration for subtitles. ${errorText}`));
|
|
39
|
+
else
|
|
40
|
+
resolve();
|
|
41
|
+
});
|
|
42
|
+
if (signal.aborted)
|
|
43
|
+
abort();
|
|
44
|
+
});
|
|
45
|
+
}
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import { createReadStream, createWriteStream } from "node:fs";
|
|
3
|
+
import { mkdir, mkdtemp, rename, rm, stat } from "node:fs/promises";
|
|
4
|
+
import { join } from "node:path";
|
|
5
|
+
import { Readable } from "node:stream";
|
|
6
|
+
import { pipeline } from "node:stream/promises";
|
|
7
|
+
const revision = "a19f851b3d42865797e410752b4c570c871e4825";
|
|
8
|
+
export const alignmentModel = {
|
|
9
|
+
filename: "wav2vec2-base-960h-cd5040c1.onnx",
|
|
10
|
+
url: `https://huggingface.co/Xenova/wav2vec2-base-960h/resolve/${revision}/onnx/model_quantized.onnx`,
|
|
11
|
+
bytes: 95_286_046,
|
|
12
|
+
sha256: "cd5040c147381580ed73258143dd8e0c28e800a09e74ee42ee2b3e8cb4d760a3",
|
|
13
|
+
};
|
|
14
|
+
export async function prepareModel(input, deps = { model: alignmentModel, fetch: globalThis.fetch }) {
|
|
15
|
+
input.signal.throwIfAborted();
|
|
16
|
+
await mkdir(input.cacheDir, { recursive: true, mode: 0o700 });
|
|
17
|
+
const target = join(input.cacheDir, deps.model.filename);
|
|
18
|
+
if (await verified(target, deps.model, input.signal))
|
|
19
|
+
return target;
|
|
20
|
+
const staging = await mkdtemp(join(input.cacheDir, ".alignment-download-"));
|
|
21
|
+
const part = join(staging, "model.part");
|
|
22
|
+
try {
|
|
23
|
+
const response = await deps.fetch(deps.model.url, {
|
|
24
|
+
signal: AbortSignal.any([input.signal, AbortSignal.timeout(300_000)]),
|
|
25
|
+
});
|
|
26
|
+
if (!response.ok || response.body === null)
|
|
27
|
+
throw new Error(`The free subtitle model could not be downloaded (HTTP ${response.status}). Check your connection and try again.`);
|
|
28
|
+
let current = 0;
|
|
29
|
+
const hash = createHash("sha256");
|
|
30
|
+
const reader = response.body.getReader();
|
|
31
|
+
async function* chunks() {
|
|
32
|
+
try {
|
|
33
|
+
for (;;) {
|
|
34
|
+
input.signal.throwIfAborted();
|
|
35
|
+
const result = await reader.read();
|
|
36
|
+
if (result.done)
|
|
37
|
+
break;
|
|
38
|
+
current += result.value.length;
|
|
39
|
+
if (current > deps.model.bytes)
|
|
40
|
+
throw new Error("The subtitle model failed download size verification.");
|
|
41
|
+
hash.update(result.value);
|
|
42
|
+
input.onProgress?.(current, deps.model.bytes);
|
|
43
|
+
yield result.value;
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
finally {
|
|
47
|
+
await reader.cancel();
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
await pipeline(Readable.from(chunks()), createWriteStream(part, { mode: 0o600 }), {
|
|
51
|
+
signal: input.signal,
|
|
52
|
+
});
|
|
53
|
+
input.signal.throwIfAborted();
|
|
54
|
+
if (current !== deps.model.bytes || hash.digest("hex") !== deps.model.sha256)
|
|
55
|
+
throw new Error("The subtitle model failed download verification. Please try again.");
|
|
56
|
+
// Each caller writes in its own directory. On Windows a completed concurrent download
|
|
57
|
+
// may already own the final filename; reuse it only after the same verification.
|
|
58
|
+
try {
|
|
59
|
+
await rename(part, target);
|
|
60
|
+
}
|
|
61
|
+
catch (error) {
|
|
62
|
+
if (await verified(target, deps.model, input.signal))
|
|
63
|
+
return target;
|
|
64
|
+
if (!isCode(error, "EEXIST") && !isCode(error, "EPERM"))
|
|
65
|
+
throw error;
|
|
66
|
+
await rm(target, { force: true });
|
|
67
|
+
await rename(part, target);
|
|
68
|
+
}
|
|
69
|
+
return target;
|
|
70
|
+
}
|
|
71
|
+
finally {
|
|
72
|
+
await rm(staging, { recursive: true, force: true });
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
async function verified(path, model, signal) {
|
|
76
|
+
try {
|
|
77
|
+
if ((await stat(path)).size !== model.bytes)
|
|
78
|
+
return false;
|
|
79
|
+
const hash = createHash("sha256");
|
|
80
|
+
await pipeline(createReadStream(path), hash, { signal });
|
|
81
|
+
return hash.digest("hex") === model.sha256;
|
|
82
|
+
}
|
|
83
|
+
catch (error) {
|
|
84
|
+
if (isCode(error, "ENOENT"))
|
|
85
|
+
return false;
|
|
86
|
+
throw error;
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
function isCode(error, code) {
|
|
90
|
+
return error instanceof Error && "code" in error && error.code === code;
|
|
91
|
+
}
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
import { vocabulary } from "./vocabulary.js";
|
|
2
|
+
export const frameSeconds = 0.02;
|
|
3
|
+
export const mismatch = "The audio does not closely match the English transcript. Check the article and audio, including any intro or outro, before generating subtitles.";
|
|
4
|
+
export function alignWindow(logits, frames, words, complete) {
|
|
5
|
+
if (frames < 1 || frames > 1100 || logits.length !== frames * 32)
|
|
6
|
+
throw new Error("The subtitle alignment window exceeds its safe size.");
|
|
7
|
+
const tokens = tokenize(words);
|
|
8
|
+
if (tokens.ids.length === 0 || tokens.ids.length > 1500)
|
|
9
|
+
throw new Error("The subtitle transcript window exceeds its safe size.");
|
|
10
|
+
const states = [0];
|
|
11
|
+
for (const id of tokens.ids)
|
|
12
|
+
states.push(id, 0);
|
|
13
|
+
const probabilities = logSoftmax(logits, frames);
|
|
14
|
+
const width = states.length;
|
|
15
|
+
const back = new Uint8Array(frames * width);
|
|
16
|
+
let previous = new Float32Array(width).fill(-Infinity);
|
|
17
|
+
previous[0] = 0;
|
|
18
|
+
for (let frame = 0; frame < frames; frame += 1) {
|
|
19
|
+
const current = new Float32Array(width).fill(-Infinity);
|
|
20
|
+
for (let state = 0; state < Math.min(width, frame * 2 + 3); state += 1) {
|
|
21
|
+
let score = previous[state] ?? -Infinity;
|
|
22
|
+
let step = 0;
|
|
23
|
+
if (state > 0 && (previous[state - 1] ?? -Infinity) > score) {
|
|
24
|
+
score = previous[state - 1] ?? -Infinity;
|
|
25
|
+
step = 1;
|
|
26
|
+
}
|
|
27
|
+
if (state > 1 &&
|
|
28
|
+
states[state] !== 0 &&
|
|
29
|
+
states[state] !== states[state - 2] &&
|
|
30
|
+
(previous[state - 2] ?? -Infinity) > score) {
|
|
31
|
+
score = previous[state - 2] ?? -Infinity;
|
|
32
|
+
step = 2;
|
|
33
|
+
}
|
|
34
|
+
current[state] = score + (probabilities[frame * 32 + (states[state] ?? 0)] ?? -Infinity);
|
|
35
|
+
back[frame * width + state] = step;
|
|
36
|
+
}
|
|
37
|
+
previous = current;
|
|
38
|
+
}
|
|
39
|
+
const state = endState(previous, tokens, complete);
|
|
40
|
+
if (!Number.isFinite(previous[state]))
|
|
41
|
+
throw new Error(mismatch);
|
|
42
|
+
const traced = trace(back, states, tokens, words, probabilities, frames, state);
|
|
43
|
+
if (traced.length === 0)
|
|
44
|
+
throw new Error(mismatch);
|
|
45
|
+
const confidence = traced.reduce((sum, word) => sum + (word.confidence ?? 0), 0) / traced.length;
|
|
46
|
+
const poor = traced.filter((word) => (word.confidence ?? 0) < 0.2).length / traced.length;
|
|
47
|
+
if (confidence < 0.48 || poor > 0.3)
|
|
48
|
+
throw new Error(mismatch);
|
|
49
|
+
return { words: traced, confidence };
|
|
50
|
+
}
|
|
51
|
+
function tokenize(words) {
|
|
52
|
+
const ids = [], owners = [], ends = [];
|
|
53
|
+
for (const [index, word] of words.entries()) {
|
|
54
|
+
if (index > 0) {
|
|
55
|
+
ids.push(4);
|
|
56
|
+
owners.push(-1);
|
|
57
|
+
}
|
|
58
|
+
for (const letter of word.spoken.replaceAll(" ", "|")) {
|
|
59
|
+
const id = vocabulary[letter];
|
|
60
|
+
if (id !== undefined) {
|
|
61
|
+
ids.push(id);
|
|
62
|
+
owners.push(index);
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
ends.push(ids.length * 2 - 1);
|
|
66
|
+
}
|
|
67
|
+
return { ids, owners, ends };
|
|
68
|
+
}
|
|
69
|
+
function endState(scores, tokens, complete) {
|
|
70
|
+
if (complete)
|
|
71
|
+
return (scores.at(-1) ?? -Infinity) > (scores.at(-2) ?? -Infinity)
|
|
72
|
+
? scores.length - 1
|
|
73
|
+
: scores.length - 2;
|
|
74
|
+
let best = 0;
|
|
75
|
+
for (const end of tokens.ends) {
|
|
76
|
+
for (const state of [end, end + 1, end + 2]) {
|
|
77
|
+
if ((scores[state] ?? -Infinity) > (scores[best] ?? -Infinity))
|
|
78
|
+
best = state;
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
return best;
|
|
82
|
+
}
|
|
83
|
+
function trace(back, states, tokens, words, probabilities, frames, finalState) {
|
|
84
|
+
const found = words.map((word) => ({
|
|
85
|
+
text: word.text,
|
|
86
|
+
start: Infinity,
|
|
87
|
+
end: 0,
|
|
88
|
+
score: 0,
|
|
89
|
+
count: 0,
|
|
90
|
+
}));
|
|
91
|
+
let state = finalState;
|
|
92
|
+
for (let frame = frames - 1; frame >= 0; frame -= 1) {
|
|
93
|
+
if (state % 2 === 1) {
|
|
94
|
+
const owner = tokens.owners[(state - 1) / 2] ?? -1;
|
|
95
|
+
const word = found[owner];
|
|
96
|
+
if (word !== undefined) {
|
|
97
|
+
word.start = Math.min(word.start, frame * frameSeconds);
|
|
98
|
+
word.end = Math.max(word.end, (frame + 1) * frameSeconds);
|
|
99
|
+
word.score += Math.exp(probabilities[frame * 32 + (states[state] ?? 0)] ?? -Infinity);
|
|
100
|
+
word.count += 1;
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
state -= back[frame * states.length + state] ?? 0;
|
|
104
|
+
}
|
|
105
|
+
return found.flatMap((word, index) => {
|
|
106
|
+
if (word.count === 0 || (tokens.ends[index] ?? Infinity) > finalState)
|
|
107
|
+
return [];
|
|
108
|
+
return [
|
|
109
|
+
{ text: word.text, start: word.start, end: word.end, confidence: word.score / word.count },
|
|
110
|
+
];
|
|
111
|
+
});
|
|
112
|
+
}
|
|
113
|
+
function logSoftmax(logits, frames) {
|
|
114
|
+
const result = new Float32Array(logits.length);
|
|
115
|
+
for (let frame = 0; frame < frames; frame += 1) {
|
|
116
|
+
const offset = frame * 32;
|
|
117
|
+
let max = -Infinity;
|
|
118
|
+
for (let label = 0; label < 32; label += 1)
|
|
119
|
+
max = Math.max(max, logits[offset + label] ?? -Infinity);
|
|
120
|
+
let sum = 0;
|
|
121
|
+
for (let label = 0; label < 32; label += 1)
|
|
122
|
+
sum += Math.exp((logits[offset + label] ?? -Infinity) - max);
|
|
123
|
+
const divisor = max + Math.log(sum);
|
|
124
|
+
for (let label = 0; label < 32; label += 1)
|
|
125
|
+
result[offset + label] = (logits[offset + label] ?? -Infinity) - divisor;
|
|
126
|
+
}
|
|
127
|
+
return result;
|
|
128
|
+
}
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
import { mkdir, mkdtemp, rm } from "node:fs/promises";
|
|
2
|
+
import { join } from "node:path";
|
|
3
|
+
import { decodeAudio } from "./audio.js";
|
|
4
|
+
import { prepareModel } from "./cache.js";
|
|
5
|
+
import { claimWorker } from "./lock.js";
|
|
6
|
+
import { runAlignmentWorker } from "./runner.js";
|
|
7
|
+
import { speechWords } from "./text.js";
|
|
8
|
+
export const alignSubtitles = async (request) => {
|
|
9
|
+
request.signal.throwIfAborted();
|
|
10
|
+
speechWords(request.text);
|
|
11
|
+
await mkdir(request.cacheDir, { recursive: true, mode: 0o700 });
|
|
12
|
+
const release = await claimWorker(request.cacheDir, request.signal);
|
|
13
|
+
try {
|
|
14
|
+
const modelPath = await prepareModel({
|
|
15
|
+
cacheDir: request.cacheDir,
|
|
16
|
+
signal: request.signal,
|
|
17
|
+
onProgress: (current, total) => request.onProgress?.(Math.round((current / total) * 20), 100),
|
|
18
|
+
});
|
|
19
|
+
request.onProgress?.(20, 100);
|
|
20
|
+
const working = await mkdtemp(join(request.cacheDir, ".alignment-audio-"));
|
|
21
|
+
try {
|
|
22
|
+
const pcmPath = join(working, "audio.f32");
|
|
23
|
+
await decodeAudio(request.ffmpeg, request.audioPath, pcmPath, request.signal);
|
|
24
|
+
request.onProgress?.(25, 100);
|
|
25
|
+
return await runAlignmentWorker({ modelPath, pcmPath, text: request.text }, request.signal, (current, total) => request.onProgress?.(25 + Math.round((current / total) * 75), 100));
|
|
26
|
+
}
|
|
27
|
+
finally {
|
|
28
|
+
await rm(working, { recursive: true, force: true });
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
finally {
|
|
32
|
+
await release();
|
|
33
|
+
}
|
|
34
|
+
};
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
import { open, readFile, rm, stat } from "node:fs/promises";
|
|
2
|
+
import { join } from "node:path";
|
|
3
|
+
import { setTimeout } from "node:timers/promises";
|
|
4
|
+
// Models run one at a time per data directory, including requests from separate app
|
|
5
|
+
// processes. A canceled waiter never removes another request's lock.
|
|
6
|
+
export async function claimWorker(cacheDir, signal) {
|
|
7
|
+
const path = join(cacheDir, "alignment-worker.lock");
|
|
8
|
+
for (;;) {
|
|
9
|
+
signal.throwIfAborted();
|
|
10
|
+
try {
|
|
11
|
+
const file = await open(path, "wx", 0o600);
|
|
12
|
+
try {
|
|
13
|
+
try {
|
|
14
|
+
await file.writeFile(String(process.pid));
|
|
15
|
+
}
|
|
16
|
+
finally {
|
|
17
|
+
await file.close();
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
catch (error) {
|
|
21
|
+
await rm(path, { force: true });
|
|
22
|
+
throw error;
|
|
23
|
+
}
|
|
24
|
+
return async () => {
|
|
25
|
+
await rm(path, { force: true });
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
catch (error) {
|
|
29
|
+
if (!(error instanceof Error && "code" in error && error.code === "EEXIST"))
|
|
30
|
+
throw error;
|
|
31
|
+
}
|
|
32
|
+
try {
|
|
33
|
+
const owner = Number(await readFile(path, "utf8"));
|
|
34
|
+
if ((!Number.isInteger(owner) || owner <= 0) &&
|
|
35
|
+
Date.now() - (await stat(path)).mtimeMs > 10_000) {
|
|
36
|
+
await rm(path, { force: true });
|
|
37
|
+
continue;
|
|
38
|
+
}
|
|
39
|
+
if (Number.isInteger(owner) && owner > 0) {
|
|
40
|
+
try {
|
|
41
|
+
process.kill(owner, 0);
|
|
42
|
+
}
|
|
43
|
+
catch (error) {
|
|
44
|
+
if (error instanceof Error && "code" in error && error.code === "ESRCH") {
|
|
45
|
+
await rm(path, { force: true });
|
|
46
|
+
continue;
|
|
47
|
+
}
|
|
48
|
+
throw error;
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
catch (error) {
|
|
53
|
+
if (!(error instanceof Error && "code" in error && error.code === "ENOENT"))
|
|
54
|
+
throw error;
|
|
55
|
+
}
|
|
56
|
+
await setTimeout(250, undefined, { signal });
|
|
57
|
+
}
|
|
58
|
+
}
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
const small = [
|
|
2
|
+
"ZERO",
|
|
3
|
+
"ONE",
|
|
4
|
+
"TWO",
|
|
5
|
+
"THREE",
|
|
6
|
+
"FOUR",
|
|
7
|
+
"FIVE",
|
|
8
|
+
"SIX",
|
|
9
|
+
"SEVEN",
|
|
10
|
+
"EIGHT",
|
|
11
|
+
"NINE",
|
|
12
|
+
"TEN",
|
|
13
|
+
"ELEVEN",
|
|
14
|
+
"TWELVE",
|
|
15
|
+
"THIRTEEN",
|
|
16
|
+
"FOURTEEN",
|
|
17
|
+
"FIFTEEN",
|
|
18
|
+
"SIXTEEN",
|
|
19
|
+
"SEVENTEEN",
|
|
20
|
+
"EIGHTEEN",
|
|
21
|
+
"NINETEEN",
|
|
22
|
+
];
|
|
23
|
+
const tens = [
|
|
24
|
+
"",
|
|
25
|
+
"",
|
|
26
|
+
"TWENTY",
|
|
27
|
+
"THIRTY",
|
|
28
|
+
"FORTY",
|
|
29
|
+
"FIFTY",
|
|
30
|
+
"SIXTY",
|
|
31
|
+
"SEVENTY",
|
|
32
|
+
"EIGHTY",
|
|
33
|
+
"NINETY",
|
|
34
|
+
];
|
|
35
|
+
const ordinal = {
|
|
36
|
+
ONE: "FIRST",
|
|
37
|
+
TWO: "SECOND",
|
|
38
|
+
THREE: "THIRD",
|
|
39
|
+
FIVE: "FIFTH",
|
|
40
|
+
EIGHT: "EIGHTH",
|
|
41
|
+
NINE: "NINTH",
|
|
42
|
+
TWELVE: "TWELFTH",
|
|
43
|
+
};
|
|
44
|
+
export function cardinal(value) {
|
|
45
|
+
if (value < 20)
|
|
46
|
+
return small[value] ?? "";
|
|
47
|
+
if (value < 100)
|
|
48
|
+
return `${tens[Math.floor(value / 10)] ?? ""} ${value % 10 === 0 ? "" : cardinal(value % 10)}`.trim();
|
|
49
|
+
for (const [size, name] of [
|
|
50
|
+
[1_000_000_000, "BILLION"],
|
|
51
|
+
[1_000_000, "MILLION"],
|
|
52
|
+
[1000, "THOUSAND"],
|
|
53
|
+
[100, "HUNDRED"],
|
|
54
|
+
]) {
|
|
55
|
+
if (value >= size)
|
|
56
|
+
return `${cardinal(Math.floor(value / size))} ${name}${value % size === 0 ? "" : ` ${cardinal(value % size)}`}`;
|
|
57
|
+
}
|
|
58
|
+
return "";
|
|
59
|
+
}
|
|
60
|
+
export function numberForms(raw) {
|
|
61
|
+
const match = /^(\d[\d,]*)(?:\.(\d+))?(s|st|nd|rd|th)?$/i.exec(raw);
|
|
62
|
+
if (match === null)
|
|
63
|
+
return [raw.replace(/\d/g, (digit) => ` ${small[Number(digit)] ?? ""} `)];
|
|
64
|
+
const digits = (match[1] ?? "").replaceAll(",", "");
|
|
65
|
+
const value = Number(digits);
|
|
66
|
+
if (!Number.isSafeInteger(value) || value >= 1_000_000_000_000)
|
|
67
|
+
return [
|
|
68
|
+
digits
|
|
69
|
+
.split("")
|
|
70
|
+
.map((digit) => small[Number(digit)])
|
|
71
|
+
.join(" "),
|
|
72
|
+
];
|
|
73
|
+
const fraction = match[2];
|
|
74
|
+
const suffix = match[3]?.toLowerCase();
|
|
75
|
+
const decimal = fraction === undefined
|
|
76
|
+
? ""
|
|
77
|
+
: ` POINT ${fraction
|
|
78
|
+
.split("")
|
|
79
|
+
.map((digit) => small[Number(digit)])
|
|
80
|
+
.join(" ")}`;
|
|
81
|
+
const ordinary = `${cardinal(value)}${decimal}`;
|
|
82
|
+
const year = value >= 1900 && value <= 2099 && value % 100 >= 10
|
|
83
|
+
? `${cardinal(Math.floor(value / 100))} ${cardinal(value % 100)}`
|
|
84
|
+
: undefined;
|
|
85
|
+
const forms = year === undefined ? [ordinary] : [year, ordinary];
|
|
86
|
+
if (suffix === "s")
|
|
87
|
+
return forms.map((form) => `${form.replace(/Y$/, "IE")}S`);
|
|
88
|
+
if (suffix !== undefined)
|
|
89
|
+
return forms.map((form) => form.replace(/[A-Z]+$/, (word) => ordinal[word] ?? (word.endsWith("Y") ? `${word.slice(0, -1)}IETH` : `${word}TH`)));
|
|
90
|
+
return forms;
|
|
91
|
+
}
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
export const workerInput = z.object({
|
|
3
|
+
modelPath: z.string(),
|
|
4
|
+
pcmPath: z.string(),
|
|
5
|
+
text: z.string(),
|
|
6
|
+
});
|
|
7
|
+
export const workerMessage = z.discriminatedUnion("type", [
|
|
8
|
+
z.object({
|
|
9
|
+
type: z.literal("progress"),
|
|
10
|
+
current: z.number().finite().nonnegative(),
|
|
11
|
+
total: z.number().finite().positive(),
|
|
12
|
+
}),
|
|
13
|
+
z.object({
|
|
14
|
+
type: z.literal("done"),
|
|
15
|
+
words: z.array(z.object({
|
|
16
|
+
text: z.string(),
|
|
17
|
+
start: z.number().finite().nonnegative(),
|
|
18
|
+
end: z.number().finite().nonnegative(),
|
|
19
|
+
confidence: z.number().finite().min(0).max(1).optional(),
|
|
20
|
+
})),
|
|
21
|
+
}),
|
|
22
|
+
z.object({ type: z.literal("error"), message: z.string() }),
|
|
23
|
+
]);
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
// Compare acoustic recognition with the already aligned transcript. This catches extra
|
|
2
|
+
// speech and missing phrases even when a few forced words individually score well.
|
|
3
|
+
export function agreesWithSpeech(expected, observed) {
|
|
4
|
+
const left = expected.toUpperCase().replace(/[^A-Z']/g, "");
|
|
5
|
+
const right = observed.toUpperCase().replace(/[^A-Z']/g, "");
|
|
6
|
+
if (left.length === 0 || right.length === 0)
|
|
7
|
+
return false;
|
|
8
|
+
let previous = Uint16Array.from({ length: right.length + 1 }, (_, index) => index);
|
|
9
|
+
for (let row = 1; row <= left.length; row += 1) {
|
|
10
|
+
const current = new Uint16Array(right.length + 1);
|
|
11
|
+
current[0] = row;
|
|
12
|
+
for (let column = 1; column <= right.length; column += 1)
|
|
13
|
+
current[column] = Math.min((previous[column] ?? 0) + 1, (current[column - 1] ?? 0) + 1, (previous[column - 1] ?? 0) + (left[row - 1] === right[column - 1] ? 0 : 1));
|
|
14
|
+
previous = current;
|
|
15
|
+
}
|
|
16
|
+
return (previous[right.length] ?? Infinity) / Math.max(left.length, right.length) <= 0.42;
|
|
17
|
+
}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
import { fork } from "node:child_process";
|
|
2
|
+
import { workerMessage } from "./protocol.js";
|
|
3
|
+
export async function runAlignmentWorker(input, signal, onProgress, worker = new URL("./worker.js", import.meta.url)) {
|
|
4
|
+
signal.throwIfAborted();
|
|
5
|
+
return new Promise((resolve, reject) => {
|
|
6
|
+
const options = {
|
|
7
|
+
stdio: ["ignore", "ignore", "pipe", "ipc"],
|
|
8
|
+
windowsHide: true,
|
|
9
|
+
execArgv: [],
|
|
10
|
+
serialization: "advanced",
|
|
11
|
+
};
|
|
12
|
+
const child = fork(worker, [], options);
|
|
13
|
+
let finishing = false;
|
|
14
|
+
let failure;
|
|
15
|
+
let output = [];
|
|
16
|
+
let stderr = "";
|
|
17
|
+
const finish = (error, words) => {
|
|
18
|
+
if (finishing)
|
|
19
|
+
return;
|
|
20
|
+
finishing = true;
|
|
21
|
+
failure = error;
|
|
22
|
+
output = words ?? [];
|
|
23
|
+
signal.removeEventListener("abort", abort);
|
|
24
|
+
child.kill("SIGKILL");
|
|
25
|
+
};
|
|
26
|
+
const abort = () => finish(new Error("Subtitle alignment was canceled."));
|
|
27
|
+
signal.addEventListener("abort", abort, { once: true });
|
|
28
|
+
child.stderr?.on("data", (data) => {
|
|
29
|
+
stderr = (stderr + data.toString("utf8")).slice(-2000);
|
|
30
|
+
});
|
|
31
|
+
child.on("error", (error) => finish(error));
|
|
32
|
+
// Wait for close, not only the IPC answer: Windows still owns the decoded audio
|
|
33
|
+
// handle until the worker has actually exited, so early cleanup can fail with EPERM.
|
|
34
|
+
child.on("close", (code) => {
|
|
35
|
+
if (!finishing)
|
|
36
|
+
finish(new Error(`Local subtitle alignment stopped (exit ${String(code)}). ${stderr}`.trim()));
|
|
37
|
+
if (failure !== undefined)
|
|
38
|
+
reject(failure);
|
|
39
|
+
else
|
|
40
|
+
resolve(output);
|
|
41
|
+
});
|
|
42
|
+
child.on("message", (raw) => {
|
|
43
|
+
if (finishing)
|
|
44
|
+
return;
|
|
45
|
+
const parsed = workerMessage.safeParse(raw);
|
|
46
|
+
if (!parsed.success) {
|
|
47
|
+
finish(new Error("The local subtitle worker returned invalid timing data."));
|
|
48
|
+
return;
|
|
49
|
+
}
|
|
50
|
+
const message = parsed.data;
|
|
51
|
+
if (message.type === "progress") {
|
|
52
|
+
try {
|
|
53
|
+
onProgress?.(message.current, message.total);
|
|
54
|
+
}
|
|
55
|
+
catch (error) {
|
|
56
|
+
finish(error instanceof Error ? error : new Error(String(error)));
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
else if (message.type === "error")
|
|
60
|
+
finish(new Error(message.message));
|
|
61
|
+
else
|
|
62
|
+
finish(undefined, message.words.map(({ confidence, ...word }) => ({
|
|
63
|
+
...word,
|
|
64
|
+
...(confidence === undefined ? {} : { confidence }),
|
|
65
|
+
})));
|
|
66
|
+
});
|
|
67
|
+
if (signal.aborted) {
|
|
68
|
+
abort();
|
|
69
|
+
return;
|
|
70
|
+
}
|
|
71
|
+
child.send(input, (error) => {
|
|
72
|
+
if (error !== null)
|
|
73
|
+
finish(error);
|
|
74
|
+
});
|
|
75
|
+
});
|
|
76
|
+
}
|