@gentbajko/slopify 0.8.1 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/alignment/index.js +1 -1
- package/dist/adapters/alignment/protocol.js +5 -0
- package/dist/adapters/alignment/quality.js +2 -2
- package/dist/adapters/alignment/runner.js +10 -2
- package/dist/adapters/alignment/window.js +60 -0
- package/dist/adapters/alignment/worker.js +18 -25
- package/dist/slices/storage/repo.js +3 -0
- package/dist/slices/subtitles/prepare.js +15 -5
- package/dist/slices/video/write-export.js +3 -0
- package/dist/web/assets/{index-GszhBVNn.js → index-C7PasGML.js} +2 -2
- package/dist/web/index.html +1 -1
- package/package.json +1 -1
|
@@ -22,7 +22,7 @@ export const alignSubtitles = async (request) => {
|
|
|
22
22
|
const pcmPath = join(working, "audio.f32");
|
|
23
23
|
await decodeAudio(request.ffmpeg, request.audioPath, pcmPath, request.signal);
|
|
24
24
|
request.onProgress?.(25, 100);
|
|
25
|
-
return await runAlignmentWorker({ modelPath, pcmPath, text: request.text }, request.signal, (current, total) => request.onProgress?.(25 + Math.round((current / total) * 75), 100));
|
|
25
|
+
return await runAlignmentWorker({ modelPath, pcmPath, text: request.text }, request.signal, (current, total) => request.onProgress?.(25 + Math.round((current / total) * 75), 100), undefined, request.onOmission);
|
|
26
26
|
}
|
|
27
27
|
finally {
|
|
28
28
|
await rm(working, { recursive: true, force: true });
|
|
@@ -4,6 +4,10 @@ export const workerInput = z.object({
|
|
|
4
4
|
pcmPath: z.string(),
|
|
5
5
|
text: z.string(),
|
|
6
6
|
});
|
|
7
|
+
export const omissionSchema = z.object({
|
|
8
|
+
start: z.number().finite().nonnegative(),
|
|
9
|
+
text: z.string().min(1).max(10000),
|
|
10
|
+
});
|
|
7
11
|
export const workerMessage = z.discriminatedUnion("type", [
|
|
8
12
|
z.object({
|
|
9
13
|
type: z.literal("progress"),
|
|
@@ -19,5 +23,6 @@ export const workerMessage = z.discriminatedUnion("type", [
|
|
|
19
23
|
confidence: z.number().finite().min(0).max(1).optional(),
|
|
20
24
|
})),
|
|
21
25
|
}),
|
|
26
|
+
omissionSchema.extend({ type: z.literal("omission") }),
|
|
22
27
|
z.object({ type: z.literal("error"), message: z.string() }),
|
|
23
28
|
]);
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
// Compare acoustic recognition with the already aligned transcript. This catches extra
|
|
2
2
|
// speech and missing phrases even when a few forced words individually score well.
|
|
3
|
-
export function agreesWithSpeech(expected, observed) {
|
|
3
|
+
export function agreesWithSpeech(expected, observed, maximumError = 0.42) {
|
|
4
4
|
const left = expected.toUpperCase().replace(/[^A-Z']/g, "");
|
|
5
5
|
const right = observed.toUpperCase().replace(/[^A-Z']/g, "");
|
|
6
6
|
if (left.length === 0 || right.length === 0)
|
|
@@ -13,5 +13,5 @@ export function agreesWithSpeech(expected, observed) {
|
|
|
13
13
|
current[column] = Math.min((previous[column] ?? 0) + 1, (current[column - 1] ?? 0) + 1, (previous[column - 1] ?? 0) + (left[row - 1] === right[column - 1] ? 0 : 1));
|
|
14
14
|
previous = current;
|
|
15
15
|
}
|
|
16
|
-
return (previous[right.length] ?? Infinity) / Math.max(left.length, right.length) <=
|
|
16
|
+
return (previous[right.length] ?? Infinity) / Math.max(left.length, right.length) <= maximumError;
|
|
17
17
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { fork } from "node:child_process";
|
|
2
2
|
import { workerMessage } from "./protocol.js";
|
|
3
|
-
export async function runAlignmentWorker(input, signal, onProgress, worker = new URL("./worker.js", import.meta.url)) {
|
|
3
|
+
export async function runAlignmentWorker(input, signal, onProgress, worker = new URL("./worker.js", import.meta.url), onOmission) {
|
|
4
4
|
signal.throwIfAborted();
|
|
5
5
|
return new Promise((resolve, reject) => {
|
|
6
6
|
const options = {
|
|
@@ -48,7 +48,15 @@ export async function runAlignmentWorker(input, signal, onProgress, worker = new
|
|
|
48
48
|
return;
|
|
49
49
|
}
|
|
50
50
|
const message = parsed.data;
|
|
51
|
-
if (message.type === "
|
|
51
|
+
if (message.type === "omission") {
|
|
52
|
+
try {
|
|
53
|
+
onOmission?.({ start: message.start, text: message.text });
|
|
54
|
+
}
|
|
55
|
+
catch (error) {
|
|
56
|
+
finish(error instanceof Error ? error : new Error(String(error)));
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
else if (message.type === "progress") {
|
|
52
60
|
try {
|
|
53
61
|
onProgress?.(message.current, message.total);
|
|
54
62
|
}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import { alignWindow, frameSeconds, mismatch } from "./ctc.js";
|
|
2
|
+
import { agreesWithSpeech } from "./quality.js";
|
|
3
|
+
import { letters } from "./vocabulary.js";
|
|
4
|
+
// Recovery can only omit a short transcript prefix, never insert guessed words or times.
|
|
5
|
+
export function alignSpeechWindow(logits, frames, candidate, complete, cutoff, skipBudget) {
|
|
6
|
+
try {
|
|
7
|
+
return { words: accepted(logits, frames, candidate, complete, cutoff), skipped: 0 };
|
|
8
|
+
}
|
|
9
|
+
catch (error) {
|
|
10
|
+
if (!(error instanceof Error) || error.message !== mismatch)
|
|
11
|
+
throw error;
|
|
12
|
+
}
|
|
13
|
+
const heard = greedy(logits, frames).split(/\s+/);
|
|
14
|
+
for (let skipped = 1; skipped <= Math.min(40, skipBudget, candidate.length - 4); skipped += 1) {
|
|
15
|
+
const remaining = candidate.slice(skipped);
|
|
16
|
+
const anchor = remaining
|
|
17
|
+
.slice(0, 4)
|
|
18
|
+
.map((word) => word.spoken)
|
|
19
|
+
.join(" ");
|
|
20
|
+
if (anchor.replace(/[^A-Z]/g, "").length < 20 ||
|
|
21
|
+
!agreesWithSpeech(anchor, heard.slice(0, anchor.split(/\s+/).length).join(" "), 0.2))
|
|
22
|
+
continue;
|
|
23
|
+
try {
|
|
24
|
+
const words = accepted(logits, frames, remaining, complete, cutoff);
|
|
25
|
+
if (words.length < 4 || words.slice(0, 4).some((word) => (word.confidence ?? 0) < 0.75))
|
|
26
|
+
continue;
|
|
27
|
+
return { words, skipped };
|
|
28
|
+
}
|
|
29
|
+
catch (error) {
|
|
30
|
+
if (!(error instanceof Error) || error.message !== mismatch)
|
|
31
|
+
throw error;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
throw new Error(mismatch);
|
|
35
|
+
}
|
|
36
|
+
function accepted(logits, frames, candidate, complete, cutoff) {
|
|
37
|
+
const words = alignWindow(logits, frames, candidate, complete).words.filter((word) => word.end <= cutoff);
|
|
38
|
+
const last = words.at(-1);
|
|
39
|
+
if (last === undefined ||
|
|
40
|
+
!agreesWithSpeech(candidate
|
|
41
|
+
.slice(0, words.length)
|
|
42
|
+
.map((word) => word.spoken)
|
|
43
|
+
.join(" "), greedy(logits, Math.min(frames, Math.ceil(last.end / frameSeconds)))))
|
|
44
|
+
throw new Error(mismatch);
|
|
45
|
+
return words;
|
|
46
|
+
}
|
|
47
|
+
export function greedy(logits, frames) {
|
|
48
|
+
let previous = -1;
|
|
49
|
+
let text = "";
|
|
50
|
+
for (let frame = 0; frame < frames; frame += 1) {
|
|
51
|
+
let best = 0;
|
|
52
|
+
for (let label = 1; label < 32; label += 1)
|
|
53
|
+
if ((logits[frame * 32 + label] ?? -Infinity) > (logits[frame * 32 + best] ?? -Infinity))
|
|
54
|
+
best = label;
|
|
55
|
+
if (best !== previous && best !== 0)
|
|
56
|
+
text += letters[best] ?? "";
|
|
57
|
+
previous = best;
|
|
58
|
+
}
|
|
59
|
+
return text.trim().replace(/\s+/g, " ");
|
|
60
|
+
}
|
|
@@ -1,10 +1,9 @@
|
|
|
1
1
|
import { open, readFile, stat } from "node:fs/promises";
|
|
2
2
|
import * as ort from "onnxruntime-web/wasm";
|
|
3
|
-
import {
|
|
3
|
+
import { mismatch } from "./ctc.js";
|
|
4
4
|
import { workerInput } from "./protocol.js";
|
|
5
|
-
import { agreesWithSpeech } from "./quality.js";
|
|
6
5
|
import { speechWords } from "./text.js";
|
|
7
|
-
import {
|
|
6
|
+
import { alignSpeechWindow, greedy } from "./window.js";
|
|
8
7
|
const sampleRate = 16000;
|
|
9
8
|
const windowSeconds = 12;
|
|
10
9
|
const overlapSeconds = 2;
|
|
@@ -31,6 +30,8 @@ async function run(input) {
|
|
|
31
30
|
const source = speechWords(input.text);
|
|
32
31
|
const output = [];
|
|
33
32
|
let cursor = 0;
|
|
33
|
+
let omitted = 0;
|
|
34
|
+
const omissionBudget = Math.min(60, Math.floor(source.length * 0.05));
|
|
34
35
|
let sampleAt = 0;
|
|
35
36
|
while (sampleAt < totalSamples && cursor < source.length) {
|
|
36
37
|
const count = Math.min(windowSeconds * sampleRate, totalSamples - sampleAt);
|
|
@@ -62,18 +63,24 @@ async function run(input) {
|
|
|
62
63
|
const candidate = candidates(source, cursor, observed);
|
|
63
64
|
const finalWindow = sampleAt + count >= totalSamples;
|
|
64
65
|
const complete = finalWindow && cursor + candidate.length === source.length;
|
|
65
|
-
const aligned = alignWindow(logits.data, frames, candidate, complete).words;
|
|
66
66
|
const cutoff = finalWindow ? count / sampleRate : count / sampleRate - overlapSeconds;
|
|
67
|
-
const
|
|
67
|
+
const recovered = alignSpeechWindow(logits.data, frames, candidate, complete, cutoff, cursor === 0 ? 0 : omissionBudget - omitted);
|
|
68
|
+
const accepted = recovered.words;
|
|
68
69
|
const last = accepted.at(-1);
|
|
69
70
|
if (last === undefined)
|
|
70
71
|
throw new Error(mismatch);
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
72
|
+
if (recovered.skipped > 0) {
|
|
73
|
+
omitted += recovered.skipped;
|
|
74
|
+
send({
|
|
75
|
+
type: "omission",
|
|
76
|
+
start: sampleAt / sampleRate,
|
|
77
|
+
text: candidate
|
|
78
|
+
.slice(0, recovered.skipped)
|
|
79
|
+
.map((word) => word.text)
|
|
80
|
+
.join(" "),
|
|
81
|
+
});
|
|
82
|
+
cursor += recovered.skipped;
|
|
83
|
+
}
|
|
77
84
|
const offset = sampleAt / sampleRate;
|
|
78
85
|
output.push(...accepted.map((word) => ({
|
|
79
86
|
...word,
|
|
@@ -147,17 +154,3 @@ function normalize(audio) {
|
|
|
147
154
|
const divisor = Math.sqrt(variance + 1e-7);
|
|
148
155
|
return Float32Array.from(audio, (value) => (value - mean) / divisor);
|
|
149
156
|
}
|
|
150
|
-
function greedy(logits, frames) {
|
|
151
|
-
let previous = -1;
|
|
152
|
-
let text = "";
|
|
153
|
-
for (let frame = 0; frame < frames; frame += 1) {
|
|
154
|
-
let best = 0;
|
|
155
|
-
for (let label = 1; label < 32; label += 1)
|
|
156
|
-
if ((logits[frame * 32 + label] ?? -Infinity) > (logits[frame * 32 + best] ?? -Infinity))
|
|
157
|
-
best = label;
|
|
158
|
-
if (best !== previous && best !== 0)
|
|
159
|
-
text += letters[best] ?? "";
|
|
160
|
-
previous = best;
|
|
161
|
-
}
|
|
162
|
-
return text.trim().replace(/\s+/g, " ");
|
|
163
|
-
}
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
import { z } from "zod";
|
|
2
2
|
import { outputRoles, stagedFileStates, stageKinds } from "./model.js";
|
|
3
3
|
const metaSchema = z.object({
|
|
4
|
+
subtitleOmissions: z
|
|
5
|
+
.array(z.object({ start: z.number().finite().nonnegative(), text: z.string() }))
|
|
6
|
+
.optional(),
|
|
4
7
|
subtitlesMode: z.enum(["off", "files", "burn-in"]).optional(),
|
|
5
8
|
promptName: z.string().optional(),
|
|
6
9
|
prompt: z.string().optional(),
|
|
@@ -28,7 +28,14 @@ const fontSchema = z.object({
|
|
|
28
28
|
assName: z.string(),
|
|
29
29
|
extension: z.enum([".ttf", ".otf", ".ttc"]),
|
|
30
30
|
});
|
|
31
|
-
const cacheSchema = z.object({
|
|
31
|
+
const cacheSchema = z.object({
|
|
32
|
+
key: z.string(),
|
|
33
|
+
words: z.array(wordSchema),
|
|
34
|
+
font: fontSchema,
|
|
35
|
+
omissions: z
|
|
36
|
+
.array(z.object({ start: z.number().finite().nonnegative(), text: z.string() }))
|
|
37
|
+
.default([]),
|
|
38
|
+
});
|
|
32
39
|
// Preparation writes into a new directory; only writeExport commits its output rows.
|
|
33
40
|
// The old caption files/font/timing stay usable if alignment or rendering fails.
|
|
34
41
|
export async function prepareSubtitles(deps, context, audio, frame) {
|
|
@@ -52,7 +59,8 @@ export async function prepareSubtitles(deps, context, audio, frame) {
|
|
|
52
59
|
const directory = mkdtempSync(join(dir, "captions-"));
|
|
53
60
|
try {
|
|
54
61
|
const font = await snapshotFont(deps, projectId, outputs, config.fontId, cache, directory);
|
|
55
|
-
const
|
|
62
|
+
const omissions = cache?.key === key ? [...cache.omissions] : [];
|
|
63
|
+
const words = cache?.key === key ? cache.words : await alignSegments(deps, context, segments, omissions);
|
|
56
64
|
context.signal.throwIfAborted();
|
|
57
65
|
const cues = captionCues(words);
|
|
58
66
|
if (cues.length === 0)
|
|
@@ -65,7 +73,7 @@ export async function prepareSubtitles(deps, context, audio, frame) {
|
|
|
65
73
|
fontName: font.assName,
|
|
66
74
|
position: config.position,
|
|
67
75
|
}), { mode: 0o600 });
|
|
68
|
-
writeFileSync(join(directory, "subtitles.json"), JSON.stringify({ key, words, font }), {
|
|
76
|
+
writeFileSync(join(directory, "subtitles.json"), JSON.stringify({ key, words, font, omissions }), {
|
|
69
77
|
mode: 0o600,
|
|
70
78
|
});
|
|
71
79
|
const files = [
|
|
@@ -77,6 +85,7 @@ export async function prepareSubtitles(deps, context, audio, frame) {
|
|
|
77
85
|
];
|
|
78
86
|
return {
|
|
79
87
|
directory,
|
|
88
|
+
omissions,
|
|
80
89
|
burnIn: config.mode === "burn-in" && project.config.sources.video !== "off",
|
|
81
90
|
assets: files.map(([role, path]) => ({
|
|
82
91
|
role,
|
|
@@ -91,7 +100,7 @@ export async function prepareSubtitles(deps, context, audio, frame) {
|
|
|
91
100
|
}
|
|
92
101
|
async function timingKey(segments, signal) {
|
|
93
102
|
// Bump when alignment normalization/model changes. Hash file contents, not timestamps.
|
|
94
|
-
const hash = createHash("sha256").update("wav2vec2-en-a19f851-
|
|
103
|
+
const hash = createHash("sha256").update("wav2vec2-en-a19f851-v2-omissions");
|
|
95
104
|
for (const segment of segments) {
|
|
96
105
|
signal.throwIfAborted();
|
|
97
106
|
hash.update(JSON.stringify({ kind: segment.kind, seconds: segment.seconds, text: segment.text }));
|
|
@@ -103,7 +112,7 @@ async function timingKey(segments, signal) {
|
|
|
103
112
|
}
|
|
104
113
|
return hash.digest("hex");
|
|
105
114
|
}
|
|
106
|
-
async function alignSegments(deps, context, segments) {
|
|
115
|
+
async function alignSegments(deps, context, segments, omissions) {
|
|
107
116
|
if (deps.alignSubtitles === undefined)
|
|
108
117
|
throw new Error("Local subtitle alignment is unavailable in this build.");
|
|
109
118
|
const words = [];
|
|
@@ -114,6 +123,7 @@ async function alignSegments(deps, context, segments) {
|
|
|
114
123
|
if (segment.path !== null) {
|
|
115
124
|
const aligned = await deps.alignSubtitles({
|
|
116
125
|
audioPath: segment.path,
|
|
126
|
+
onOmission: (omission) => omissions.push({ ...omission, start: omission.start + offset }),
|
|
117
127
|
text: segment.text,
|
|
118
128
|
cacheDir: join(deps.paths.dataDir, "models", "english-subtitles"),
|
|
119
129
|
ffmpeg: deps.ffmpeg,
|
|
@@ -90,6 +90,9 @@ export async function writeExport(deps, context, output) {
|
|
|
90
90
|
}
|
|
91
91
|
store(deps, projectId, "render_params", "render.json", null);
|
|
92
92
|
store(deps, projectId, output.role, output.filename, totalMs, {
|
|
93
|
+
...(output.subtitles?.omissions?.length
|
|
94
|
+
? { subtitleOmissions: output.subtitles.omissions }
|
|
95
|
+
: {}),
|
|
93
96
|
subtitlesMode: output.subtitles === undefined ? "off" : output.subtitles.burnIn ? "burn-in" : "files",
|
|
94
97
|
});
|
|
95
98
|
for (const asset of output.subtitles?.assets ?? [])
|