@gentbajko/slopify 0.8.1 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -22,7 +22,7 @@ export const alignSubtitles = async (request) => {
22
22
  const pcmPath = join(working, "audio.f32");
23
23
  await decodeAudio(request.ffmpeg, request.audioPath, pcmPath, request.signal);
24
24
  request.onProgress?.(25, 100);
25
- return await runAlignmentWorker({ modelPath, pcmPath, text: request.text }, request.signal, (current, total) => request.onProgress?.(25 + Math.round((current / total) * 75), 100));
25
+ return await runAlignmentWorker({ modelPath, pcmPath, text: request.text }, request.signal, (current, total) => request.onProgress?.(25 + Math.round((current / total) * 75), 100), undefined, request.onOmission);
26
26
  }
27
27
  finally {
28
28
  await rm(working, { recursive: true, force: true });
@@ -4,6 +4,10 @@ export const workerInput = z.object({
4
4
  pcmPath: z.string(),
5
5
  text: z.string(),
6
6
  });
7
+ export const omissionSchema = z.object({
8
+ start: z.number().finite().nonnegative(),
9
+ text: z.string().min(1).max(10000),
10
+ });
7
11
  export const workerMessage = z.discriminatedUnion("type", [
8
12
  z.object({
9
13
  type: z.literal("progress"),
@@ -19,5 +23,6 @@ export const workerMessage = z.discriminatedUnion("type", [
19
23
  confidence: z.number().finite().min(0).max(1).optional(),
20
24
  })),
21
25
  }),
26
+ omissionSchema.extend({ type: z.literal("omission") }),
22
27
  z.object({ type: z.literal("error"), message: z.string() }),
23
28
  ]);
@@ -1,6 +1,6 @@
1
1
  // Compare acoustic recognition with the already aligned transcript. This catches extra
2
2
  // speech and missing phrases even when a few forced words individually score well.
3
- export function agreesWithSpeech(expected, observed) {
3
+ export function agreesWithSpeech(expected, observed, maximumError = 0.42) {
4
4
  const left = expected.toUpperCase().replace(/[^A-Z']/g, "");
5
5
  const right = observed.toUpperCase().replace(/[^A-Z']/g, "");
6
6
  if (left.length === 0 || right.length === 0)
@@ -13,5 +13,5 @@ export function agreesWithSpeech(expected, observed) {
13
13
  current[column] = Math.min((previous[column] ?? 0) + 1, (current[column - 1] ?? 0) + 1, (previous[column - 1] ?? 0) + (left[row - 1] === right[column - 1] ? 0 : 1));
14
14
  previous = current;
15
15
  }
16
- return (previous[right.length] ?? Infinity) / Math.max(left.length, right.length) <= 0.42;
16
+ return (previous[right.length] ?? Infinity) / Math.max(left.length, right.length) <= maximumError;
17
17
  }
@@ -1,6 +1,6 @@
1
1
  import { fork } from "node:child_process";
2
2
  import { workerMessage } from "./protocol.js";
3
- export async function runAlignmentWorker(input, signal, onProgress, worker = new URL("./worker.js", import.meta.url)) {
3
+ export async function runAlignmentWorker(input, signal, onProgress, worker = new URL("./worker.js", import.meta.url), onOmission) {
4
4
  signal.throwIfAborted();
5
5
  return new Promise((resolve, reject) => {
6
6
  const options = {
@@ -48,7 +48,15 @@ export async function runAlignmentWorker(input, signal, onProgress, worker = new
48
48
  return;
49
49
  }
50
50
  const message = parsed.data;
51
- if (message.type === "progress") {
51
+ if (message.type === "omission") {
52
+ try {
53
+ onOmission?.({ start: message.start, text: message.text });
54
+ }
55
+ catch (error) {
56
+ finish(error instanceof Error ? error : new Error(String(error)));
57
+ }
58
+ }
59
+ else if (message.type === "progress") {
52
60
  try {
53
61
  onProgress?.(message.current, message.total);
54
62
  }
@@ -0,0 +1,60 @@
1
+ import { alignWindow, frameSeconds, mismatch } from "./ctc.js";
2
+ import { agreesWithSpeech } from "./quality.js";
3
+ import { letters } from "./vocabulary.js";
4
+ // Recovery can only omit a short transcript prefix, never insert guessed words or times.
5
+ export function alignSpeechWindow(logits, frames, candidate, complete, cutoff, skipBudget) {
6
+ try {
7
+ return { words: accepted(logits, frames, candidate, complete, cutoff), skipped: 0 };
8
+ }
9
+ catch (error) {
10
+ if (!(error instanceof Error) || error.message !== mismatch)
11
+ throw error;
12
+ }
13
+ const heard = greedy(logits, frames).split(/\s+/);
14
+ for (let skipped = 1; skipped <= Math.min(40, skipBudget, candidate.length - 4); skipped += 1) {
15
+ const remaining = candidate.slice(skipped);
16
+ const anchor = remaining
17
+ .slice(0, 4)
18
+ .map((word) => word.spoken)
19
+ .join(" ");
20
+ if (anchor.replace(/[^A-Z]/g, "").length < 20 ||
21
+ !agreesWithSpeech(anchor, heard.slice(0, anchor.split(/\s+/).length).join(" "), 0.2))
22
+ continue;
23
+ try {
24
+ const words = accepted(logits, frames, remaining, complete, cutoff);
25
+ if (words.length < 4 || words.slice(0, 4).some((word) => (word.confidence ?? 0) < 0.75))
26
+ continue;
27
+ return { words, skipped };
28
+ }
29
+ catch (error) {
30
+ if (!(error instanceof Error) || error.message !== mismatch)
31
+ throw error;
32
+ }
33
+ }
34
+ throw new Error(mismatch);
35
+ }
36
+ function accepted(logits, frames, candidate, complete, cutoff) {
37
+ const words = alignWindow(logits, frames, candidate, complete).words.filter((word) => word.end <= cutoff);
38
+ const last = words.at(-1);
39
+ if (last === undefined ||
40
+ !agreesWithSpeech(candidate
41
+ .slice(0, words.length)
42
+ .map((word) => word.spoken)
43
+ .join(" "), greedy(logits, Math.min(frames, Math.ceil(last.end / frameSeconds)))))
44
+ throw new Error(mismatch);
45
+ return words;
46
+ }
47
+ export function greedy(logits, frames) {
48
+ let previous = -1;
49
+ let text = "";
50
+ for (let frame = 0; frame < frames; frame += 1) {
51
+ let best = 0;
52
+ for (let label = 1; label < 32; label += 1)
53
+ if ((logits[frame * 32 + label] ?? -Infinity) > (logits[frame * 32 + best] ?? -Infinity))
54
+ best = label;
55
+ if (best !== previous && best !== 0)
56
+ text += letters[best] ?? "";
57
+ previous = best;
58
+ }
59
+ return text.trim().replace(/\s+/g, " ");
60
+ }
@@ -1,10 +1,9 @@
1
1
  import { open, readFile, stat } from "node:fs/promises";
2
2
  import * as ort from "onnxruntime-web/wasm";
3
- import { alignWindow, frameSeconds, mismatch } from "./ctc.js";
3
+ import { mismatch } from "./ctc.js";
4
4
  import { workerInput } from "./protocol.js";
5
- import { agreesWithSpeech } from "./quality.js";
6
5
  import { speechWords } from "./text.js";
7
- import { letters } from "./vocabulary.js";
6
+ import { alignSpeechWindow, greedy } from "./window.js";
8
7
  const sampleRate = 16000;
9
8
  const windowSeconds = 12;
10
9
  const overlapSeconds = 2;
@@ -31,6 +30,8 @@ async function run(input) {
31
30
  const source = speechWords(input.text);
32
31
  const output = [];
33
32
  let cursor = 0;
33
+ let omitted = 0;
34
+ const omissionBudget = Math.min(60, Math.floor(source.length * 0.05));
34
35
  let sampleAt = 0;
35
36
  while (sampleAt < totalSamples && cursor < source.length) {
36
37
  const count = Math.min(windowSeconds * sampleRate, totalSamples - sampleAt);
@@ -62,18 +63,24 @@ async function run(input) {
62
63
  const candidate = candidates(source, cursor, observed);
63
64
  const finalWindow = sampleAt + count >= totalSamples;
64
65
  const complete = finalWindow && cursor + candidate.length === source.length;
65
- const aligned = alignWindow(logits.data, frames, candidate, complete).words;
66
66
  const cutoff = finalWindow ? count / sampleRate : count / sampleRate - overlapSeconds;
67
- const accepted = aligned.filter((word) => word.end <= cutoff);
67
+ const recovered = alignSpeechWindow(logits.data, frames, candidate, complete, cutoff, cursor === 0 ? 0 : omissionBudget - omitted);
68
+ const accepted = recovered.words;
68
69
  const last = accepted.at(-1);
69
70
  if (last === undefined)
70
71
  throw new Error(mismatch);
71
- const heard = greedy(logits.data, Math.min(frames, Math.ceil(last.end / frameSeconds)));
72
- if (!agreesWithSpeech(candidate
73
- .slice(0, accepted.length)
74
- .map((word) => word.spoken)
75
- .join(" "), heard))
76
- throw new Error(mismatch);
72
+ if (recovered.skipped > 0) {
73
+ omitted += recovered.skipped;
74
+ send({
75
+ type: "omission",
76
+ start: sampleAt / sampleRate,
77
+ text: candidate
78
+ .slice(0, recovered.skipped)
79
+ .map((word) => word.text)
80
+ .join(" "),
81
+ });
82
+ cursor += recovered.skipped;
83
+ }
77
84
  const offset = sampleAt / sampleRate;
78
85
  output.push(...accepted.map((word) => ({
79
86
  ...word,
@@ -147,17 +154,3 @@ function normalize(audio) {
147
154
  const divisor = Math.sqrt(variance + 1e-7);
148
155
  return Float32Array.from(audio, (value) => (value - mean) / divisor);
149
156
  }
150
- function greedy(logits, frames) {
151
- let previous = -1;
152
- let text = "";
153
- for (let frame = 0; frame < frames; frame += 1) {
154
- let best = 0;
155
- for (let label = 1; label < 32; label += 1)
156
- if ((logits[frame * 32 + label] ?? -Infinity) > (logits[frame * 32 + best] ?? -Infinity))
157
- best = label;
158
- if (best !== previous && best !== 0)
159
- text += letters[best] ?? "";
160
- previous = best;
161
- }
162
- return text.trim().replace(/\s+/g, " ");
163
- }
@@ -1,6 +1,9 @@
1
1
  import { z } from "zod";
2
2
  import { outputRoles, stagedFileStates, stageKinds } from "./model.js";
3
3
  const metaSchema = z.object({
4
+ subtitleOmissions: z
5
+ .array(z.object({ start: z.number().finite().nonnegative(), text: z.string() }))
6
+ .optional(),
4
7
  subtitlesMode: z.enum(["off", "files", "burn-in"]).optional(),
5
8
  promptName: z.string().optional(),
6
9
  prompt: z.string().optional(),
@@ -28,7 +28,14 @@ const fontSchema = z.object({
28
28
  assName: z.string(),
29
29
  extension: z.enum([".ttf", ".otf", ".ttc"]),
30
30
  });
31
- const cacheSchema = z.object({ key: z.string(), words: z.array(wordSchema), font: fontSchema });
31
+ const cacheSchema = z.object({
32
+ key: z.string(),
33
+ words: z.array(wordSchema),
34
+ font: fontSchema,
35
+ omissions: z
36
+ .array(z.object({ start: z.number().finite().nonnegative(), text: z.string() }))
37
+ .default([]),
38
+ });
32
39
  // Preparation writes into a new directory; only writeExport commits its output rows.
33
40
  // The old caption files/font/timing stay usable if alignment or rendering fails.
34
41
  export async function prepareSubtitles(deps, context, audio, frame) {
@@ -52,7 +59,8 @@ export async function prepareSubtitles(deps, context, audio, frame) {
52
59
  const directory = mkdtempSync(join(dir, "captions-"));
53
60
  try {
54
61
  const font = await snapshotFont(deps, projectId, outputs, config.fontId, cache, directory);
55
- const words = cache?.key === key ? cache.words : await alignSegments(deps, context, segments);
62
+ const omissions = cache?.key === key ? [...cache.omissions] : [];
63
+ const words = cache?.key === key ? cache.words : await alignSegments(deps, context, segments, omissions);
56
64
  context.signal.throwIfAborted();
57
65
  const cues = captionCues(words);
58
66
  if (cues.length === 0)
@@ -65,7 +73,7 @@ export async function prepareSubtitles(deps, context, audio, frame) {
65
73
  fontName: font.assName,
66
74
  position: config.position,
67
75
  }), { mode: 0o600 });
68
- writeFileSync(join(directory, "subtitles.json"), JSON.stringify({ key, words, font }), {
76
+ writeFileSync(join(directory, "subtitles.json"), JSON.stringify({ key, words, font, omissions }), {
69
77
  mode: 0o600,
70
78
  });
71
79
  const files = [
@@ -77,6 +85,7 @@ export async function prepareSubtitles(deps, context, audio, frame) {
77
85
  ];
78
86
  return {
79
87
  directory,
88
+ omissions,
80
89
  burnIn: config.mode === "burn-in" && project.config.sources.video !== "off",
81
90
  assets: files.map(([role, path]) => ({
82
91
  role,
@@ -91,7 +100,7 @@ export async function prepareSubtitles(deps, context, audio, frame) {
91
100
  }
92
101
  async function timingKey(segments, signal) {
93
102
  // Bump when alignment normalization/model changes. Hash file contents, not timestamps.
94
- const hash = createHash("sha256").update("wav2vec2-en-a19f851-v1");
103
+ const hash = createHash("sha256").update("wav2vec2-en-a19f851-v2-omissions");
95
104
  for (const segment of segments) {
96
105
  signal.throwIfAborted();
97
106
  hash.update(JSON.stringify({ kind: segment.kind, seconds: segment.seconds, text: segment.text }));
@@ -103,7 +112,7 @@ async function timingKey(segments, signal) {
103
112
  }
104
113
  return hash.digest("hex");
105
114
  }
106
- async function alignSegments(deps, context, segments) {
115
+ async function alignSegments(deps, context, segments, omissions) {
107
116
  if (deps.alignSubtitles === undefined)
108
117
  throw new Error("Local subtitle alignment is unavailable in this build.");
109
118
  const words = [];
@@ -114,6 +123,7 @@ async function alignSegments(deps, context, segments) {
114
123
  if (segment.path !== null) {
115
124
  const aligned = await deps.alignSubtitles({
116
125
  audioPath: segment.path,
126
+ onOmission: (omission) => omissions.push({ ...omission, start: omission.start + offset }),
117
127
  text: segment.text,
118
128
  cacheDir: join(deps.paths.dataDir, "models", "english-subtitles"),
119
129
  ffmpeg: deps.ffmpeg,
@@ -90,6 +90,9 @@ export async function writeExport(deps, context, output) {
90
90
  }
91
91
  store(deps, projectId, "render_params", "render.json", null);
92
92
  store(deps, projectId, output.role, output.filename, totalMs, {
93
+ ...(output.subtitles?.omissions?.length
94
+ ? { subtitleOmissions: output.subtitles.omissions }
95
+ : {}),
93
96
  subtitlesMode: output.subtitles === undefined ? "off" : output.subtitles.burnIn ? "burn-in" : "files",
94
97
  });
95
98
  for (const asset of output.subtitles?.assets ?? [])