@gentbajko/slopify 0.8.0 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/alignment/index.js +1 -1
- package/dist/adapters/alignment/protocol.js +5 -0
- package/dist/adapters/alignment/quality.js +2 -2
- package/dist/adapters/alignment/runner.js +10 -2
- package/dist/adapters/alignment/window.js +60 -0
- package/dist/adapters/alignment/worker.js +18 -25
- package/dist/edge/http/app.js +2 -0
- package/dist/edge/http/open-folder.js +61 -0
- package/dist/edge/open-folder.js +25 -0
- package/dist/main.js +2 -0
- package/dist/slices/admission/repo.js +7 -1
- package/dist/slices/control/index.js +2 -3
- package/dist/slices/control/providers.js +5 -1
- package/dist/slices/narration/chunk.js +21 -22
- package/dist/slices/storage/repo.js +3 -0
- package/dist/slices/subtitles/prepare.js +15 -5
- package/dist/slices/video/write-export.js +3 -0
- package/dist/web/assets/{index-CbYEcBOa.js → index-C7PasGML.js} +20 -20
- package/dist/web/assets/index-DkC4WXhk.css +1 -0
- package/dist/web/index.html +2 -2
- package/package.json +1 -1
- package/dist/web/assets/index-6zz8telY.css +0 -1
|
@@ -22,7 +22,7 @@ export const alignSubtitles = async (request) => {
|
|
|
22
22
|
const pcmPath = join(working, "audio.f32");
|
|
23
23
|
await decodeAudio(request.ffmpeg, request.audioPath, pcmPath, request.signal);
|
|
24
24
|
request.onProgress?.(25, 100);
|
|
25
|
-
return await runAlignmentWorker({ modelPath, pcmPath, text: request.text }, request.signal, (current, total) => request.onProgress?.(25 + Math.round((current / total) * 75), 100));
|
|
25
|
+
return await runAlignmentWorker({ modelPath, pcmPath, text: request.text }, request.signal, (current, total) => request.onProgress?.(25 + Math.round((current / total) * 75), 100), undefined, request.onOmission);
|
|
26
26
|
}
|
|
27
27
|
finally {
|
|
28
28
|
await rm(working, { recursive: true, force: true });
|
|
@@ -4,6 +4,10 @@ export const workerInput = z.object({
|
|
|
4
4
|
pcmPath: z.string(),
|
|
5
5
|
text: z.string(),
|
|
6
6
|
});
|
|
7
|
+
export const omissionSchema = z.object({
|
|
8
|
+
start: z.number().finite().nonnegative(),
|
|
9
|
+
text: z.string().min(1).max(10000),
|
|
10
|
+
});
|
|
7
11
|
export const workerMessage = z.discriminatedUnion("type", [
|
|
8
12
|
z.object({
|
|
9
13
|
type: z.literal("progress"),
|
|
@@ -19,5 +23,6 @@ export const workerMessage = z.discriminatedUnion("type", [
|
|
|
19
23
|
confidence: z.number().finite().min(0).max(1).optional(),
|
|
20
24
|
})),
|
|
21
25
|
}),
|
|
26
|
+
omissionSchema.extend({ type: z.literal("omission") }),
|
|
22
27
|
z.object({ type: z.literal("error"), message: z.string() }),
|
|
23
28
|
]);
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
// Compare acoustic recognition with the already aligned transcript. This catches extra
|
|
2
2
|
// speech and missing phrases even when a few forced words individually score well.
|
|
3
|
-
export function agreesWithSpeech(expected, observed) {
|
|
3
|
+
export function agreesWithSpeech(expected, observed, maximumError = 0.42) {
|
|
4
4
|
const left = expected.toUpperCase().replace(/[^A-Z']/g, "");
|
|
5
5
|
const right = observed.toUpperCase().replace(/[^A-Z']/g, "");
|
|
6
6
|
if (left.length === 0 || right.length === 0)
|
|
@@ -13,5 +13,5 @@ export function agreesWithSpeech(expected, observed) {
|
|
|
13
13
|
current[column] = Math.min((previous[column] ?? 0) + 1, (current[column - 1] ?? 0) + 1, (previous[column - 1] ?? 0) + (left[row - 1] === right[column - 1] ? 0 : 1));
|
|
14
14
|
previous = current;
|
|
15
15
|
}
|
|
16
|
-
return (previous[right.length] ?? Infinity) / Math.max(left.length, right.length) <=
|
|
16
|
+
return (previous[right.length] ?? Infinity) / Math.max(left.length, right.length) <= maximumError;
|
|
17
17
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { fork } from "node:child_process";
|
|
2
2
|
import { workerMessage } from "./protocol.js";
|
|
3
|
-
export async function runAlignmentWorker(input, signal, onProgress, worker = new URL("./worker.js", import.meta.url)) {
|
|
3
|
+
export async function runAlignmentWorker(input, signal, onProgress, worker = new URL("./worker.js", import.meta.url), onOmission) {
|
|
4
4
|
signal.throwIfAborted();
|
|
5
5
|
return new Promise((resolve, reject) => {
|
|
6
6
|
const options = {
|
|
@@ -48,7 +48,15 @@ export async function runAlignmentWorker(input, signal, onProgress, worker = new
|
|
|
48
48
|
return;
|
|
49
49
|
}
|
|
50
50
|
const message = parsed.data;
|
|
51
|
-
if (message.type === "
|
|
51
|
+
if (message.type === "omission") {
|
|
52
|
+
try {
|
|
53
|
+
onOmission?.({ start: message.start, text: message.text });
|
|
54
|
+
}
|
|
55
|
+
catch (error) {
|
|
56
|
+
finish(error instanceof Error ? error : new Error(String(error)));
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
else if (message.type === "progress") {
|
|
52
60
|
try {
|
|
53
61
|
onProgress?.(message.current, message.total);
|
|
54
62
|
}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import { alignWindow, frameSeconds, mismatch } from "./ctc.js";
|
|
2
|
+
import { agreesWithSpeech } from "./quality.js";
|
|
3
|
+
import { letters } from "./vocabulary.js";
|
|
4
|
+
// Recovery can only omit a short transcript prefix, never insert guessed words or times.
|
|
5
|
+
export function alignSpeechWindow(logits, frames, candidate, complete, cutoff, skipBudget) {
|
|
6
|
+
try {
|
|
7
|
+
return { words: accepted(logits, frames, candidate, complete, cutoff), skipped: 0 };
|
|
8
|
+
}
|
|
9
|
+
catch (error) {
|
|
10
|
+
if (!(error instanceof Error) || error.message !== mismatch)
|
|
11
|
+
throw error;
|
|
12
|
+
}
|
|
13
|
+
const heard = greedy(logits, frames).split(/\s+/);
|
|
14
|
+
for (let skipped = 1; skipped <= Math.min(40, skipBudget, candidate.length - 4); skipped += 1) {
|
|
15
|
+
const remaining = candidate.slice(skipped);
|
|
16
|
+
const anchor = remaining
|
|
17
|
+
.slice(0, 4)
|
|
18
|
+
.map((word) => word.spoken)
|
|
19
|
+
.join(" ");
|
|
20
|
+
if (anchor.replace(/[^A-Z]/g, "").length < 20 ||
|
|
21
|
+
!agreesWithSpeech(anchor, heard.slice(0, anchor.split(/\s+/).length).join(" "), 0.2))
|
|
22
|
+
continue;
|
|
23
|
+
try {
|
|
24
|
+
const words = accepted(logits, frames, remaining, complete, cutoff);
|
|
25
|
+
if (words.length < 4 || words.slice(0, 4).some((word) => (word.confidence ?? 0) < 0.75))
|
|
26
|
+
continue;
|
|
27
|
+
return { words, skipped };
|
|
28
|
+
}
|
|
29
|
+
catch (error) {
|
|
30
|
+
if (!(error instanceof Error) || error.message !== mismatch)
|
|
31
|
+
throw error;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
throw new Error(mismatch);
|
|
35
|
+
}
|
|
36
|
+
function accepted(logits, frames, candidate, complete, cutoff) {
|
|
37
|
+
const words = alignWindow(logits, frames, candidate, complete).words.filter((word) => word.end <= cutoff);
|
|
38
|
+
const last = words.at(-1);
|
|
39
|
+
if (last === undefined ||
|
|
40
|
+
!agreesWithSpeech(candidate
|
|
41
|
+
.slice(0, words.length)
|
|
42
|
+
.map((word) => word.spoken)
|
|
43
|
+
.join(" "), greedy(logits, Math.min(frames, Math.ceil(last.end / frameSeconds)))))
|
|
44
|
+
throw new Error(mismatch);
|
|
45
|
+
return words;
|
|
46
|
+
}
|
|
47
|
+
export function greedy(logits, frames) {
|
|
48
|
+
let previous = -1;
|
|
49
|
+
let text = "";
|
|
50
|
+
for (let frame = 0; frame < frames; frame += 1) {
|
|
51
|
+
let best = 0;
|
|
52
|
+
for (let label = 1; label < 32; label += 1)
|
|
53
|
+
if ((logits[frame * 32 + label] ?? -Infinity) > (logits[frame * 32 + best] ?? -Infinity))
|
|
54
|
+
best = label;
|
|
55
|
+
if (best !== previous && best !== 0)
|
|
56
|
+
text += letters[best] ?? "";
|
|
57
|
+
previous = best;
|
|
58
|
+
}
|
|
59
|
+
return text.trim().replace(/\s+/g, " ");
|
|
60
|
+
}
|
|
@@ -1,10 +1,9 @@
|
|
|
1
1
|
import { open, readFile, stat } from "node:fs/promises";
|
|
2
2
|
import * as ort from "onnxruntime-web/wasm";
|
|
3
|
-
import {
|
|
3
|
+
import { mismatch } from "./ctc.js";
|
|
4
4
|
import { workerInput } from "./protocol.js";
|
|
5
|
-
import { agreesWithSpeech } from "./quality.js";
|
|
6
5
|
import { speechWords } from "./text.js";
|
|
7
|
-
import {
|
|
6
|
+
import { alignSpeechWindow, greedy } from "./window.js";
|
|
8
7
|
const sampleRate = 16000;
|
|
9
8
|
const windowSeconds = 12;
|
|
10
9
|
const overlapSeconds = 2;
|
|
@@ -31,6 +30,8 @@ async function run(input) {
|
|
|
31
30
|
const source = speechWords(input.text);
|
|
32
31
|
const output = [];
|
|
33
32
|
let cursor = 0;
|
|
33
|
+
let omitted = 0;
|
|
34
|
+
const omissionBudget = Math.min(60, Math.floor(source.length * 0.05));
|
|
34
35
|
let sampleAt = 0;
|
|
35
36
|
while (sampleAt < totalSamples && cursor < source.length) {
|
|
36
37
|
const count = Math.min(windowSeconds * sampleRate, totalSamples - sampleAt);
|
|
@@ -62,18 +63,24 @@ async function run(input) {
|
|
|
62
63
|
const candidate = candidates(source, cursor, observed);
|
|
63
64
|
const finalWindow = sampleAt + count >= totalSamples;
|
|
64
65
|
const complete = finalWindow && cursor + candidate.length === source.length;
|
|
65
|
-
const aligned = alignWindow(logits.data, frames, candidate, complete).words;
|
|
66
66
|
const cutoff = finalWindow ? count / sampleRate : count / sampleRate - overlapSeconds;
|
|
67
|
-
const
|
|
67
|
+
const recovered = alignSpeechWindow(logits.data, frames, candidate, complete, cutoff, cursor === 0 ? 0 : omissionBudget - omitted);
|
|
68
|
+
const accepted = recovered.words;
|
|
68
69
|
const last = accepted.at(-1);
|
|
69
70
|
if (last === undefined)
|
|
70
71
|
throw new Error(mismatch);
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
72
|
+
if (recovered.skipped > 0) {
|
|
73
|
+
omitted += recovered.skipped;
|
|
74
|
+
send({
|
|
75
|
+
type: "omission",
|
|
76
|
+
start: sampleAt / sampleRate,
|
|
77
|
+
text: candidate
|
|
78
|
+
.slice(0, recovered.skipped)
|
|
79
|
+
.map((word) => word.text)
|
|
80
|
+
.join(" "),
|
|
81
|
+
});
|
|
82
|
+
cursor += recovered.skipped;
|
|
83
|
+
}
|
|
77
84
|
const offset = sampleAt / sampleRate;
|
|
78
85
|
output.push(...accepted.map((word) => ({
|
|
79
86
|
...word,
|
|
@@ -147,17 +154,3 @@ function normalize(audio) {
|
|
|
147
154
|
const divisor = Math.sqrt(variance + 1e-7);
|
|
148
155
|
return Float32Array.from(audio, (value) => (value - mean) / divisor);
|
|
149
156
|
}
|
|
150
|
-
function greedy(logits, frames) {
|
|
151
|
-
let previous = -1;
|
|
152
|
-
let text = "";
|
|
153
|
-
for (let frame = 0; frame < frames; frame += 1) {
|
|
154
|
-
let best = 0;
|
|
155
|
-
for (let label = 1; label < 32; label += 1)
|
|
156
|
-
if ((logits[frame * 32 + label] ?? -Infinity) > (logits[frame * 32 + best] ?? -Infinity))
|
|
157
|
-
best = label;
|
|
158
|
-
if (best !== previous && best !== 0)
|
|
159
|
-
text += letters[best] ?? "";
|
|
160
|
-
previous = best;
|
|
161
|
-
}
|
|
162
|
-
return text.trim().replace(/\s+/g, " ");
|
|
163
|
-
}
|
package/dist/edge/http/app.js
CHANGED
|
@@ -7,6 +7,7 @@ import { audioPreviewRoutes } from "./audio-preview.js";
|
|
|
7
7
|
import { entryRoutes } from "./entries.js";
|
|
8
8
|
import { fileRoutes } from "./files.js";
|
|
9
9
|
import { fontsRoutes } from "./fonts.js";
|
|
10
|
+
import { openFolderRoutes } from "./open-folder.js";
|
|
10
11
|
import { planningRoutes } from "./planning.js";
|
|
11
12
|
import { problem, problemFromError, titleOf } from "./problem.js";
|
|
12
13
|
import { projectRoutes } from "./projects.js";
|
|
@@ -30,6 +31,7 @@ function apiRoutes(deps, startedAt) {
|
|
|
30
31
|
.route("/staging", stagingRoutes(deps))
|
|
31
32
|
.route("/projects", planningRoutes(deps))
|
|
32
33
|
.route("/projects", projectRoutes(deps))
|
|
34
|
+
.route("/projects", openFolderRoutes(deps))
|
|
33
35
|
.route("/projects", audioPreviewRoutes(deps))
|
|
34
36
|
.route("/update", updateRoutes(deps))
|
|
35
37
|
// The re-run and cancel actions sit on the same prefix as the project itself; they
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
import { dirname } from "node:path";
|
|
2
|
+
import { zValidator } from "@hono/zod-validator";
|
|
3
|
+
import { Hono } from "hono";
|
|
4
|
+
import { z } from "zod";
|
|
5
|
+
import { assetOf, findDownload } from "../../slices/storage/downloads.js";
|
|
6
|
+
import { outputsOf } from "../../slices/storage/repo.js";
|
|
7
|
+
import { onInvalid, problem, titleOf } from "./problem.js";
|
|
8
|
+
export function openFolderRoutes(deps) {
|
|
9
|
+
return new Hono().post("/:id/open-folder", zValidator("param", z.object({
|
|
10
|
+
id: z
|
|
11
|
+
.string()
|
|
12
|
+
.min(1)
|
|
13
|
+
.max(64)
|
|
14
|
+
.regex(/^[0-9A-Za-z_-]+$/),
|
|
15
|
+
}), onInvalid), zValidator("json", z.object({
|
|
16
|
+
asset: z
|
|
17
|
+
.string()
|
|
18
|
+
.max(64)
|
|
19
|
+
.regex(/^(?:[a-z0-9-]+|images\.zip)$/),
|
|
20
|
+
}), onInvalid), async (c) => {
|
|
21
|
+
const origin = c.req.header("origin");
|
|
22
|
+
if (origin !== undefined && origin !== new URL(c.req.url).origin) {
|
|
23
|
+
return problem(c, {
|
|
24
|
+
status: 403,
|
|
25
|
+
title: titleOf(403),
|
|
26
|
+
detail: "Open folders from Slopify itself.",
|
|
27
|
+
});
|
|
28
|
+
}
|
|
29
|
+
const { id } = c.req.valid("param");
|
|
30
|
+
const { asset } = c.req.valid("json");
|
|
31
|
+
// The archive is virtual; open the first existing image's folder without building a zip.
|
|
32
|
+
const assets = asset === "images.zip"
|
|
33
|
+
? outputsOf(deps.db, id)
|
|
34
|
+
.filter((output) => output.role === "image" || output.role === "thumbnail")
|
|
35
|
+
.map(assetOf)
|
|
36
|
+
: [asset];
|
|
37
|
+
const download = assets
|
|
38
|
+
.map((name) => findDownload(deps, id, name))
|
|
39
|
+
.find((result) => result.ok);
|
|
40
|
+
if (download === undefined || !download.ok) {
|
|
41
|
+
return problem(c, {
|
|
42
|
+
status: 404,
|
|
43
|
+
title: titleOf(404),
|
|
44
|
+
detail: "No saved file was found for this output. Re-run the stage if the file was removed.",
|
|
45
|
+
});
|
|
46
|
+
}
|
|
47
|
+
try {
|
|
48
|
+
if (deps.openFolder === undefined)
|
|
49
|
+
throw new Error("No folder opener configured");
|
|
50
|
+
await deps.openFolder(dirname(download.download.path));
|
|
51
|
+
return c.json({ opened: true });
|
|
52
|
+
}
|
|
53
|
+
catch {
|
|
54
|
+
return problem(c, {
|
|
55
|
+
status: 503,
|
|
56
|
+
title: titleOf(503),
|
|
57
|
+
detail: "Could not open the file manager on the machine running Slopify. Make sure a desktop session is available.",
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
});
|
|
61
|
+
}
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import { execFile } from "node:child_process";
|
|
2
|
+
import { promisify } from "node:util";
|
|
3
|
+
import { isWsl } from "./open-browser.js";
|
|
4
|
+
const execute = promisify(execFile);
|
|
5
|
+
export function folderCommand(platform, path, wsl = false) {
|
|
6
|
+
if (platform === "win32" || wsl)
|
|
7
|
+
return ["explorer.exe", [path]];
|
|
8
|
+
return [platform === "darwin" ? "open" : "xdg-open", [path]];
|
|
9
|
+
}
|
|
10
|
+
export async function openFolder(path) {
|
|
11
|
+
const wsl = isWsl(process.platform);
|
|
12
|
+
const target = wsl
|
|
13
|
+
? (await execute("wslpath", ["-w", path], { timeout: 5000 })).stdout.trim()
|
|
14
|
+
: path;
|
|
15
|
+
const [command, args] = folderCommand(process.platform, target, wsl);
|
|
16
|
+
try {
|
|
17
|
+
await execute(command, args, { timeout: 10000 });
|
|
18
|
+
}
|
|
19
|
+
catch (error) {
|
|
20
|
+
// Explorer commonly reports 1 even when it successfully opens an existing window.
|
|
21
|
+
if (command === "explorer.exe" && error instanceof Error && "code" in error && error.code === 1)
|
|
22
|
+
return;
|
|
23
|
+
throw error;
|
|
24
|
+
}
|
|
25
|
+
}
|
package/dist/main.js
CHANGED
|
@@ -11,6 +11,7 @@ import { curateRegistry } from "./catalog/registry.js";
|
|
|
11
11
|
import { createCatalogueStore } from "./catalog/store.js";
|
|
12
12
|
import { createHub } from "./edge/events/hub.js";
|
|
13
13
|
import { createApp } from "./edge/http/app.js";
|
|
14
|
+
import { openFolder } from "./edge/open-folder.js";
|
|
14
15
|
import { createAudioPreviewStore } from "./kernel/audio-preview.js";
|
|
15
16
|
import { systemClock } from "./kernel/clock.js";
|
|
16
17
|
import { openDb } from "./kernel/db/index.js";
|
|
@@ -169,6 +170,7 @@ export async function boot(config) {
|
|
|
169
170
|
},
|
|
170
171
|
});
|
|
171
172
|
const app = createApp({
|
|
173
|
+
openFolder,
|
|
172
174
|
db,
|
|
173
175
|
paths,
|
|
174
176
|
hub,
|
|
@@ -43,7 +43,13 @@ export const runDraftSchema = z.object({
|
|
|
43
43
|
}),
|
|
44
44
|
// Optional until Play carries the control; unknown keys are stripped by this schema, so
|
|
45
45
|
// a mode not listed here would never reach the audio stage.
|
|
46
|
-
chunking: z
|
|
46
|
+
chunking: z
|
|
47
|
+
.object({
|
|
48
|
+
mode: z.enum(chunkModes),
|
|
49
|
+
words: z.number().optional(),
|
|
50
|
+
characters: z.number().int().min(1).max(1000000).optional(),
|
|
51
|
+
})
|
|
52
|
+
.optional(),
|
|
47
53
|
silenceGapSeconds: z.number(),
|
|
48
54
|
subtitles: subtitleConfigSchema.optional(),
|
|
49
55
|
});
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { transact } from "../../kernel/db/tx.js";
|
|
2
2
|
import { derive, satisfied } from "../../kernel/runner/graph.js";
|
|
3
3
|
import { finishStage, projectById, resetStage, setProjectPaused, stagesOf, updateProjectConfig, } from "../admission/repo.js";
|
|
4
|
+
import { sameChunking } from "../narration/chunk.js";
|
|
4
5
|
import { clearUnfinishedAudio } from "../reruns/index.js";
|
|
5
6
|
import { providers as providerCatalog } from "../settings/model.js";
|
|
6
7
|
import { hasKey, listVoices } from "../settings/repo.js";
|
|
@@ -137,9 +138,7 @@ export function changeProviders(deps, id, changes) {
|
|
|
137
138
|
(changes.audio.provider !== project.config.audio?.provider ||
|
|
138
139
|
changes.audio.model !== project.config.audio?.model ||
|
|
139
140
|
changes.audio.voice !== project.config.audio?.voice);
|
|
140
|
-
const chunkingChanged = changes.chunking !== undefined &&
|
|
141
|
-
(changes.chunking.mode !== (project.config.chunking?.mode ?? "whole") ||
|
|
142
|
-
(changes.chunking.words ?? 500) !== (project.config.chunking?.words ?? 500));
|
|
141
|
+
const chunkingChanged = changes.chunking !== undefined && !sameChunking(changes.chunking, project.config.chunking);
|
|
143
142
|
const orphaned = transact(deps.db, () => {
|
|
144
143
|
const files = audioChanged || chunkingChanged ? clearUnfinishedAudio(deps, id) : [];
|
|
145
144
|
updateProjectConfig(deps.db, id, {
|
|
@@ -14,7 +14,11 @@ export const providerChangesSchema = z
|
|
|
14
14
|
audio: choice.extend({ voice: z.string().trim().min(1).max(200) }).optional(),
|
|
15
15
|
images: choice.optional(),
|
|
16
16
|
chunking: z
|
|
17
|
-
.object({
|
|
17
|
+
.object({
|
|
18
|
+
mode: z.enum(chunkModes),
|
|
19
|
+
words: z.number().int().min(1).max(10000).optional(),
|
|
20
|
+
characters: z.number().int().min(1).max(1000000).optional(),
|
|
21
|
+
})
|
|
18
22
|
.strict()
|
|
19
23
|
.optional(),
|
|
20
24
|
})
|
|
@@ -1,12 +1,6 @@
|
|
|
1
|
-
|
|
2
|
-
// paragraph, and every ~N words is consecutive chunks each ending at the last sentence
|
|
3
|
-
// boundary at or before N words, N defaulting to 500.
|
|
4
|
-
//
|
|
5
|
-
// Pure, and the only place the rule lives. Chunking sits in the functional core beside
|
|
6
|
-
// substitution and the render plan, so it is testable with no I/O and `run.ts` does nothing
|
|
7
|
-
// but call it.
|
|
8
|
-
export const chunkModes = ["whole", "paragraph", "words"];
|
|
1
|
+
export const chunkModes = ["whole", "paragraph", "words", "characters"];
|
|
9
2
|
export const defaultChunkWords = 500;
|
|
3
|
+
export const defaultChunkCharacters = 3000;
|
|
10
4
|
// A run created before Play carried the control sends the whole text as one request,
|
|
11
5
|
// which is the first case and the one that adds nothing the user did not ask for.
|
|
12
6
|
export const defaultChunking = { mode: "whole" };
|
|
@@ -20,7 +14,9 @@ export function chunkNarration(text, chunking) {
|
|
|
20
14
|
case "paragraph":
|
|
21
15
|
return nonEmpty(text.split(paragraphBreak));
|
|
22
16
|
case "words":
|
|
23
|
-
return
|
|
17
|
+
return sentenceRuns(text, chunking.words ?? defaultChunkWords, wordsIn);
|
|
18
|
+
case "characters":
|
|
19
|
+
return sentenceRuns(text, chunking.characters ?? defaultChunkCharacters, (part) => Array.from(part.trim()).length);
|
|
24
20
|
}
|
|
25
21
|
}
|
|
26
22
|
// What "N words" counts: runs of non-space. Exported because it is half of the rule -
|
|
@@ -28,29 +24,32 @@ export function chunkNarration(text, chunking) {
|
|
|
28
24
|
export function wordsIn(text) {
|
|
29
25
|
return text.split(/\s+/).filter((word) => word !== "").length;
|
|
30
26
|
}
|
|
31
|
-
function
|
|
32
|
-
|
|
33
|
-
|
|
27
|
+
export function sameChunking(left, right) {
|
|
28
|
+
const a = left ?? defaultChunking;
|
|
29
|
+
const b = right ?? defaultChunking;
|
|
30
|
+
if (a.mode !== b.mode)
|
|
31
|
+
return false;
|
|
32
|
+
if (a.mode === "words")
|
|
33
|
+
return (a.words ?? defaultChunkWords) === (b.words ?? defaultChunkWords);
|
|
34
|
+
if (a.mode === "characters")
|
|
35
|
+
return (a.characters ?? defaultChunkCharacters) === (b.characters ?? defaultChunkCharacters);
|
|
36
|
+
return true;
|
|
37
|
+
}
|
|
38
|
+
function sentenceRuns(text, budget, measure) {
|
|
34
39
|
const limit = Math.max(1, Math.floor(budget));
|
|
35
40
|
const chunks = [];
|
|
36
41
|
let current = "";
|
|
37
|
-
let count = 0;
|
|
38
42
|
for (const sentence of sentences(text)) {
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
// past the budget starts the next chunk instead of being split.
|
|
42
|
-
if (count > 0 && count + words > limit) {
|
|
43
|
+
// Measure the joined text so spaces between sentences count toward a character budget.
|
|
44
|
+
if (current.trim() && measure(current + sentence) > limit) {
|
|
43
45
|
chunks.push(current);
|
|
44
46
|
current = "";
|
|
45
|
-
count = 0;
|
|
46
47
|
}
|
|
47
48
|
current += sentence;
|
|
48
|
-
count += words;
|
|
49
49
|
}
|
|
50
50
|
chunks.push(current);
|
|
51
|
-
//
|
|
52
|
-
//
|
|
53
|
-
// limit is what refuses it, as an error on the stage.
|
|
51
|
+
// Like word chunking, a sentence longer than the budget stays whole on its own.
|
|
52
|
+
// The provider-specific planning layer applies any hard request limit afterwards.
|
|
54
53
|
return nonEmpty(chunks);
|
|
55
54
|
}
|
|
56
55
|
// ceiling: segmented as English. `Intl.Segmenter` is the platform's own sentence breaker
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
import { z } from "zod";
|
|
2
2
|
import { outputRoles, stagedFileStates, stageKinds } from "./model.js";
|
|
3
3
|
const metaSchema = z.object({
|
|
4
|
+
subtitleOmissions: z
|
|
5
|
+
.array(z.object({ start: z.number().finite().nonnegative(), text: z.string() }))
|
|
6
|
+
.optional(),
|
|
4
7
|
subtitlesMode: z.enum(["off", "files", "burn-in"]).optional(),
|
|
5
8
|
promptName: z.string().optional(),
|
|
6
9
|
prompt: z.string().optional(),
|
|
@@ -28,7 +28,14 @@ const fontSchema = z.object({
|
|
|
28
28
|
assName: z.string(),
|
|
29
29
|
extension: z.enum([".ttf", ".otf", ".ttc"]),
|
|
30
30
|
});
|
|
31
|
-
const cacheSchema = z.object({
|
|
31
|
+
const cacheSchema = z.object({
|
|
32
|
+
key: z.string(),
|
|
33
|
+
words: z.array(wordSchema),
|
|
34
|
+
font: fontSchema,
|
|
35
|
+
omissions: z
|
|
36
|
+
.array(z.object({ start: z.number().finite().nonnegative(), text: z.string() }))
|
|
37
|
+
.default([]),
|
|
38
|
+
});
|
|
32
39
|
// Preparation writes into a new directory; only writeExport commits its output rows.
|
|
33
40
|
// The old caption files/font/timing stay usable if alignment or rendering fails.
|
|
34
41
|
export async function prepareSubtitles(deps, context, audio, frame) {
|
|
@@ -52,7 +59,8 @@ export async function prepareSubtitles(deps, context, audio, frame) {
|
|
|
52
59
|
const directory = mkdtempSync(join(dir, "captions-"));
|
|
53
60
|
try {
|
|
54
61
|
const font = await snapshotFont(deps, projectId, outputs, config.fontId, cache, directory);
|
|
55
|
-
const
|
|
62
|
+
const omissions = cache?.key === key ? [...cache.omissions] : [];
|
|
63
|
+
const words = cache?.key === key ? cache.words : await alignSegments(deps, context, segments, omissions);
|
|
56
64
|
context.signal.throwIfAborted();
|
|
57
65
|
const cues = captionCues(words);
|
|
58
66
|
if (cues.length === 0)
|
|
@@ -65,7 +73,7 @@ export async function prepareSubtitles(deps, context, audio, frame) {
|
|
|
65
73
|
fontName: font.assName,
|
|
66
74
|
position: config.position,
|
|
67
75
|
}), { mode: 0o600 });
|
|
68
|
-
writeFileSync(join(directory, "subtitles.json"), JSON.stringify({ key, words, font }), {
|
|
76
|
+
writeFileSync(join(directory, "subtitles.json"), JSON.stringify({ key, words, font, omissions }), {
|
|
69
77
|
mode: 0o600,
|
|
70
78
|
});
|
|
71
79
|
const files = [
|
|
@@ -77,6 +85,7 @@ export async function prepareSubtitles(deps, context, audio, frame) {
|
|
|
77
85
|
];
|
|
78
86
|
return {
|
|
79
87
|
directory,
|
|
88
|
+
omissions,
|
|
80
89
|
burnIn: config.mode === "burn-in" && project.config.sources.video !== "off",
|
|
81
90
|
assets: files.map(([role, path]) => ({
|
|
82
91
|
role,
|
|
@@ -91,7 +100,7 @@ export async function prepareSubtitles(deps, context, audio, frame) {
|
|
|
91
100
|
}
|
|
92
101
|
async function timingKey(segments, signal) {
|
|
93
102
|
// Bump when alignment normalization/model changes. Hash file contents, not timestamps.
|
|
94
|
-
const hash = createHash("sha256").update("wav2vec2-en-a19f851-
|
|
103
|
+
const hash = createHash("sha256").update("wav2vec2-en-a19f851-v2-omissions");
|
|
95
104
|
for (const segment of segments) {
|
|
96
105
|
signal.throwIfAborted();
|
|
97
106
|
hash.update(JSON.stringify({ kind: segment.kind, seconds: segment.seconds, text: segment.text }));
|
|
@@ -103,7 +112,7 @@ async function timingKey(segments, signal) {
|
|
|
103
112
|
}
|
|
104
113
|
return hash.digest("hex");
|
|
105
114
|
}
|
|
106
|
-
async function alignSegments(deps, context, segments) {
|
|
115
|
+
async function alignSegments(deps, context, segments, omissions) {
|
|
107
116
|
if (deps.alignSubtitles === undefined)
|
|
108
117
|
throw new Error("Local subtitle alignment is unavailable in this build.");
|
|
109
118
|
const words = [];
|
|
@@ -114,6 +123,7 @@ async function alignSegments(deps, context, segments) {
|
|
|
114
123
|
if (segment.path !== null) {
|
|
115
124
|
const aligned = await deps.alignSubtitles({
|
|
116
125
|
audioPath: segment.path,
|
|
126
|
+
onOmission: (omission) => omissions.push({ ...omission, start: omission.start + offset }),
|
|
117
127
|
text: segment.text,
|
|
118
128
|
cacheDir: join(deps.paths.dataDir, "models", "english-subtitles"),
|
|
119
129
|
ffmpeg: deps.ffmpeg,
|
|
@@ -90,6 +90,9 @@ export async function writeExport(deps, context, output) {
|
|
|
90
90
|
}
|
|
91
91
|
store(deps, projectId, "render_params", "render.json", null);
|
|
92
92
|
store(deps, projectId, output.role, output.filename, totalMs, {
|
|
93
|
+
...(output.subtitles?.omissions?.length
|
|
94
|
+
? { subtitleOmissions: output.subtitles.omissions }
|
|
95
|
+
: {}),
|
|
93
96
|
subtitlesMode: output.subtitles === undefined ? "off" : output.subtitles.burnIn ? "burn-in" : "files",
|
|
94
97
|
});
|
|
95
98
|
for (const asset of output.subtitles?.assets ?? [])
|