@gentbajko/slopify 3.0.2 → 3.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/alignment/text.js +1 -1
- package/dist/adapters/alignment/window.js +87 -4
- package/dist/adapters/alignment/worker.js +1 -1
- package/dist/edge/http/files.js +9 -1
- package/dist/edge/http/revision-files.js +7 -1
- package/dist/edge/http/waveform.js +77 -0
- package/dist/extension/slopify-studio-chrome.zip +0 -0
- package/dist/extension/slopify-studio-firefox.zip +0 -0
- package/dist/{adapters/alignment/numbers.js → kernel/ports/number-words.js} +18 -4
- package/dist/patch-notes/3.0.3.md +15 -0
- package/dist/patch-notes/3.0.4.md +50 -0
- package/dist/patch-notes/index.json +12 -0
- package/dist/slices/admission/rules.js +27 -11
- package/dist/slices/article/plain.js +19 -0
- package/dist/slices/article/split.js +33 -0
- package/dist/slices/document/pages.js +33 -4
- package/dist/slices/loudness/line-level.js +138 -0
- package/dist/slices/narration/pauses.js +30 -4
- package/dist/slices/rebuild/recipe-build.js +8 -1
- package/dist/slices/rebuild/recipe-exports.js +43 -29
- package/dist/slices/rebuild/recipe-lines.js +17 -0
- package/dist/slices/rebuild/runtime-export.js +14 -2
- package/dist/slices/rebuild/runtime-lines.js +36 -0
- package/dist/slices/rebuild/runtime-subtitles.js +6 -3
- package/dist/slices/rebuild/runtime-voices.js +4 -1
- package/dist/tutorials/Editing-a-Project.md +2 -2
- package/dist/tutorials/Play-Narration.md +1 -1
- package/dist/tutorials/Play-Overview.md +1 -1
- package/dist/tutorials/Play-Title-and-Article.md +1 -1
- package/dist/tutorials/Project-Page.md +2 -2
- package/dist/web/assets/index-KWtMGqKB.css +1 -0
- package/dist/web/assets/{index-C4McGNke.js → index-haZKPEuq.js} +117 -117
- package/dist/web/assets/{pdf-COowcLcz.js → pdf-BEwm41kk.js} +1 -1
- package/dist/web/index.html +2 -2
- package/package.json +1 -1
- package/dist/web/assets/index-DBLSfqA-.css +0 -1
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
import { rmSync, writeFileSync } from "node:fs";
|
|
2
|
+
import { runFfmpeg } from "../video/ffmpeg.js";
|
|
3
|
+
// Level the volume, line by line, for a narration with several speakers. The pieces are levelled
|
|
4
|
+
// before they are joined (`level-pieces.ts`), but a piece is one request to the voice provider,
|
|
5
|
+
// and with native dialogue one request speaks several people's lines. Its average hides them: a
|
|
6
|
+
// narrator with a few loud words averages the same as a character who speaks up throughout, and
|
|
7
|
+
// the character sits 2 dB above the narrator all the way. Once the words are timed, each line's
|
|
8
|
+
// typical level (the median of its 400 ms momentary loudness while it speaks) is brought to
|
|
9
|
+
// within `withinLu` of the narrator's; the gain changes in the pause between two lines, so no
|
|
10
|
+
// word is cut into and nothing moves in time, and the captions keep their timing.
|
|
11
|
+
// ceiling: how far from the narrator's typical level a line may stay.
|
|
12
|
+
export const withinLu = 1;
|
|
13
|
+
// ceiling: the most a line is moved either way; a line further off than this is left to the
|
|
14
|
+
// listener rather than turned into noise or distortion.
|
|
15
|
+
const gainCapDb = 10;
|
|
16
|
+
// A momentary reading under this is a pause, not speech.
|
|
17
|
+
const speechFloor = -45;
|
|
18
|
+
// The first readings of a line still hold the line before it (the window is 400 ms long).
|
|
19
|
+
const settleSeconds = 0.4;
|
|
20
|
+
// Consecutive words by one speaker, from the timed words, in `offset`'s time (the file's own).
|
|
21
|
+
export function linesOf(words, offset) {
|
|
22
|
+
const lines = [];
|
|
23
|
+
for (const word of words) {
|
|
24
|
+
if (word.speaker === undefined)
|
|
25
|
+
continue;
|
|
26
|
+
const last = lines.at(-1);
|
|
27
|
+
if (last !== undefined && last.speaker === word.speaker)
|
|
28
|
+
lines[lines.length - 1] = { ...last, end: word.end - offset };
|
|
29
|
+
else
|
|
30
|
+
lines.push({ speaker: word.speaker, start: word.start - offset, end: word.end - offset });
|
|
31
|
+
}
|
|
32
|
+
return lines;
|
|
33
|
+
}
|
|
34
|
+
// Each line's gain in dB, and the time it starts at: the middle of the pause before the line.
|
|
35
|
+
export function lineGains(lines, frames, narrator = "narrator") {
|
|
36
|
+
const typical = lines.map((line) => {
|
|
37
|
+
const spoken = (from) => frames
|
|
38
|
+
.filter((frame) => frame.t >= from && frame.t <= line.end && frame.momentary > speechFloor)
|
|
39
|
+
.map((frame) => frame.momentary)
|
|
40
|
+
.sort((a, b) => a - b);
|
|
41
|
+
// A short line has too few settled readings; its whole span is the best there is.
|
|
42
|
+
const settled = spoken(line.start + settleSeconds);
|
|
43
|
+
const readings = settled.length >= 3 ? settled : spoken(line.start);
|
|
44
|
+
return readings.length === 0 ? undefined : readings[Math.floor(readings.length / 2)];
|
|
45
|
+
});
|
|
46
|
+
const median = (values) => [...values].sort((a, b) => a - b)[Math.floor(values.length / 2)];
|
|
47
|
+
const known = (values) => values.filter((value) => value !== undefined);
|
|
48
|
+
const reference = median(known(typical.filter((_value, at) => lines[at]?.speaker === narrator))) ??
|
|
49
|
+
median(known(typical));
|
|
50
|
+
if (reference === undefined)
|
|
51
|
+
return [];
|
|
52
|
+
return lines.map((line, at) => {
|
|
53
|
+
const level = typical[at];
|
|
54
|
+
const off = level === undefined ? 0 : level - reference;
|
|
55
|
+
const gain = off > withinLu ? withinLu - off : off < -withinLu ? -withinLu - off : 0;
|
|
56
|
+
const before = lines[at - 1];
|
|
57
|
+
return {
|
|
58
|
+
at: before === undefined ? 0 : (before.end + line.start) / 2,
|
|
59
|
+
gainDb: Math.max(-gainCapDb, Math.min(gainCapDb, Math.round(gain * 100) / 100)),
|
|
60
|
+
};
|
|
61
|
+
});
|
|
62
|
+
}
|
|
63
|
+
// ffmpeg's EBU R128 readings, ten a second, as its verbose log prints them.
|
|
64
|
+
export function parseFrames(stderr) {
|
|
65
|
+
const frames = [];
|
|
66
|
+
for (const match of stderr.matchAll(/t:\s*([0-9.]+)\s+TARGET:[^\n]*?\sM:\s*(-?[0-9.]+)/g))
|
|
67
|
+
frames.push({ t: Number(match[1]), momentary: Number(match[2]) });
|
|
68
|
+
return frames;
|
|
69
|
+
}
|
|
70
|
+
// `input` levelled line by line into `output` (16-bit PCM WAV). `words` are timed on a timeline
|
|
71
|
+
// where the file starts at `offset` seconds. False, with nothing written, when there is nothing
|
|
72
|
+
// to level: no speakers in the words, or every line already within `withinLu`.
|
|
73
|
+
export async function levelLines(run, input, output, words, offset) {
|
|
74
|
+
const lines = linesOf(words, offset);
|
|
75
|
+
if (new Set(lines.map((line) => line.speaker)).size < 2)
|
|
76
|
+
return false;
|
|
77
|
+
let stderr = "";
|
|
78
|
+
await runFfmpeg({
|
|
79
|
+
bin: run.bin,
|
|
80
|
+
args: [
|
|
81
|
+
"-hide_banner",
|
|
82
|
+
"-nostdin",
|
|
83
|
+
"-v",
|
|
84
|
+
"verbose",
|
|
85
|
+
"-nostats",
|
|
86
|
+
"-i",
|
|
87
|
+
input,
|
|
88
|
+
"-af",
|
|
89
|
+
"ebur128",
|
|
90
|
+
"-f",
|
|
91
|
+
"null",
|
|
92
|
+
"-",
|
|
93
|
+
],
|
|
94
|
+
signal: run.signal,
|
|
95
|
+
log: run.log,
|
|
96
|
+
onProgress: () => { },
|
|
97
|
+
onStderr: (text) => {
|
|
98
|
+
stderr += text;
|
|
99
|
+
},
|
|
100
|
+
});
|
|
101
|
+
const gains = lineGains(lines, parseFrames(stderr));
|
|
102
|
+
if (gains.every((one) => one.gainDb === 0))
|
|
103
|
+
return false;
|
|
104
|
+
// The gain for each line, set at its start by a timed command to the volume filter.
|
|
105
|
+
const commands = `${output}.lines.txt`;
|
|
106
|
+
writeFileSync(commands, gains
|
|
107
|
+
.map((one) => `${Math.max(0, one.at).toFixed(3)} volume@lines volume ${(10 ** (one.gainDb / 20)).toFixed(5)};`)
|
|
108
|
+
.join("\n"), { mode: 0o600 });
|
|
109
|
+
try {
|
|
110
|
+
await runFfmpeg({
|
|
111
|
+
bin: run.bin,
|
|
112
|
+
args: [
|
|
113
|
+
"-hide_banner",
|
|
114
|
+
"-nostdin",
|
|
115
|
+
"-loglevel",
|
|
116
|
+
"error",
|
|
117
|
+
"-nostats",
|
|
118
|
+
"-y",
|
|
119
|
+
"-i",
|
|
120
|
+
input,
|
|
121
|
+
"-af",
|
|
122
|
+
`asendcmd=f=${commands.replaceAll("\\", "/").replaceAll(":", "\\\\:")},volume@lines=volume=1:precision=float:eval=frame`,
|
|
123
|
+
"-c:a",
|
|
124
|
+
"pcm_s16le",
|
|
125
|
+
"-f",
|
|
126
|
+
"wav",
|
|
127
|
+
output,
|
|
128
|
+
],
|
|
129
|
+
signal: run.signal,
|
|
130
|
+
log: run.log,
|
|
131
|
+
onProgress: () => { },
|
|
132
|
+
});
|
|
133
|
+
}
|
|
134
|
+
finally {
|
|
135
|
+
rmSync(commands, { force: true });
|
|
136
|
+
}
|
|
137
|
+
return true;
|
|
138
|
+
}
|
|
@@ -1,23 +1,37 @@
|
|
|
1
1
|
import { mkdirSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
|
+
import { numberForms } from "../../kernel/ports/number-words.js";
|
|
3
4
|
import { measureFile } from "../loudness/loudnorm.js";
|
|
4
5
|
import { pieceLufs, pieceTruePeak } from "../loudness/model.js";
|
|
5
6
|
import { probeDurationMs, runFfmpeg } from "../video/ffmpeg.js";
|
|
6
7
|
import { sentences } from "./chunk.js";
|
|
7
|
-
//
|
|
8
|
+
// How long a stretch of text takes to say, in letters. Digits count as their words ("1982" as
|
|
9
|
+
// "nineteen eighty two", English's being a fair measure for any language's), an IPA spelling's
|
|
10
|
+
// slashes and stress marks count nothing, and a run of spacing counts one. Counted as written,
|
|
11
|
+
// a paragraph full of years and issue numbers put its end seconds early.
|
|
12
|
+
export function spokenLength(text) {
|
|
13
|
+
return Array.from(text
|
|
14
|
+
.replace(/\d[\d,]*(?:\.\d+)?(?:s|st|nd|rd|th)?\b/gi, (raw) => numberForms(raw)[0] ?? raw)
|
|
15
|
+
.replace(/[/ˈˌː]/g, "")
|
|
16
|
+
.replace(/\s+/g, " ")).length;
|
|
17
|
+
}
|
|
18
|
+
// The sentence ends inside a text, not counting its last. A name's initial ("Mary J. Blake")
|
|
19
|
+
// ends no sentence, whatever the sentence splitter makes of its full stop.
|
|
8
20
|
export function textBoundaries(text) {
|
|
9
21
|
const parts = [...sentences(text)];
|
|
10
|
-
const total =
|
|
22
|
+
const total = parts.reduce((sum, part) => sum + spokenLength(part), 0);
|
|
11
23
|
if (parts.length < 2 || total === 0)
|
|
12
24
|
return [];
|
|
13
25
|
const boundaries = [];
|
|
14
26
|
let offset = 0;
|
|
15
27
|
for (const part of parts.slice(0, -1)) {
|
|
16
|
-
offset +=
|
|
28
|
+
offset += spokenLength(part);
|
|
17
29
|
const trailing = /\s*$/.exec(part)?.[0] ?? "";
|
|
18
30
|
// Only a real sentence: text that is only spacing is no sentence of its own.
|
|
19
31
|
if (part.trim() === "")
|
|
20
32
|
continue;
|
|
33
|
+
if (/(?:^|[\s("'])\p{Lu}\.\s*$/u.test(part))
|
|
34
|
+
continue;
|
|
21
35
|
boundaries.push({
|
|
22
36
|
ratio: offset / total,
|
|
23
37
|
kind: /\n[ \t\r]*\n/.test(trailing) || /\n/.test(trailing) ? "paragraph" : "sentence",
|
|
@@ -54,6 +68,12 @@ const windowShare = 0.2;
|
|
|
54
68
|
const durationWeight = 3;
|
|
55
69
|
// A stretch this near either end of the piece is the piece's own lead-in or tail.
|
|
56
70
|
const edgeSeconds = 0.02;
|
|
71
|
+
// ceiling: a stretch shorter than this share of the voice's usual sentence pause (the median of
|
|
72
|
+
// its as many longest stretches as the piece has sentence ends) is a breath between words, and
|
|
73
|
+
// never a sentence end. The words only say roughly where an end is: a voice's pace swings with
|
|
74
|
+
// its delivery cues, and one narration put a paragraph end 17 seconds after where its words
|
|
75
|
+
// did. Lengthening a breath there cut "sixty-two" in two.
|
|
76
|
+
const breathShare = 0.4;
|
|
57
77
|
export function planPauses(input) {
|
|
58
78
|
const { durationSeconds, silences, boundaries, settings } = input;
|
|
59
79
|
const leadingSilence = silences.find((one) => one.start <= edgeSeconds);
|
|
@@ -65,7 +85,13 @@ export function planPauses(input) {
|
|
|
65
85
|
const speechStart = leading;
|
|
66
86
|
const speechEnd = Math.max(speechStart, durationSeconds - trailing);
|
|
67
87
|
const speech = speechEnd - speechStart;
|
|
68
|
-
const
|
|
88
|
+
const between = silences.filter((one) => one !== leadingSilence && one !== trailingSilence);
|
|
89
|
+
const longest = between
|
|
90
|
+
.map((one) => one.end - one.start)
|
|
91
|
+
.sort((a, b) => b - a)
|
|
92
|
+
.slice(0, boundaries.length);
|
|
93
|
+
const usual = longest[Math.floor((longest.length - 1) / 2)] ?? 0;
|
|
94
|
+
const inner = between.filter((one) => one.end - one.start >= usual * breathShare);
|
|
69
95
|
if (speech <= 0 || inner.length === 0 || boundaries.length === 0)
|
|
70
96
|
return { inserts: [], leading, trailing };
|
|
71
97
|
const window = Math.max(windowSeconds, speech * windowShare);
|
|
@@ -3,6 +3,7 @@ import { audioRecipes } from "./recipe-audio.js";
|
|
|
3
3
|
import { documentRecipes } from "./recipe-document.js";
|
|
4
4
|
import { editPlan } from "./recipe-edit.js";
|
|
5
5
|
import { exportRecipes } from "./recipe-exports.js";
|
|
6
|
+
import { linePlan } from "./recipe-lines.js";
|
|
6
7
|
import { masterPlan } from "./recipe-loudness.js";
|
|
7
8
|
import { resourceIdentity } from "./recipe-model.js";
|
|
8
9
|
import { imageReference, referenceRecipe } from "./recipe-reference.js";
|
|
@@ -11,6 +12,12 @@ import { shortsRecipes } from "./recipe-shorts.js";
|
|
|
11
12
|
import { textRecipes } from "./recipe-text.js";
|
|
12
13
|
import { thumbnailRecipes, visualAssets, visualRecipes } from "./recipe-visual.js";
|
|
13
14
|
import { youtubeRecipes } from "./recipe-youtube.js";
|
|
15
|
+
// The video's sound: mastered, and on a multi-voice run levelled line by line first.
|
|
16
|
+
function withLines(master, lines) {
|
|
17
|
+
return lines.keys.length === 0
|
|
18
|
+
? master
|
|
19
|
+
: { values: [...master.values, ...lines.values], keys: [...master.keys, ...lines.keys] };
|
|
20
|
+
}
|
|
14
21
|
export function buildRecipes(context) {
|
|
15
22
|
const text = textRecipes(context);
|
|
16
23
|
const audio = audioRecipes(context, text);
|
|
@@ -31,6 +38,6 @@ export function buildRecipes(context) {
|
|
|
31
38
|
...shortsRecipes(context, exports, drawnFrom, audio.cards ?? [], masterPlan(context, audio.levels, "video")),
|
|
32
39
|
...thumbnail,
|
|
33
40
|
...documentRecipes(context, text, thumbnail),
|
|
34
|
-
...visualAssets(context, visualRecipes(context.config, context.content, audio.mediaFingerprint, captions?.fingerprint ?? null, (images) => editPlan(context, exports, youtube, images, audio.cards ?? []), drawnFrom, timing === undefined ? null : resourceIdentity(context, timing), masterPlan(context, audio.levels, "video"))),
|
|
41
|
+
...visualAssets(context, visualRecipes(context.config, context.content, audio.mediaFingerprint, captions?.fingerprint ?? null, (images) => editPlan(context, exports, youtube, images, audio.cards ?? []), drawnFrom, timing === undefined ? null : resourceIdentity(context, timing), withLines(masterPlan(context, audio.levels, "video"), linePlan(context, timing)))),
|
|
35
42
|
]);
|
|
36
43
|
}
|
|
@@ -7,6 +7,7 @@ import { audioExportArgs } from "../video/audio-export-args.js";
|
|
|
7
7
|
import { editNeedsTiming } from "../video/edit-settings.js";
|
|
8
8
|
import { usesVoices } from "../voices/model.js";
|
|
9
9
|
import { portraitValues } from "../voices/portraits.js";
|
|
10
|
+
import { linePlan, linesLevelled } from "./recipe-lines.js";
|
|
10
11
|
import { masterPlan } from "./recipe-loudness.js";
|
|
11
12
|
import { recipe, resourceIdentity, } from "./recipe-model.js";
|
|
12
13
|
export function exportRecipes(context, audio) {
|
|
@@ -16,6 +17,42 @@ export function exportRecipes(context, audio) {
|
|
|
16
17
|
const recipes = [];
|
|
17
18
|
// Level the volume: the listening files are mastered to the audio files' target.
|
|
18
19
|
const master = masterPlan(context, audio.levels, "audioFiles");
|
|
20
|
+
const captions = config.subtitles !== undefined && config.subtitles.mode !== "off";
|
|
21
|
+
// The YouTube description's chapters, the shorts' clips and captions, a short's own
|
|
22
|
+
// captions, and the video's cuts, chapter cards and chapter openers use the same word timing,
|
|
23
|
+
// so it runs for them even with captions off; only the caption files below wait for captions.
|
|
24
|
+
// A multi-voice run with Level the volume on levels its speakers line by line from it too.
|
|
25
|
+
const voices = usesVoices(config) ? config.voices : undefined;
|
|
26
|
+
const lined = linesLevelled(config);
|
|
27
|
+
const timed = captions ||
|
|
28
|
+
usesYoutubeDescription(config) ||
|
|
29
|
+
usesShorts(config) ||
|
|
30
|
+
usesShortMode(config) ||
|
|
31
|
+
editNeedsTiming(config) ||
|
|
32
|
+
reviewsNarration(config) ||
|
|
33
|
+
voices?.audioFiles === true ||
|
|
34
|
+
lined;
|
|
35
|
+
const timing = timed
|
|
36
|
+
? recipe(context, "subtitles:timing", "video", {
|
|
37
|
+
kind: "local",
|
|
38
|
+
version: 1,
|
|
39
|
+
operation: timingOperation(projectLanguage(config)),
|
|
40
|
+
// The lead-in moves every word, so the edge silence is part of the timing.
|
|
41
|
+
values: [
|
|
42
|
+
audio.timeline,
|
|
43
|
+
config.silenceGapSeconds,
|
|
44
|
+
// The project language; an English project reads "en" here as it always did.
|
|
45
|
+
config.language ?? config.subtitles?.language ?? "en",
|
|
46
|
+
config.edgeSilenceSeconds,
|
|
47
|
+
// Each word learns its speaker and turn on a multi-voice run.
|
|
48
|
+
...(voices === undefined ? [] : ["voice-words-v1"]),
|
|
49
|
+
],
|
|
50
|
+
}, audio.keys)
|
|
51
|
+
: undefined;
|
|
52
|
+
const lines = linePlan(context, timing);
|
|
53
|
+
// The timing goes ahead of what waits for it; any other run keeps its old order.
|
|
54
|
+
if (lined && timing !== undefined)
|
|
55
|
+
recipes.push(timing);
|
|
19
56
|
if (config.sources.video === "off")
|
|
20
57
|
recipes.push(recipe(context, "export:wav", "video", {
|
|
21
58
|
kind: "local",
|
|
@@ -25,37 +62,13 @@ export function exportRecipes(context, audio) {
|
|
|
25
62
|
audio.mediaFingerprint,
|
|
26
63
|
audioExportArgs([{ kind: "body", path: "$body", seconds: 0 }], "$output"),
|
|
27
64
|
...master.values,
|
|
65
|
+
...lines.values,
|
|
28
66
|
],
|
|
29
|
-
}, [...audio.keys, ...master.keys]));
|
|
30
|
-
|
|
31
|
-
// The YouTube description's chapters, the shorts' clips and captions, a short's own
|
|
32
|
-
// captions, and the video's cuts, chapter cards and chapter openers use the same word timing,
|
|
33
|
-
// so it runs for them even with captions off; only the caption files below wait for captions.
|
|
34
|
-
const voices = usesVoices(config) ? config.voices : undefined;
|
|
35
|
-
if (!captions &&
|
|
36
|
-
!usesYoutubeDescription(config) &&
|
|
37
|
-
!usesShorts(config) &&
|
|
38
|
-
!usesShortMode(config) &&
|
|
39
|
-
!editNeedsTiming(config) &&
|
|
40
|
-
!reviewsNarration(config) &&
|
|
41
|
-
voices?.audioFiles !== true)
|
|
67
|
+
}, [...audio.keys, ...master.keys, ...lines.keys]));
|
|
68
|
+
if (timing === undefined)
|
|
42
69
|
return recipes;
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
version: 1,
|
|
46
|
-
operation: timingOperation(projectLanguage(config)),
|
|
47
|
-
// The lead-in moves every word, so the edge silence is part of the timing.
|
|
48
|
-
values: [
|
|
49
|
-
audio.timeline,
|
|
50
|
-
config.silenceGapSeconds,
|
|
51
|
-
// The project language; an English project reads "en" here as it always did.
|
|
52
|
-
config.language ?? config.subtitles?.language ?? "en",
|
|
53
|
-
config.edgeSilenceSeconds,
|
|
54
|
-
// Each word learns its speaker and turn on a multi-voice run.
|
|
55
|
-
...(voices === undefined ? [] : ["voice-words-v1"]),
|
|
56
|
-
],
|
|
57
|
-
}, audio.keys);
|
|
58
|
-
recipes.push(timing);
|
|
70
|
+
if (!lined)
|
|
71
|
+
recipes.push(timing);
|
|
59
72
|
if (voices?.audioFiles === true)
|
|
60
73
|
recipes.push(recipe(context, "voices:files", "video", {
|
|
61
74
|
kind: "local",
|
|
@@ -67,6 +80,7 @@ export function exportRecipes(context, audio) {
|
|
|
67
80
|
(audio.sections ?? []).map((section) => [section.title, section.firstTurn]),
|
|
68
81
|
config.title,
|
|
69
82
|
...master.values,
|
|
83
|
+
...lines.values,
|
|
70
84
|
// The book's title and chapter become the files' album and track tags; a project
|
|
71
85
|
// that is no chapter of a book keeps the values it always had.
|
|
72
86
|
...(voices.book === undefined
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import { usesLoudness } from "../loudness/model.js";
|
|
2
|
+
import { usesVoices } from "../voices/model.js";
|
|
3
|
+
// Level the volume line by line (`loudness/line-level.ts`): a multi-voice run with Level the
|
|
4
|
+
// volume on. Its sound waits for the word timing, which says where each speaker's line is.
|
|
5
|
+
export function linesLevelled(config) {
|
|
6
|
+
return usesVoices(config) && usesLoudness(config) && config.sources.audio === "generate";
|
|
7
|
+
}
|
|
8
|
+
// What levelling line by line adds to a recipe that plays the sound: it waits for the word
|
|
9
|
+
// timing. No fingerprint value: the levelling follows from the narration and its timing, which
|
|
10
|
+
// the recipe already carries, so a project made before it (and the bundled samples) keeps its
|
|
11
|
+
// files, and whatever is rendered from now on is levelled. A change to how lines are levelled
|
|
12
|
+
// (`withinLu`) is the time to add one, and to rebuild the samples.
|
|
13
|
+
export function linePlan(context, timing) {
|
|
14
|
+
if (!linesLevelled(context.config) || timing === undefined)
|
|
15
|
+
return { values: [], keys: [] };
|
|
16
|
+
return { values: [], keys: [timing.key] };
|
|
17
|
+
}
|
|
@@ -11,10 +11,12 @@ import { runFfmpeg } from "../video/ffmpeg.js";
|
|
|
11
11
|
import { planRender } from "../video/plan.js";
|
|
12
12
|
import { renderSlideshow } from "../video/slideshow.js";
|
|
13
13
|
import { writePortraits } from "../voices/portraits.js";
|
|
14
|
+
import { linesLevelled } from "./recipe-lines.js";
|
|
14
15
|
import { exportBed } from "./runtime-export-bed.js";
|
|
15
16
|
import { exportEdit } from "./runtime-export-edit.js";
|
|
16
17
|
import { exportSnapshot, retainedOutput, revisionAudio, } from "./runtime-export-inputs.js";
|
|
17
18
|
import { executeShortExport } from "./runtime-export-short.js";
|
|
19
|
+
import { lineLevelledAudio } from "./runtime-lines.js";
|
|
18
20
|
import { preparedResult, preparedText, publishResult } from "./runtime-publication.js";
|
|
19
21
|
export async function executeExportRecipe(deps, context, piece) {
|
|
20
22
|
if (deps.db
|
|
@@ -38,7 +40,15 @@ export async function executeExportRecipe(deps, context, piece) {
|
|
|
38
40
|
const pending = allocateAsset(deps, context.work.projectId, filename);
|
|
39
41
|
const prepared = [];
|
|
40
42
|
let directory;
|
|
43
|
+
// A multi-voice run's body narration levelled line by line (`runtime-lines.ts`), here
|
|
44
|
+
// until the export is written.
|
|
45
|
+
const linesDirectory = linesLevelled(view.revision.config)
|
|
46
|
+
? mkdtempSync(join(projectDir(deps.paths, context.work.projectId), "lines-"))
|
|
47
|
+
: undefined;
|
|
41
48
|
try {
|
|
49
|
+
const sound = linesDirectory === undefined
|
|
50
|
+
? audio
|
|
51
|
+
: await lineLevelledAudio(deps, context, view, audio, linesDirectory);
|
|
42
52
|
const captions = snapshot.view.outputs.filter((row) => row.selected && row.available && row.state === "ready" && row.workKey === "subtitles:files");
|
|
43
53
|
const burn = !wav && view.revision.config.subtitles?.mode === "burn-in";
|
|
44
54
|
if (burn)
|
|
@@ -54,7 +64,7 @@ export async function executeExportRecipe(deps, context, piece) {
|
|
|
54
64
|
: "off";
|
|
55
65
|
const config = view.revision.config;
|
|
56
66
|
const segment = (kind) => {
|
|
57
|
-
const row =
|
|
67
|
+
const row = sound.find((one) => one.kind === kind);
|
|
58
68
|
return row?.path === null || row === undefined
|
|
59
69
|
? undefined
|
|
60
70
|
: { path: row.path, seconds: row.seconds };
|
|
@@ -125,7 +135,7 @@ export async function executeExportRecipe(deps, context, piece) {
|
|
|
125
135
|
try {
|
|
126
136
|
await runFfmpeg({
|
|
127
137
|
bin: deps.ffmpeg,
|
|
128
|
-
args: audioExportArgs(
|
|
138
|
+
args: audioExportArgs(sound, mixed),
|
|
129
139
|
signal: context.signal,
|
|
130
140
|
log: deps.log,
|
|
131
141
|
onProgress,
|
|
@@ -201,6 +211,8 @@ export async function executeExportRecipe(deps, context, piece) {
|
|
|
201
211
|
discardPreparedAssets(deps, [pending, ...prepared.map((one) => one.asset)]);
|
|
202
212
|
if (directory !== undefined)
|
|
203
213
|
rmSync(directory, { recursive: true, force: true });
|
|
214
|
+
if (linesDirectory !== undefined)
|
|
215
|
+
rmSync(linesDirectory, { recursive: true, force: true });
|
|
204
216
|
}
|
|
205
217
|
}
|
|
206
218
|
// The project's images in slideshow order, as files.
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import { readFileSync } from "node:fs";
|
|
2
|
+
import { join } from "node:path";
|
|
3
|
+
import { levelLines } from "../loudness/line-level.js";
|
|
4
|
+
import { outputPath } from "../storage/layout.js";
|
|
5
|
+
import { linesLevelled } from "./recipe-lines.js";
|
|
6
|
+
import { wordsSchema } from "./runtime-subtitles.js";
|
|
7
|
+
// The sound a multi-voice run plays, with its body narration levelled line by line
|
|
8
|
+
// (`loudness/line-level.ts`) into `directory`, for the video, the audio export and the audio
|
|
9
|
+
// files alike. The words are timed on the timeline these segments lay out, so the body starts
|
|
10
|
+
// at the seconds of the segments before it. Anything else plays as it was.
|
|
11
|
+
export async function lineLevelledAudio(deps, context, view, audio, directory) {
|
|
12
|
+
if (!linesLevelled(view.revision.config))
|
|
13
|
+
return audio;
|
|
14
|
+
const timing = view.outputs.find((row) => row.selected &&
|
|
15
|
+
row.available &&
|
|
16
|
+
row.state === "ready" &&
|
|
17
|
+
row.workKey === "subtitles:timing" &&
|
|
18
|
+
row.output.role === "subtitle_words");
|
|
19
|
+
if (timing === undefined)
|
|
20
|
+
throw new Error("The narration's word timing, which evens out the speakers' volume, is missing. Use More → Render the video again in the Video section, then Try again.");
|
|
21
|
+
const { words } = wordsSchema.parse(JSON.parse(readFileSync(outputPath(deps.paths, context.work.projectId, timing.output.path), "utf8")));
|
|
22
|
+
const run = { bin: deps.ffmpeg, log: deps.log, signal: context.signal };
|
|
23
|
+
const levelled = [];
|
|
24
|
+
let offset = 0;
|
|
25
|
+
for (const segment of audio) {
|
|
26
|
+
if (segment.kind === "body" && segment.path !== null) {
|
|
27
|
+
const target = join(directory, "body-lines.wav");
|
|
28
|
+
const done = await levelLines(run, segment.path, target, words, offset);
|
|
29
|
+
levelled.push(done ? { ...segment, path: target } : segment);
|
|
30
|
+
}
|
|
31
|
+
else
|
|
32
|
+
levelled.push(segment);
|
|
33
|
+
offset += segment.seconds;
|
|
34
|
+
}
|
|
35
|
+
return levelled;
|
|
36
|
+
}
|
|
@@ -283,9 +283,12 @@ export function describeMismatch(chunks, segment, mismatch) {
|
|
|
283
283
|
const expected = mismatch.expected === "" ? "the end of the text" : `"${mismatch.expected}…"`;
|
|
284
284
|
const heard = mismatch.heard === "" ? "no more speech" : `"${mismatch.heard.toLowerCase()}…"`;
|
|
285
285
|
const fix = chunk === undefined
|
|
286
|
-
?
|
|
287
|
-
: `
|
|
288
|
-
|
|
286
|
+
? `Listen at ${clock(mismatch.at)}: if words are missing or wrong, regenerate that part in Edit project → Narration, then Continue the run.`
|
|
287
|
+
: `Listen at ${clock(mismatch.at)}: if words are missing or wrong, regenerate narration chunk ${String(at + 1)} in Edit project → Narration, then Continue the run.`;
|
|
288
|
+
// A voice that reads a year, an abbreviation or a name its own way reads it the same way
|
|
289
|
+
// again, so a remake would cost a chunk and fail here once more.
|
|
290
|
+
const same = "If it says them right, only another way (a year, an abbreviation or a name read its own way), regenerating gives the same reading: use Download diagnostics in Settings and report it.";
|
|
291
|
+
return `Subtitles stopped matching the audio at ${where}. The text expected ${expected} but the audio has ${heard} The recording there probably skips or changes words. ${fix} ${same}`;
|
|
289
292
|
}
|
|
290
293
|
function comparable(text) {
|
|
291
294
|
return text
|
|
@@ -10,6 +10,7 @@ import { runFfmpeg } from "../video/ffmpeg.js";
|
|
|
10
10
|
import { audioChapters, audioFileArgs, ffmetadata } from "../voices/audio-files.js";
|
|
11
11
|
import { turnStarts } from "../voices/timing.js";
|
|
12
12
|
import { exportSnapshot, revisionAudio } from "./runtime-export-inputs.js";
|
|
13
|
+
import { lineLevelledAudio } from "./runtime-lines.js";
|
|
13
14
|
import { preparedResult, publishResult } from "./runtime-publication.js";
|
|
14
15
|
import { wordsSchema } from "./runtime-subtitles.js";
|
|
15
16
|
// The listening files of a multi-voice run: the MP3 and the M4B of the narration timeline,
|
|
@@ -64,9 +65,11 @@ export async function executeVoicesRecipe(deps, context, piece) {
|
|
|
64
65
|
let sound = audio;
|
|
65
66
|
if (goal !== undefined) {
|
|
66
67
|
const mixed = join(directory, "mix.wav");
|
|
68
|
+
// A multi-voice run's speakers levelled line by line first (`runtime-lines.ts`).
|
|
69
|
+
const levelled = await lineLevelledAudio(deps, context, view, audio, directory);
|
|
67
70
|
await runFfmpeg({
|
|
68
71
|
bin: deps.ffmpeg,
|
|
69
|
-
args: audioExportArgs(
|
|
72
|
+
args: audioExportArgs(levelled, mixed),
|
|
70
73
|
signal: context.signal,
|
|
71
74
|
log: deps.log,
|
|
72
75
|
onProgress: () => { },
|
|
@@ -68,14 +68,14 @@ When the project was narrated as one whole request (Chunking set to the whole ar
|
|
|
68
68
|
| **Replace image N** | Uses your own picture in its place. |
|
|
69
69
|
| **Move image N earlier** / **later** | Changes the order. |
|
|
70
70
|
| **Delete image N** | Removes it. Deleting the last image also turns Images and Video off. |
|
|
71
|
-
| **Regenerate image N
|
|
71
|
+
| **Regenerate image N** | Marks the image to be drawn again from its prompt, one paid image call, with a new result, when you save. The image says it is marked, and **Keep image N** takes the mark off. The current image stays until the new one is made. |
|
|
72
72
|
| **Add generated image** | Adds a new image with an empty prompt at the end. |
|
|
73
73
|
| **Add provided image** | Adds your own picture at the end. |
|
|
74
74
|
| **Add a video clip** | Adds a clip (MP4, MOV, M4V, WebM or MKV). It plays muted in an image's place, trimmed, slowed (to half speed at most) or looped to fit. |
|
|
75
75
|
|
|
76
76
|
A project holds at most 60 images, or 240 with **More images for long videos** on.
|
|
77
77
|
|
|
78
|
-
The **Regenerate** button on an image in the project's Images section
|
|
78
|
+
The **Regenerate** button on an image in the project's Images section, and **Regenerate all** beside Download all, do it without the settings: they ask once, then make the new images straight away. The video keeps the old images, marked outdated, until you remake it, so you can regenerate several images and render the video once. If the settings have unsaved changes of their own, the images are marked there instead, to go with them when you save.
|
|
79
79
|
|
|
80
80
|
### Image prompts and More images for long videos
|
|
81
81
|
|
|
@@ -61,7 +61,7 @@ Has the text model add delivery directions and sounds, such as a sigh or a laugh
|
|
|
61
61
|
| --- | --- | --- |
|
|
62
62
|
| **Narration Preparation** | Pick a narration prompt from the Library, or **Off**. One text-model call per narration chunk and per intro or outro entry. | Off |
|
|
63
63
|
|
|
64
|
-
It needs the Inworld TTS-2 model. With another model, the field says "Choose Inworld TTS-2 or turn preparation Off." With several speakers it works per speaker: each turn of a speaker on Inworld TTS-2 is prepared on its own, with the speaker's name and role, so every voice gets its own cues. At least one speaker must use TTS-2; turns of speakers on other voices are spoken as written. See [Multiple Voices](Multiple-Voices).
|
|
64
|
+
It needs the Inworld TTS-2 model. With another model, the field says "Choose Inworld TTS-2 or turn preparation Off." With several speakers it works per speaker: each turn of a speaker on Inworld TTS-2 is prepared on its own, with the speaker's name and role, so every voice gets its own cues. At least one speaker must use TTS-2; turns of speakers on other voices are spoken as written. See [Multiple Voices](Multiple-Voices). Preparation is written by the text model: when nothing earlier needs one (a pasted article, say), **Text generation** appears in this row.
|
|
65
65
|
|
|
66
66
|
### Pronunciation Glossary
|
|
67
67
|
|
|
@@ -104,7 +104,7 @@ If you have the same draft open in another tab or window and it was saved there,
|
|
|
104
104
|
| Row | What it holds | Details |
|
|
105
105
|
| --- | --- | --- |
|
|
106
106
|
| **Title and keywords** | The title pattern and every keyword value | This page |
|
|
107
|
-
| **Article** | Article source, article prompt, text generation (LLM, model, thinking), research | [Play Title and Article](Play-Title-and-Article) |
|
|
107
|
+
| **Article** | Article source, article prompt, text generation (LLM, model, thinking) when the article or research is written, research | [Play Title and Article](Play-Title-and-Article) |
|
|
108
108
|
| **Narration** | Narration source, TTS, model, voice, speakers, Audio Advanced (chunking, intro, outro, preparation, glossary, aliases, tables and figures) | [Play Narration](Play-Narration) |
|
|
109
109
|
| **Images** | Images source, provider, model, effort, image prompts, establishing image, more images for long videos | [Play Images](Play-Images) |
|
|
110
110
|
| **Video and style** | Video source, seconds per image, zoom, motion, cuts, the Look, ambient sound, pauses and volume, silence, frame format, captions, preview text | [Play Video and Style](Play-Video-and-Style) |
|
|
@@ -52,7 +52,7 @@ Include everything that should be read, headings too. Headings become chapters.
|
|
|
52
52
|
|
|
53
53
|
One text model writes everything this run needs in words: the article, the research, the thumbnail's image prompt, intro and outro entries set to LLM, Narration Preparation, table and figure descriptions, the YouTube description and picking shorts. Each is a separate call, counted in usage.
|
|
54
54
|
|
|
55
|
-
The **Text generation** section appears
|
|
55
|
+
The **Text generation** section appears whenever the run needs a text model, in the row of the first thing that needs it, and its line says what it is used for. It is under **Article** when the article or research is written for you; under **Narration** when only the narration needs it, such as an audiobook whose speakers are worked out from a book you pasted; and under **Outputs** when only the YouTube description, the shorts or the thumbnail prompt need it. There is one text model per run, wherever it shows. **Settings** beside it opens Settings, where you set the default for new drafts.
|
|
56
56
|
|
|
57
57
|
| Option | What it does | Default |
|
|
58
58
|
| --- | --- | --- |
|
|
@@ -109,9 +109,9 @@ A section for a stage that is switched off says "*Article* is off for this run."
|
|
|
109
109
|
|
|
110
110
|
### Re-run a whole stage
|
|
111
111
|
|
|
112
|
-
Each stage section has
|
|
112
|
+
Each stage section's head has a button for making that whole stage again, named for what it makes:
|
|
113
113
|
|
|
114
|
-
|
|
|
114
|
+
| Button | What it does |
|
|
115
115
|
| --- | --- |
|
|
116
116
|
| **Research again** | Fresh research, then the affected article, narration, thumbnail and exports. |
|
|
117
117
|
| **Write the article again** | A fresh article, then the affected narration, thumbnail and exports. |
|