@gentbajko/slopify 3.0.3 → 3.0.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/dist/adapters/alignment/text.js +1 -1
  2. package/dist/adapters/alignment/window.js +87 -4
  3. package/dist/adapters/alignment/worker.js +1 -1
  4. package/dist/edge/http/files.js +9 -1
  5. package/dist/edge/http/revision-files.js +7 -1
  6. package/dist/edge/http/waveform.js +77 -0
  7. package/dist/extension/slopify-studio-chrome.zip +0 -0
  8. package/dist/extension/slopify-studio-firefox.zip +0 -0
  9. package/dist/{adapters/alignment/numbers.js → kernel/ports/number-words.js} +18 -4
  10. package/dist/patch-notes/3.0.4.md +50 -0
  11. package/dist/patch-notes/index.json +6 -0
  12. package/dist/slices/admission/rules.js +27 -11
  13. package/dist/slices/article/plain.js +19 -0
  14. package/dist/slices/article/split.js +33 -0
  15. package/dist/slices/document/pages.js +33 -4
  16. package/dist/slices/loudness/line-level.js +138 -0
  17. package/dist/slices/narration/pauses.js +30 -4
  18. package/dist/slices/rebuild/recipe-build.js +8 -1
  19. package/dist/slices/rebuild/recipe-exports.js +43 -29
  20. package/dist/slices/rebuild/recipe-lines.js +17 -0
  21. package/dist/slices/rebuild/runtime-export.js +14 -2
  22. package/dist/slices/rebuild/runtime-lines.js +36 -0
  23. package/dist/slices/rebuild/runtime-subtitles.js +6 -3
  24. package/dist/slices/rebuild/runtime-voices.js +4 -1
  25. package/dist/tutorials/Play-Narration.md +1 -1
  26. package/dist/tutorials/Play-Overview.md +1 -1
  27. package/dist/tutorials/Play-Title-and-Article.md +1 -1
  28. package/dist/tutorials/Project-Page.md +2 -2
  29. package/dist/web/assets/index-KWtMGqKB.css +1 -0
  30. package/dist/web/assets/{index-C37k_oL0.js → index-haZKPEuq.js} +117 -117
  31. package/dist/web/assets/{pdf-DmL8a5oI.js → pdf-BEwm41kk.js} +1 -1
  32. package/dist/web/index.html +2 -2
  33. package/package.json +1 -1
  34. package/dist/web/assets/index-DBLSfqA-.css +0 -1
@@ -0,0 +1,138 @@
1
+ import { rmSync, writeFileSync } from "node:fs";
2
+ import { runFfmpeg } from "../video/ffmpeg.js";
3
+ // Level the volume, line by line, for a narration with several speakers. The pieces are levelled
4
+ // before they are joined (`level-pieces.ts`), but a piece is one request to the voice provider,
5
+ // and with native dialogue one request speaks several people's lines. Its average hides them: a
6
+ // narrator with a few loud words averages the same as a character who speaks up throughout, and
7
+ // the character sits 2 dB above the narrator all the way. Once the words are timed, each line's
8
+ // typical level (the median of its 400 ms momentary loudness while it speaks) is brought to
9
+ // within `withinLu` of the narrator's; the gain changes in the pause between two lines, so no
10
+ // word is cut into and nothing moves in time, and the captions keep their timing.
11
+ // ceiling: how far from the narrator's typical level a line may stay.
12
+ export const withinLu = 1;
13
+ // ceiling: the most a line is moved either way; a line further off than this is left to the
14
+ // listener rather than turned into noise or distortion.
15
+ const gainCapDb = 10;
16
+ // A momentary reading under this is a pause, not speech.
17
+ const speechFloor = -45;
18
+ // The first readings of a line still hold the line before it (the window is 400 ms long).
19
+ const settleSeconds = 0.4;
20
+ // Consecutive words by one speaker, from the timed words, in `offset`'s time (the file's own).
21
+ export function linesOf(words, offset) {
22
+ const lines = [];
23
+ for (const word of words) {
24
+ if (word.speaker === undefined)
25
+ continue;
26
+ const last = lines.at(-1);
27
+ if (last !== undefined && last.speaker === word.speaker)
28
+ lines[lines.length - 1] = { ...last, end: word.end - offset };
29
+ else
30
+ lines.push({ speaker: word.speaker, start: word.start - offset, end: word.end - offset });
31
+ }
32
+ return lines;
33
+ }
34
+ // Each line's gain in dB, and the time it starts at: the middle of the pause before the line.
35
+ export function lineGains(lines, frames, narrator = "narrator") {
36
+ const typical = lines.map((line) => {
37
+ const spoken = (from) => frames
38
+ .filter((frame) => frame.t >= from && frame.t <= line.end && frame.momentary > speechFloor)
39
+ .map((frame) => frame.momentary)
40
+ .sort((a, b) => a - b);
41
+ // A short line has too few settled readings; its whole span is the best there is.
42
+ const settled = spoken(line.start + settleSeconds);
43
+ const readings = settled.length >= 3 ? settled : spoken(line.start);
44
+ return readings.length === 0 ? undefined : readings[Math.floor(readings.length / 2)];
45
+ });
46
+ const median = (values) => [...values].sort((a, b) => a - b)[Math.floor(values.length / 2)];
47
+ const known = (values) => values.filter((value) => value !== undefined);
48
+ const reference = median(known(typical.filter((_value, at) => lines[at]?.speaker === narrator))) ??
49
+ median(known(typical));
50
+ if (reference === undefined)
51
+ return [];
52
+ return lines.map((line, at) => {
53
+ const level = typical[at];
54
+ const off = level === undefined ? 0 : level - reference;
55
+ const gain = off > withinLu ? withinLu - off : off < -withinLu ? -withinLu - off : 0;
56
+ const before = lines[at - 1];
57
+ return {
58
+ at: before === undefined ? 0 : (before.end + line.start) / 2,
59
+ gainDb: Math.max(-gainCapDb, Math.min(gainCapDb, Math.round(gain * 100) / 100)),
60
+ };
61
+ });
62
+ }
63
+ // ffmpeg's EBU R128 readings, ten a second, as its verbose log prints them.
64
+ export function parseFrames(stderr) {
65
+ const frames = [];
66
+ for (const match of stderr.matchAll(/t:\s*([0-9.]+)\s+TARGET:[^\n]*?\sM:\s*(-?[0-9.]+)/g))
67
+ frames.push({ t: Number(match[1]), momentary: Number(match[2]) });
68
+ return frames;
69
+ }
70
+ // `input` levelled line by line into `output` (16-bit PCM WAV). `words` are timed on a timeline
71
+ // where the file starts at `offset` seconds. False, with nothing written, when there is nothing
72
+ // to level: no speakers in the words, or every line already within `withinLu`.
73
+ export async function levelLines(run, input, output, words, offset) {
74
+ const lines = linesOf(words, offset);
75
+ if (new Set(lines.map((line) => line.speaker)).size < 2)
76
+ return false;
77
+ let stderr = "";
78
+ await runFfmpeg({
79
+ bin: run.bin,
80
+ args: [
81
+ "-hide_banner",
82
+ "-nostdin",
83
+ "-v",
84
+ "verbose",
85
+ "-nostats",
86
+ "-i",
87
+ input,
88
+ "-af",
89
+ "ebur128",
90
+ "-f",
91
+ "null",
92
+ "-",
93
+ ],
94
+ signal: run.signal,
95
+ log: run.log,
96
+ onProgress: () => { },
97
+ onStderr: (text) => {
98
+ stderr += text;
99
+ },
100
+ });
101
+ const gains = lineGains(lines, parseFrames(stderr));
102
+ if (gains.every((one) => one.gainDb === 0))
103
+ return false;
104
+ // The gain for each line, set at its start by a timed command to the volume filter.
105
+ const commands = `${output}.lines.txt`;
106
+ writeFileSync(commands, gains
107
+ .map((one) => `${Math.max(0, one.at).toFixed(3)} volume@lines volume ${(10 ** (one.gainDb / 20)).toFixed(5)};`)
108
+ .join("\n"), { mode: 0o600 });
109
+ try {
110
+ await runFfmpeg({
111
+ bin: run.bin,
112
+ args: [
113
+ "-hide_banner",
114
+ "-nostdin",
115
+ "-loglevel",
116
+ "error",
117
+ "-nostats",
118
+ "-y",
119
+ "-i",
120
+ input,
121
+ "-af",
122
+ `asendcmd=f=${commands.replaceAll("\\", "/").replaceAll(":", "\\\\:")},volume@lines=volume=1:precision=float:eval=frame`,
123
+ "-c:a",
124
+ "pcm_s16le",
125
+ "-f",
126
+ "wav",
127
+ output,
128
+ ],
129
+ signal: run.signal,
130
+ log: run.log,
131
+ onProgress: () => { },
132
+ });
133
+ }
134
+ finally {
135
+ rmSync(commands, { force: true });
136
+ }
137
+ return true;
138
+ }
@@ -1,23 +1,37 @@
1
1
  import { mkdirSync } from "node:fs";
2
2
  import { join } from "node:path";
3
+ import { numberForms } from "../../kernel/ports/number-words.js";
3
4
  import { measureFile } from "../loudness/loudnorm.js";
4
5
  import { pieceLufs, pieceTruePeak } from "../loudness/model.js";
5
6
  import { probeDurationMs, runFfmpeg } from "../video/ffmpeg.js";
6
7
  import { sentences } from "./chunk.js";
7
- // The sentence ends inside a text, not counting its last.
8
+ // How long a stretch of text takes to say, in letters. Digits count as their words ("1982" as
9
+ // "nineteen eighty two", English's being a fair measure for any language's), an IPA spelling's
10
+ // slashes and stress marks count nothing, and a run of spacing counts one. Counted as written,
11
+ // a paragraph full of years and issue numbers put its end seconds early.
12
+ export function spokenLength(text) {
13
+ return Array.from(text
14
+ .replace(/\d[\d,]*(?:\.\d+)?(?:s|st|nd|rd|th)?\b/gi, (raw) => numberForms(raw)[0] ?? raw)
15
+ .replace(/[/ˈˌː]/g, "")
16
+ .replace(/\s+/g, " ")).length;
17
+ }
18
+ // The sentence ends inside a text, not counting its last. A name's initial ("Mary J. Blake")
19
+ // ends no sentence, whatever the sentence splitter makes of its full stop.
8
20
  export function textBoundaries(text) {
9
21
  const parts = [...sentences(text)];
10
- const total = Array.from(text).length;
22
+ const total = parts.reduce((sum, part) => sum + spokenLength(part), 0);
11
23
  if (parts.length < 2 || total === 0)
12
24
  return [];
13
25
  const boundaries = [];
14
26
  let offset = 0;
15
27
  for (const part of parts.slice(0, -1)) {
16
- offset += Array.from(part).length;
28
+ offset += spokenLength(part);
17
29
  const trailing = /\s*$/.exec(part)?.[0] ?? "";
18
30
  // Only a real sentence: text that is only spacing is no sentence of its own.
19
31
  if (part.trim() === "")
20
32
  continue;
33
+ if (/(?:^|[\s("'])\p{Lu}\.\s*$/u.test(part))
34
+ continue;
21
35
  boundaries.push({
22
36
  ratio: offset / total,
23
37
  kind: /\n[ \t\r]*\n/.test(trailing) || /\n/.test(trailing) ? "paragraph" : "sentence",
@@ -54,6 +68,12 @@ const windowShare = 0.2;
54
68
  const durationWeight = 3;
55
69
  // A stretch this near either end of the piece is the piece's own lead-in or tail.
56
70
  const edgeSeconds = 0.02;
71
+ // ceiling: a stretch shorter than this share of the voice's usual sentence pause (the median of
72
+ // its as many longest stretches as the piece has sentence ends) is a breath between words, and
73
+ // never a sentence end. The words only say roughly where an end is: a voice's pace swings with
74
+ // its delivery cues, and one narration put a paragraph end 17 seconds after where its words
75
+ // did. Lengthening a breath there cut "sixty-two" in two.
76
+ const breathShare = 0.4;
57
77
  export function planPauses(input) {
58
78
  const { durationSeconds, silences, boundaries, settings } = input;
59
79
  const leadingSilence = silences.find((one) => one.start <= edgeSeconds);
@@ -65,7 +85,13 @@ export function planPauses(input) {
65
85
  const speechStart = leading;
66
86
  const speechEnd = Math.max(speechStart, durationSeconds - trailing);
67
87
  const speech = speechEnd - speechStart;
68
- const inner = silences.filter((one) => one !== leadingSilence && one !== trailingSilence);
88
+ const between = silences.filter((one) => one !== leadingSilence && one !== trailingSilence);
89
+ const longest = between
90
+ .map((one) => one.end - one.start)
91
+ .sort((a, b) => b - a)
92
+ .slice(0, boundaries.length);
93
+ const usual = longest[Math.floor((longest.length - 1) / 2)] ?? 0;
94
+ const inner = between.filter((one) => one.end - one.start >= usual * breathShare);
69
95
  if (speech <= 0 || inner.length === 0 || boundaries.length === 0)
70
96
  return { inserts: [], leading, trailing };
71
97
  const window = Math.max(windowSeconds, speech * windowShare);
@@ -3,6 +3,7 @@ import { audioRecipes } from "./recipe-audio.js";
3
3
  import { documentRecipes } from "./recipe-document.js";
4
4
  import { editPlan } from "./recipe-edit.js";
5
5
  import { exportRecipes } from "./recipe-exports.js";
6
+ import { linePlan } from "./recipe-lines.js";
6
7
  import { masterPlan } from "./recipe-loudness.js";
7
8
  import { resourceIdentity } from "./recipe-model.js";
8
9
  import { imageReference, referenceRecipe } from "./recipe-reference.js";
@@ -11,6 +12,12 @@ import { shortsRecipes } from "./recipe-shorts.js";
11
12
  import { textRecipes } from "./recipe-text.js";
12
13
  import { thumbnailRecipes, visualAssets, visualRecipes } from "./recipe-visual.js";
13
14
  import { youtubeRecipes } from "./recipe-youtube.js";
15
+ // The video's sound: mastered, and on a multi-voice run levelled line by line first.
16
+ function withLines(master, lines) {
17
+ return lines.keys.length === 0
18
+ ? master
19
+ : { values: [...master.values, ...lines.values], keys: [...master.keys, ...lines.keys] };
20
+ }
14
21
  export function buildRecipes(context) {
15
22
  const text = textRecipes(context);
16
23
  const audio = audioRecipes(context, text);
@@ -31,6 +38,6 @@ export function buildRecipes(context) {
31
38
  ...shortsRecipes(context, exports, drawnFrom, audio.cards ?? [], masterPlan(context, audio.levels, "video")),
32
39
  ...thumbnail,
33
40
  ...documentRecipes(context, text, thumbnail),
34
- ...visualAssets(context, visualRecipes(context.config, context.content, audio.mediaFingerprint, captions?.fingerprint ?? null, (images) => editPlan(context, exports, youtube, images, audio.cards ?? []), drawnFrom, timing === undefined ? null : resourceIdentity(context, timing), masterPlan(context, audio.levels, "video"))),
41
+ ...visualAssets(context, visualRecipes(context.config, context.content, audio.mediaFingerprint, captions?.fingerprint ?? null, (images) => editPlan(context, exports, youtube, images, audio.cards ?? []), drawnFrom, timing === undefined ? null : resourceIdentity(context, timing), withLines(masterPlan(context, audio.levels, "video"), linePlan(context, timing)))),
35
42
  ]);
36
43
  }
@@ -7,6 +7,7 @@ import { audioExportArgs } from "../video/audio-export-args.js";
7
7
  import { editNeedsTiming } from "../video/edit-settings.js";
8
8
  import { usesVoices } from "../voices/model.js";
9
9
  import { portraitValues } from "../voices/portraits.js";
10
+ import { linePlan, linesLevelled } from "./recipe-lines.js";
10
11
  import { masterPlan } from "./recipe-loudness.js";
11
12
  import { recipe, resourceIdentity, } from "./recipe-model.js";
12
13
  export function exportRecipes(context, audio) {
@@ -16,6 +17,42 @@ export function exportRecipes(context, audio) {
16
17
  const recipes = [];
17
18
  // Level the volume: the listening files are mastered to the audio files' target.
18
19
  const master = masterPlan(context, audio.levels, "audioFiles");
20
+ const captions = config.subtitles !== undefined && config.subtitles.mode !== "off";
21
+ // The YouTube description's chapters, the shorts' clips and captions, a short's own
22
+ // captions, and the video's cuts, chapter cards and chapter openers use the same word timing,
23
+ // so it runs for them even with captions off; only the caption files below wait for captions.
24
+ // A multi-voice run with Level the volume on levels its speakers line by line from it too.
25
+ const voices = usesVoices(config) ? config.voices : undefined;
26
+ const lined = linesLevelled(config);
27
+ const timed = captions ||
28
+ usesYoutubeDescription(config) ||
29
+ usesShorts(config) ||
30
+ usesShortMode(config) ||
31
+ editNeedsTiming(config) ||
32
+ reviewsNarration(config) ||
33
+ voices?.audioFiles === true ||
34
+ lined;
35
+ const timing = timed
36
+ ? recipe(context, "subtitles:timing", "video", {
37
+ kind: "local",
38
+ version: 1,
39
+ operation: timingOperation(projectLanguage(config)),
40
+ // The lead-in moves every word, so the edge silence is part of the timing.
41
+ values: [
42
+ audio.timeline,
43
+ config.silenceGapSeconds,
44
+ // The project language; an English project reads "en" here as it always did.
45
+ config.language ?? config.subtitles?.language ?? "en",
46
+ config.edgeSilenceSeconds,
47
+ // Each word learns its speaker and turn on a multi-voice run.
48
+ ...(voices === undefined ? [] : ["voice-words-v1"]),
49
+ ],
50
+ }, audio.keys)
51
+ : undefined;
52
+ const lines = linePlan(context, timing);
53
+ // The timing goes ahead of what waits for it; any other run keeps its old order.
54
+ if (lined && timing !== undefined)
55
+ recipes.push(timing);
19
56
  if (config.sources.video === "off")
20
57
  recipes.push(recipe(context, "export:wav", "video", {
21
58
  kind: "local",
@@ -25,37 +62,13 @@ export function exportRecipes(context, audio) {
25
62
  audio.mediaFingerprint,
26
63
  audioExportArgs([{ kind: "body", path: "$body", seconds: 0 }], "$output"),
27
64
  ...master.values,
65
+ ...lines.values,
28
66
  ],
29
- }, [...audio.keys, ...master.keys]));
30
- const captions = config.subtitles !== undefined && config.subtitles.mode !== "off";
31
- // The YouTube description's chapters, the shorts' clips and captions, a short's own
32
- // captions, and the video's cuts, chapter cards and chapter openers use the same word timing,
33
- // so it runs for them even with captions off; only the caption files below wait for captions.
34
- const voices = usesVoices(config) ? config.voices : undefined;
35
- if (!captions &&
36
- !usesYoutubeDescription(config) &&
37
- !usesShorts(config) &&
38
- !usesShortMode(config) &&
39
- !editNeedsTiming(config) &&
40
- !reviewsNarration(config) &&
41
- voices?.audioFiles !== true)
67
+ }, [...audio.keys, ...master.keys, ...lines.keys]));
68
+ if (timing === undefined)
42
69
  return recipes;
43
- const timing = recipe(context, "subtitles:timing", "video", {
44
- kind: "local",
45
- version: 1,
46
- operation: timingOperation(projectLanguage(config)),
47
- // The lead-in moves every word, so the edge silence is part of the timing.
48
- values: [
49
- audio.timeline,
50
- config.silenceGapSeconds,
51
- // The project language; an English project reads "en" here as it always did.
52
- config.language ?? config.subtitles?.language ?? "en",
53
- config.edgeSilenceSeconds,
54
- // Each word learns its speaker and turn on a multi-voice run.
55
- ...(voices === undefined ? [] : ["voice-words-v1"]),
56
- ],
57
- }, audio.keys);
58
- recipes.push(timing);
70
+ if (!lined)
71
+ recipes.push(timing);
59
72
  if (voices?.audioFiles === true)
60
73
  recipes.push(recipe(context, "voices:files", "video", {
61
74
  kind: "local",
@@ -67,6 +80,7 @@ export function exportRecipes(context, audio) {
67
80
  (audio.sections ?? []).map((section) => [section.title, section.firstTurn]),
68
81
  config.title,
69
82
  ...master.values,
83
+ ...lines.values,
70
84
  // The book's title and chapter become the files' album and track tags; a project
71
85
  // that is no chapter of a book keeps the values it always had.
72
86
  ...(voices.book === undefined
@@ -0,0 +1,17 @@
1
+ import { usesLoudness } from "../loudness/model.js";
2
+ import { usesVoices } from "../voices/model.js";
3
+ // Level the volume line by line (`loudness/line-level.ts`): a multi-voice run with Level the
4
+ // volume on. Its sound waits for the word timing, which says where each speaker's line is.
5
+ export function linesLevelled(config) {
6
+ return usesVoices(config) && usesLoudness(config) && config.sources.audio === "generate";
7
+ }
8
+ // What levelling line by line adds to a recipe that plays the sound: it waits for the word
9
+ // timing. No fingerprint value: the levelling follows from the narration and its timing, which
10
+ // the recipe already carries, so a project made before it (and the bundled samples) keeps its
11
+ // files, and whatever is rendered from now on is levelled. A change to how lines are levelled
12
+ // (`withinLu`) is the time to add one, and to rebuild the samples.
13
+ export function linePlan(context, timing) {
14
+ if (!linesLevelled(context.config) || timing === undefined)
15
+ return { values: [], keys: [] };
16
+ return { values: [], keys: [timing.key] };
17
+ }
@@ -11,10 +11,12 @@ import { runFfmpeg } from "../video/ffmpeg.js";
11
11
  import { planRender } from "../video/plan.js";
12
12
  import { renderSlideshow } from "../video/slideshow.js";
13
13
  import { writePortraits } from "../voices/portraits.js";
14
+ import { linesLevelled } from "./recipe-lines.js";
14
15
  import { exportBed } from "./runtime-export-bed.js";
15
16
  import { exportEdit } from "./runtime-export-edit.js";
16
17
  import { exportSnapshot, retainedOutput, revisionAudio, } from "./runtime-export-inputs.js";
17
18
  import { executeShortExport } from "./runtime-export-short.js";
19
+ import { lineLevelledAudio } from "./runtime-lines.js";
18
20
  import { preparedResult, preparedText, publishResult } from "./runtime-publication.js";
19
21
  export async function executeExportRecipe(deps, context, piece) {
20
22
  if (deps.db
@@ -38,7 +40,15 @@ export async function executeExportRecipe(deps, context, piece) {
38
40
  const pending = allocateAsset(deps, context.work.projectId, filename);
39
41
  const prepared = [];
40
42
  let directory;
43
+ // A multi-voice run's body narration levelled line by line (`runtime-lines.ts`), here
44
+ // until the export is written.
45
+ const linesDirectory = linesLevelled(view.revision.config)
46
+ ? mkdtempSync(join(projectDir(deps.paths, context.work.projectId), "lines-"))
47
+ : undefined;
41
48
  try {
49
+ const sound = linesDirectory === undefined
50
+ ? audio
51
+ : await lineLevelledAudio(deps, context, view, audio, linesDirectory);
42
52
  const captions = snapshot.view.outputs.filter((row) => row.selected && row.available && row.state === "ready" && row.workKey === "subtitles:files");
43
53
  const burn = !wav && view.revision.config.subtitles?.mode === "burn-in";
44
54
  if (burn)
@@ -54,7 +64,7 @@ export async function executeExportRecipe(deps, context, piece) {
54
64
  : "off";
55
65
  const config = view.revision.config;
56
66
  const segment = (kind) => {
57
- const row = audio.find((one) => one.kind === kind);
67
+ const row = sound.find((one) => one.kind === kind);
58
68
  return row?.path === null || row === undefined
59
69
  ? undefined
60
70
  : { path: row.path, seconds: row.seconds };
@@ -125,7 +135,7 @@ export async function executeExportRecipe(deps, context, piece) {
125
135
  try {
126
136
  await runFfmpeg({
127
137
  bin: deps.ffmpeg,
128
- args: audioExportArgs(audio, mixed),
138
+ args: audioExportArgs(sound, mixed),
129
139
  signal: context.signal,
130
140
  log: deps.log,
131
141
  onProgress,
@@ -201,6 +211,8 @@ export async function executeExportRecipe(deps, context, piece) {
201
211
  discardPreparedAssets(deps, [pending, ...prepared.map((one) => one.asset)]);
202
212
  if (directory !== undefined)
203
213
  rmSync(directory, { recursive: true, force: true });
214
+ if (linesDirectory !== undefined)
215
+ rmSync(linesDirectory, { recursive: true, force: true });
204
216
  }
205
217
  }
206
218
  // The project's images in slideshow order, as files.
@@ -0,0 +1,36 @@
1
+ import { readFileSync } from "node:fs";
2
+ import { join } from "node:path";
3
+ import { levelLines } from "../loudness/line-level.js";
4
+ import { outputPath } from "../storage/layout.js";
5
+ import { linesLevelled } from "./recipe-lines.js";
6
+ import { wordsSchema } from "./runtime-subtitles.js";
7
+ // The sound a multi-voice run plays, with its body narration levelled line by line
8
+ // (`loudness/line-level.ts`) into `directory`, for the video, the audio export and the audio
9
+ // files alike. The words are timed on the timeline these segments lay out, so the body starts
10
+ // at the seconds of the segments before it. Anything else plays as it was.
11
+ export async function lineLevelledAudio(deps, context, view, audio, directory) {
12
+ if (!linesLevelled(view.revision.config))
13
+ return audio;
14
+ const timing = view.outputs.find((row) => row.selected &&
15
+ row.available &&
16
+ row.state === "ready" &&
17
+ row.workKey === "subtitles:timing" &&
18
+ row.output.role === "subtitle_words");
19
+ if (timing === undefined)
20
+ throw new Error("The narration's word timing, which evens out the speakers' volume, is missing. Use More → Render the video again in the Video section, then Try again.");
21
+ const { words } = wordsSchema.parse(JSON.parse(readFileSync(outputPath(deps.paths, context.work.projectId, timing.output.path), "utf8")));
22
+ const run = { bin: deps.ffmpeg, log: deps.log, signal: context.signal };
23
+ const levelled = [];
24
+ let offset = 0;
25
+ for (const segment of audio) {
26
+ if (segment.kind === "body" && segment.path !== null) {
27
+ const target = join(directory, "body-lines.wav");
28
+ const done = await levelLines(run, segment.path, target, words, offset);
29
+ levelled.push(done ? { ...segment, path: target } : segment);
30
+ }
31
+ else
32
+ levelled.push(segment);
33
+ offset += segment.seconds;
34
+ }
35
+ return levelled;
36
+ }
@@ -283,9 +283,12 @@ export function describeMismatch(chunks, segment, mismatch) {
283
283
  const expected = mismatch.expected === "" ? "the end of the text" : `"${mismatch.expected}…"`;
284
284
  const heard = mismatch.heard === "" ? "no more speech" : `"${mismatch.heard.toLowerCase()}…"`;
285
285
  const fix = chunk === undefined
286
- ? "Check that part of the narration, regenerate it in Edit project → Narration, then Continue the run."
287
- : `In Edit project → Narration, regenerate narration chunk ${String(at + 1)}, then Continue the run.`;
288
- return `Subtitles stopped matching the audio at ${where}. The text expected ${expected} but the audio has ${heard} The recording there probably skips or changes words. ${fix}`;
286
+ ? `Listen at ${clock(mismatch.at)}: if words are missing or wrong, regenerate that part in Edit project → Narration, then Continue the run.`
287
+ : `Listen at ${clock(mismatch.at)}: if words are missing or wrong, regenerate narration chunk ${String(at + 1)} in Edit project → Narration, then Continue the run.`;
288
+ // A voice that reads a year, an abbreviation or a name its own way reads it the same way
289
+ // again, so a remake would cost a chunk and fail here once more.
290
+ const same = "If it says them right, only another way (a year, an abbreviation or a name read its own way), regenerating gives the same reading: use Download diagnostics in Settings and report it.";
291
+ return `Subtitles stopped matching the audio at ${where}. The text expected ${expected} but the audio has ${heard} The recording there probably skips or changes words. ${fix} ${same}`;
289
292
  }
290
293
  function comparable(text) {
291
294
  return text
@@ -10,6 +10,7 @@ import { runFfmpeg } from "../video/ffmpeg.js";
10
10
  import { audioChapters, audioFileArgs, ffmetadata } from "../voices/audio-files.js";
11
11
  import { turnStarts } from "../voices/timing.js";
12
12
  import { exportSnapshot, revisionAudio } from "./runtime-export-inputs.js";
13
+ import { lineLevelledAudio } from "./runtime-lines.js";
13
14
  import { preparedResult, publishResult } from "./runtime-publication.js";
14
15
  import { wordsSchema } from "./runtime-subtitles.js";
15
16
  // The listening files of a multi-voice run: the MP3 and the M4B of the narration timeline,
@@ -64,9 +65,11 @@ export async function executeVoicesRecipe(deps, context, piece) {
64
65
  let sound = audio;
65
66
  if (goal !== undefined) {
66
67
  const mixed = join(directory, "mix.wav");
68
+ // A multi-voice run's speakers levelled line by line first (`runtime-lines.ts`).
69
+ const levelled = await lineLevelledAudio(deps, context, view, audio, directory);
67
70
  await runFfmpeg({
68
71
  bin: deps.ffmpeg,
69
- args: audioExportArgs(audio, mixed),
72
+ args: audioExportArgs(levelled, mixed),
70
73
  signal: context.signal,
71
74
  log: deps.log,
72
75
  onProgress: () => { },
@@ -61,7 +61,7 @@ Has the text model add delivery directions and sounds, such as a sigh or a laugh
61
61
  | --- | --- | --- |
62
62
  | **Narration Preparation** | Pick a narration prompt from the Library, or **Off**. One text-model call per narration chunk and per intro or outro entry. | Off |
63
63
 
64
- It needs the Inworld TTS-2 model. With another model, the field says "Choose Inworld TTS-2 or turn preparation Off." With several speakers it works per speaker: each turn of a speaker on Inworld TTS-2 is prepared on its own, with the speaker's name and role, so every voice gets its own cues. At least one speaker must use TTS-2; turns of speakers on other voices are spoken as written. See [Multiple Voices](Multiple-Voices). When preparation is on, the row offers **Choose text generation under Article** to jump to the text model.
64
+ It needs the Inworld TTS-2 model. With another model, the field says "Choose Inworld TTS-2 or turn preparation Off." With several speakers it works per speaker: each turn of a speaker on Inworld TTS-2 is prepared on its own, with the speaker's name and role, so every voice gets its own cues. At least one speaker must use TTS-2; turns of speakers on other voices are spoken as written. See [Multiple Voices](Multiple-Voices). Preparation is written by the text model: when nothing earlier needs one (a pasted article, say), **Text generation** appears in this row.
65
65
 
66
66
  ### Pronunciation Glossary
67
67
 
@@ -104,7 +104,7 @@ If you have the same draft open in another tab or window and it was saved there,
104
104
  | Row | What it holds | Details |
105
105
  | --- | --- | --- |
106
106
  | **Title and keywords** | The title pattern and every keyword value | This page |
107
- | **Article** | Article source, article prompt, text generation (LLM, model, thinking), research | [Play Title and Article](Play-Title-and-Article) |
107
+ | **Article** | Article source, article prompt, text generation (LLM, model, thinking) when the article or research is written, research | [Play Title and Article](Play-Title-and-Article) |
108
108
  | **Narration** | Narration source, TTS, model, voice, speakers, Audio Advanced (chunking, intro, outro, preparation, glossary, aliases, tables and figures) | [Play Narration](Play-Narration) |
109
109
  | **Images** | Images source, provider, model, effort, image prompts, establishing image, more images for long videos | [Play Images](Play-Images) |
110
110
  | **Video and style** | Video source, seconds per image, zoom, motion, cuts, the Look, ambient sound, pauses and volume, silence, frame format, captions, preview text | [Play Video and Style](Play-Video-and-Style) |
@@ -52,7 +52,7 @@ Include everything that should be read, headings too. Headings become chapters.
52
52
 
53
53
  One text model writes everything this run needs in words: the article, the research, the thumbnail's image prompt, intro and outro entries set to LLM, Narration Preparation, table and figure descriptions, the YouTube description and picking shorts. Each is a separate call, counted in usage.
54
54
 
55
- The **Text generation** section appears under the Article row whenever the run needs a text model. **Settings** beside it opens Settings, where you set the default for new drafts.
55
+ The **Text generation** section appears whenever the run needs a text model, in the row of the first thing that needs it, and its line says what it is used for. It is under **Article** when the article or research is written for you; under **Narration** when only the narration needs it, such as an audiobook whose speakers are worked out from a book you pasted; and under **Outputs** when only the YouTube description, the shorts or the thumbnail prompt need it. There is one text model per run, wherever it shows. **Settings** beside it opens Settings, where you set the default for new drafts.
56
56
 
57
57
  | Option | What it does | Default |
58
58
  | --- | --- | --- |
@@ -109,9 +109,9 @@ A section for a stage that is switched off says "*Article* is off for this run."
109
109
 
110
110
  ### Re-run a whole stage
111
111
 
112
- Each stage section has its own **More actions for ...** menu (the three dots in the section head) with the rare, whole-stage actions:
112
+ Each stage section's head has a button for making that whole stage again, named for what it makes:
113
113
 
114
- | Menu item | What it does |
114
+ | Button | What it does |
115
115
  | --- | --- |
116
116
  | **Research again** | Fresh research, then the affected article, narration, thumbnail and exports. |
117
117
  | **Write the article again** | A fresh article, then the affected narration, thumbnail and exports. |