voila-recorder 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/AGENTS.md CHANGED
@@ -58,8 +58,12 @@ npx -y voila-recorder rerender <dir> --voice bf_emma
58
58
  - Narration style: short sentences, product language, 8-15 words per beat.
59
59
  - Prefer `zoom` with a `selector` over a raw `level`: voila measures the element
60
60
  and picks the level and camera centre so nothing gets cropped.
61
- - Voices: 28 English (US/UK). af_heart (A) is the default, af_bella (A-) and
62
- bf_emma (British) are the other good ones. `speed` 0.5-1.6 sets pace.
61
+ - Voices: Kokoro is English only (af_heart A, af_bella A-, bf_emma British).
62
+ `speed` 0.5-1.6 sets pace.
63
+ - Other languages: set `voice:` per narration step, so one demo can mix them.
64
+ `say:Monica` uses a macOS system voice (~50 languages, `voila voices --all`);
65
+ `--tts-cmd 'engine -o {out} "{text}"'` plugs in any engine on any platform;
66
+ `audio: clip.mp3` on a step uses a file you already have.
63
67
  - Sign-in walls: recording refuses to film a login page. Run `voila login <url>`
64
68
  (or the voila_login tool), let the HUMAN sign in in the window that opens, and
65
69
  the session persists in a local profile for every later recording.
package/README.md CHANGED
@@ -73,6 +73,28 @@ on-device, no cloud, no API keys ([audio.js](audio.js)). Falls back to macOS
73
73
  `say` if Kokoro can't load. Voices: `af_heart` (default), `af_bella`,
74
74
  `am_adam`, … (`voice` param). Disable with `narrate: false` / `--no-narrate`.
75
75
 
76
+ **Other languages, and mixing them.** Kokoro's JS port speaks English only, so
77
+ non-English narration comes from one of two other engines, chosen per step:
78
+
79
+ ```yaml
80
+ - action: hover
81
+ selector: h1
82
+ narration: "This part is English." # Kokoro
83
+ - action: scroll_to
84
+ selector: "#pricing"
85
+ voice: "say:Mónica" # macOS system voice
86
+ narration: "Esta parte está en español."
87
+ - action: wait
88
+ ms: 500
89
+ audio: ./clips/intro-ja.mp3 # a clip you already have
90
+ ```
91
+
92
+ `voila voices --all` lists the ~180 system voices across ~50 languages on
93
+ macOS. On Linux and Windows use any engine you like:
94
+ `--tts-cmd 'piper -m es.onnx -f {out} -- "{text}"'` ({out}, {text}, {voice}).
95
+ Every step is still paced to its own spoken clip, so mixed-language demos stay
96
+ in sync.
97
+
76
98
  ## Recipes — demos as code
77
99
 
78
100
  Every video ships with its source: `recipe.json` (URL + steps + narration +
package/RECIPE.md CHANGED
@@ -51,6 +51,10 @@ Each step: `{ action, ...params, caption?, narration?, optional?, pause? }`
51
51
  | `zoom` | `level` (1-3) or `selector`, `ms` | camera zoom; with a selector it frames that element |
52
52
  | `wait` | `ms` | hold (cursor keeps breathing on long holds) |
53
53
 
54
+ Any step may also carry `voice:` (Kokoro id, `say:Name`, or an id for your
55
+ `--tts-cmd` engine) and `audio:` (a ready-made clip, skipping TTS entirely).
56
+ Because voice is per step, a recipe can mix languages.
57
+
54
58
  `caption` renders as a lower-third; `narration` is spoken by on-device TTS and
55
59
  **paces the segment** — the recording holds until the clip finishes, so a
56
60
  recreated demo re-times itself to whatever voice regenerates it.
package/audio.js CHANGED
@@ -31,7 +31,7 @@ async function getKokoro() {
31
31
  return kokoroInstance;
32
32
  }
33
33
 
34
- async function synthKokoro(texts, dir, voice, onStatus, speed = 1) {
34
+ async function _unusedSynthKokoro(texts, dir, voice, onStatus, speed = 1) {
35
35
  // Fail loudly on a bad voice name rather than silently using the default.
36
36
  if (voice && /^[a-z]{2}_/.test(voice) && !voices.isValid(voice)) {
37
37
  throw new Error(`unknown voice "${voice}". Try: ${voices.suggest(voice).join(', ')} (run \`voila voices\` for all ${voices.ranked().length})`);
@@ -77,7 +77,7 @@ async function pickSayVoice(preferred) {
77
77
  return 'Samantha';
78
78
  }
79
79
 
80
- async function synthSay(texts, dir, voice, onStatus) {
80
+ async function _unusedSynthSay(texts, dir, voice, onStatus) {
81
81
  const v = await pickSayVoice(voice);
82
82
  onStatus(`narrating with say (${v})`);
83
83
  const clips = [];
@@ -90,25 +90,113 @@ async function synthSay(texts, dir, voice, onStatus) {
90
90
  }
91
91
 
92
92
  // --- pipeline entry ----------------------------------------------------------
93
+ // Each narration item is {text, voice?, audio?}. Voice decides the engine:
94
+ // af_heart -> Kokoro (English, best quality, cross platform)
95
+ // say:Monica -> a system voice (macOS; ~50 languages)
96
+ // (--tts-cmd set) -> any external engine, any language
97
+ // A step can also point at a ready-made file with `audio:`, which bypasses TTS.
98
+ // Because the voice is per item, one demo can mix languages.
99
+
100
+ function engineFor(voice, ttsCmd) {
101
+ // Supplying --tts-cmd means "use my engine", unless a step names a specific
102
+ // Kokoro or system voice, which then wins for that step.
103
+ if (!voice) {
104
+ if (ttsCmd) return 'cmd';
105
+ return process.env.VOILA_TTS === 'say' ? 'say' : 'kokoro';
106
+ }
107
+ if (String(voice).startsWith('cmd:') || (ttsCmd && String(voice).startsWith('cmd'))) return 'cmd';
108
+ if (String(voice).startsWith('say:')) return 'say';
109
+ if (voices.isValid(voice)) return 'kokoro';
110
+ if (voices.isSystemVoice(voice)) return 'say';
111
+ if (ttsCmd) return 'cmd';
112
+ throw new Error(
113
+ `unknown voice "${voice}". Kokoro (English): ${voices.suggest(voice).join(', ')}. ` +
114
+ `For other languages use a system voice like "say:Monica" (see \`voila voices --all\`) ` +
115
+ `or supply --tts-cmd for your own engine.`
116
+ );
117
+ }
118
+
119
+ async function sayOne(text, file, voiceName, speed) {
120
+ const rate = Math.round(185 * (speed || 1));
121
+ await run('say', ['-v', voiceName, '-r', String(rate), '-o', file, text]);
122
+ }
123
+
124
+ async function cmdOne(text, file, voiceName, ttsCmd) {
125
+ const cmd = ttsCmd
126
+ .replaceAll('{text}', text.replace(/"/g, '\\"'))
127
+ .replaceAll('{out}', file)
128
+ .replaceAll('{voice}', voiceName || '');
129
+ await new Promise((res, rej) => {
130
+ require('child_process').exec(cmd, { maxBuffer: 1e7 }, (err, _o, se) =>
131
+ err ? rej(new Error(`tts-cmd failed: ${String(se || err.message).slice(0, 200)}`)) : res());
132
+ });
133
+ if (!fs.existsSync(file)) throw new Error(`tts-cmd produced no file at ${file}`);
134
+ }
93
135
 
94
- // Synthesize narration clips up front so the recorder can pace segments to the
95
- // spoken durations. Returns {clips: [{file, durMs}], voice, backend}.
96
- async function prepareNarration(texts, dir, voice, onStatus = () => {}, speed = 1) {
136
+ // Synthesize every narration clip up front so the recorder can pace segments
137
+ // to real spoken durations. Returns {clips:[{file,durMs}], voice, backend}.
138
+ async function prepareNarration(items, dir, defaultVoice, onStatus = () => {}, speed = 1, ttsCmd = null) {
97
139
  fs.mkdirSync(dir, { recursive: true });
98
- const backend = process.env.VOILA_TTS || 'kokoro';
99
- if (backend === 'kokoro') {
100
- try {
101
- return await synthKokoro(texts, dir, voice, onStatus, speed);
102
- } catch (e) {
103
- if (/unknown voice/.test(e.message)) throw e; // user error, not a fallback case
104
- onStatus(`kokoro unavailable (${e.message.slice(0, 80)})`);
140
+ const norm = items.map(it => (typeof it === 'string' ? { text: it } : it));
141
+ const clips = [];
142
+ const used = new Set();
143
+ let kokoro = null;
144
+
145
+ for (let i = 0; i < norm.length; i++) {
146
+ const it = norm[i];
147
+ const voice = it.voice || defaultVoice || null;
148
+
149
+ // A pre-made audio file wins over any engine.
150
+ if (it.audio) {
151
+ if (!fs.existsSync(it.audio)) throw new Error(`audio file not found: ${it.audio}`);
152
+ clips.push({ file: it.audio, durMs: await ffDurationMs(it.audio) });
153
+ used.add('file');
154
+ continue;
155
+ }
156
+
157
+ const engine = engineFor(voice, ttsCmd);
158
+ const ext = engine === 'kokoro' ? 'wav' : engine === 'say' ? 'aiff' : 'wav';
159
+ const file = path.join(dir, `seg${i}.${ext}`);
160
+
161
+ if (engine === 'kokoro') {
162
+ if (!kokoro) {
163
+ onStatus('loading Kokoro TTS');
164
+ require('./modelcache').useStableCache();
165
+ const { KokoroTTS } = require('kokoro-js');
166
+ kokoro = await KokoroTTS.from_pretrained('onnx-community/Kokoro-82M-v1.0-ONNX', { dtype: 'q8' });
167
+ }
168
+ const v = voice && voices.isValid(voice) ? voice : 'af_heart';
169
+ onStatus(`narrating ${i + 1}/${norm.length} with Kokoro (${v})`);
170
+ const audio = await kokoro.generate(it.text, { voice: v, speed });
171
+ await audio.save(file);
172
+ clips.push({
173
+ file,
174
+ durMs: audio.audio && audio.sampling_rate
175
+ ? Math.round((audio.audio.length / audio.sampling_rate) * 1000)
176
+ : await ffDurationMs(file),
177
+ });
178
+ used.add(`kokoro:${v}`);
179
+ } else if (engine === 'say') {
180
+ if (process.platform !== 'darwin') {
181
+ throw new Error(`system voices need macOS. Use a Kokoro voice for English, or --tts-cmd on this platform.`);
182
+ }
183
+ const name = String(voice).replace(/^say:/, '');
184
+ onStatus(`narrating ${i + 1}/${norm.length} with system voice (${name})`);
185
+ await sayOne(it.text, file, name, speed);
186
+ clips.push({ file, durMs: await ffDurationMs(file) });
187
+ used.add(`say:${name}`);
188
+ } else {
189
+ onStatus(`narrating ${i + 1}/${norm.length} with tts-cmd`);
190
+ await cmdOne(it.text, file, String(voice || '').replace(/^cmd:/, ''), ttsCmd);
191
+ clips.push({ file, durMs: await ffDurationMs(file) });
192
+ used.add('cmd');
105
193
  }
106
194
  }
107
- if (process.platform === 'darwin') return synthSay(texts, dir, voice, onStatus);
108
- throw new Error('no TTS backend available');
195
+
196
+ return { clips, voice: [...used].join(', ') || 'none', backend: [...used].join(', ') };
109
197
  }
110
198
 
111
- async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1, prepared = null, onStatus = () => {} } = {}) {
199
+ async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1, ttsCmd = null, prepared = null, onStatus = () => {} } = {}) {
112
200
  const segs = (meta.segments || []).filter(s => s.narration);
113
201
  if (!segs.length) {
114
202
  fs.copyFileSync(videoIn, videoOut);
@@ -120,7 +208,9 @@ async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1,
120
208
  let synth = prepared && prepared.clips.length === segs.length ? prepared : null;
121
209
  if (!synth) {
122
210
  try {
123
- synth = await prepareNarration(segs.map(s => s.narration), dir, voice, onStatus, speed);
211
+ synth = await prepareNarration(
212
+ segs.map(s => ({ text: s.narration, voice: s.voice, audio: s.audio })),
213
+ dir, voice, onStatus, speed, ttsCmd);
124
214
  } catch (e) {
125
215
  onStatus(`narration skipped: ${e.message}`);
126
216
  fs.copyFileSync(videoIn, videoOut);
@@ -154,7 +244,7 @@ async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1,
154
244
  ff.on('error', rej);
155
245
  });
156
246
 
157
- fs.rmSync(dir, { recursive: true, force: true });
247
+ fs.rmSync(dir, { recursive: true, force: true }); // only generated clips live here
158
248
  return { narrated: true, voice: synth.voice, backend: synth.backend, segments: clips.length };
159
249
  }
160
250
 
package/cli.js CHANGED
@@ -20,10 +20,10 @@ function arg(name, fallback = null) {
20
20
  const USAGE = `usage:
21
21
  voila doctor (check + download everything voila needs)
22
22
  voila outline <url> [--device desktop|mobile|tablet]
23
- voila record <url> [--steps f.yaml] [--device mobile] [--voice name] [--speed 1] [--no-narrate] [--headful] [--keep-frames] [--no-dismiss] [--dismiss sel] [--out dir] [--profile dir]
23
+ voila record <url> [--steps f.yaml] [--device mobile] [--voice name] [--speed 1] [--tts-cmd tmpl] [--no-narrate] [--headful] [--keep-frames] [--no-dismiss] [--dismiss sel] [--out dir] [--profile dir]
24
24
  voila review <video.mp4> [--frames 12] [--out dir]
25
25
  voila login <url> [--profile dir] (sign in yourself; session is saved locally)
26
- voila voices (list every narration voice, best first)
26
+ voila voices [--all] (Kokoro voices; --all adds system voices for other languages)
27
27
  voila fork <video.mp4> [--url u] [--voice v] [--print] [--out dir]
28
28
  voila rerender <dir> [--voice v] [--speed n] (needs --keep-frames on the original)
29
29
  voila skill (install the voila skill into ~/.claude/skills)
@@ -49,7 +49,9 @@ const USAGE = `usage:
49
49
  process.exit(r.ready ? 0 : 1);
50
50
  }
51
51
  if (cmd === 'voices') {
52
- console.log(require('./voices').format());
52
+ const v = require('./voices');
53
+ console.log(process.argv.includes('--all') ? v.formatAll() : v.format());
54
+ if (!process.argv.includes('--all')) console.log('\nNon-English? run: voila voices --all');
53
55
  return;
54
56
  }
55
57
  if (cmd === 'skill') {
@@ -124,6 +126,7 @@ const USAGE = `usage:
124
126
  const r = await rerender(dir, {
125
127
  voice: arg('--voice'),
126
128
  speed: Number(arg('--speed', '1')) || 1,
129
+ ttsCmd: arg('--tts-cmd'),
127
130
  narrate: !process.argv.includes('--no-narrate'),
128
131
  onStatus: m => console.error('[voila]', m),
129
132
  });
@@ -166,6 +169,7 @@ const USAGE = `usage:
166
169
  narrate: !process.argv.includes('--no-narrate'),
167
170
  voice: arg('--voice'),
168
171
  speed: Number(arg('--speed', '1')) || 1,
172
+ ttsCmd: arg('--tts-cmd'),
169
173
  keepFrames: process.argv.includes('--keep-frames'),
170
174
  onStatus: s => console.error('[voila]', s),
171
175
  });
package/mcp.js CHANGED
@@ -37,7 +37,7 @@ function getSession(device) {
37
37
  return sessions.get(key);
38
38
  }
39
39
 
40
- const server = new McpServer({ name: 'voila', version: '0.6.0' });
40
+ const server = new McpServer({ name: 'voila', version: '0.7.0' });
41
41
  const deviceParam = z.enum(['desktop', 'mobile', 'tablet']).optional().default('desktop');
42
42
 
43
43
  server.tool(
@@ -61,6 +61,7 @@ server.tool(
61
61
  'caption is burned into the video as a lower-third; narration is spoken via on-device TTS (Kokoro) at that step, ' +
62
62
  'and segment pacing automatically stretches to fit each narration clip — no need to pad waits. ' +
63
63
  'Steps marked optional:true are skipped on failure instead of aborting. ' +
64
+ 'A step may set its own voice: (mixing languages within one demo) or audio: (a ready-made clip). ' +
64
65
  'device selects the recorded viewport (mobile emulates an iPhone-class device). ' +
65
66
  'On failure the error names the failing step and includes the live page outline — patch the steps and retry. ' +
66
67
  'Returns the MP4 path, the recipe path, and any warnings.',
@@ -68,16 +69,17 @@ server.tool(
68
69
  url: z.string().url(),
69
70
  steps_yaml: z.string().optional(),
70
71
  narrate: z.boolean().optional().default(true),
71
- voice: z.string().optional().describe('narration voice id, e.g. af_heart (A), af_bella (A-), bf_emma (B-, British). Call voila_voices for the full list.'),
72
+ voice: z.string().optional().describe('default narration voice. Kokoro is ENGLISH ONLY (af_heart, af_bella, bf_emma). For other languages use a macOS system voice ("say:Monica") or set tts_cmd. Per-step `voice:` overrides this, so one demo can mix languages.'),
72
73
  speed: z.number().min(0.5).max(1.6).optional().default(1).describe('narration speed; 0.9 reads calmer'),
74
+ tts_cmd: z.string().optional().describe('external TTS engine template for any language/platform, e.g. \'piper -m es.onnx -f {out} -- "{text}"\'. Placeholders: {out} {text} {voice}.'),
73
75
  device: deviceParam,
74
76
  },
75
- async ({ url, steps_yaml, narrate, voice, speed, device }) => enqueue(async () => {
77
+ async ({ url, steps_yaml, narrate, voice, speed, tts_cmd, device }) => enqueue(async () => {
76
78
  const steps = steps_yaml ? yaml.load(steps_yaml) : null;
77
79
  const workDir = path.join(__dirname, 'recordings', `mcp-${Date.now()}`);
78
80
  fs.mkdirSync(workDir, { recursive: true });
79
81
  const result = await produceDemo(getSession(device), {
80
- url, steps, workDir, narrate, voice: voice || null, speed,
82
+ url, steps, workDir, narrate, voice: voice || null, speed, ttsCmd: tts_cmd || null,
81
83
  onStatus: () => {},
82
84
  });
83
85
  return {
@@ -116,11 +118,19 @@ server.tool(
116
118
 
117
119
  server.tool(
118
120
  'voila_voices',
119
- 'List every narration voice with its quality grade, best first. Use before voila_record when the ' +
120
- 'user asks for a different voice, an accent, or a male or female narrator.',
121
- {},
122
- async () => ({
123
- content: [{ type: 'text', text: JSON.stringify(voiceCatalogue.ranked(), null, 2) }],
121
+ 'List narration voices. Kokoro voices are English only, with quality grades. Pass system:true to ' +
122
+ 'also get the machine\'s system voices, which is how you narrate other languages (macOS only; on ' +
123
+ 'Linux or Windows use tts_cmd instead). Use before voila_record when the user asks for a different ' +
124
+ 'voice, an accent, a male or female narrator, or a non-English language.',
125
+ { system: z.boolean().optional().default(false) },
126
+ async ({ system }) => ({
127
+ content: [{ type: 'text', text: JSON.stringify({
128
+ kokoro: voiceCatalogue.ranked(),
129
+ englishOnly: true,
130
+ system: system ? voiceCatalogue.systemVoices() : undefined,
131
+ systemLanguages: system ? Object.keys(voiceCatalogue.systemLanguages()) : undefined,
132
+ note: 'Non-English: use a system voice (say:Name) on macOS, or tts_cmd on any platform.',
133
+ }, null, 2) }],
124
134
  })
125
135
  );
126
136
 
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "voila-recorder",
3
- "version": "0.6.0",
4
- "description": "Permission-free product demo recorder: URL in, narrated auto-zoomed MP4 out \u2014 with the recipe embedded in the video. Agent-native (MCP), fully on-device.",
3
+ "version": "0.7.0",
4
+ "description": "Permission-free product demo recorder: URL in, narrated auto-zoomed MP4 out — with the recipe embedded in the video. Agent-native (MCP), fully on-device.",
5
5
  "license": "MIT",
6
6
  "main": "pipeline.js",
7
7
  "bin": {
@@ -62,4 +62,4 @@
62
62
  "playwright": "^1.48.0",
63
63
  "sharp": "^0.33.5"
64
64
  }
65
- }
65
+ }
package/pipeline.js CHANGED
@@ -44,15 +44,16 @@ function embedRecipe(videoIn, videoOut, recipe) {
44
44
  });
45
45
  }
46
46
 
47
- async function produceDemo(session, { url, mode = 'auto', steps = null, workDir, voice = null, speed = 1, narrate = true, keepFrames = false, onStatus = () => {} }) {
47
+ async function produceDemo(session, { url, mode = 'auto', steps = null, workDir, voice = null, speed = 1, ttsCmd = null, narrate = true, keepFrames = false, onStatus = () => {} }) {
48
48
  // Steps mode: synthesize narration BEFORE recording so segment pacing and
49
49
  // caption lifetimes match the spoken clip durations exactly.
50
50
  let prepared = null;
51
51
  if (steps && narrate) {
52
- const texts = steps.filter(s => s.narration).map(s => s.narration);
53
- if (texts.length) {
52
+ const items = steps.filter(s => s.narration)
53
+ .map(s => ({ text: s.narration, voice: s.voice, audio: s.audio }));
54
+ if (items.length) {
54
55
  try {
55
- prepared = await prepareNarration(texts, path.join(workDir, 'tts'), voice, onStatus, speed);
56
+ prepared = await prepareNarration(items, path.join(workDir, 'tts'), voice, onStatus, speed, ttsCmd);
56
57
  let i = 0;
57
58
  for (const s of steps) if (s.narration) s._narrDurMs = prepared.clips[i++].durMs;
58
59
  } catch (e) {
@@ -69,7 +70,7 @@ async function produceDemo(session, { url, mode = 'auto', steps = null, workDir,
69
70
  await render(meta, raw, { onStatus });
70
71
 
71
72
  let narration = { narrated: false };
72
- if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, prepared, onStatus });
73
+ if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, ttsCmd, prepared, onStatus });
73
74
  else fs.copyFileSync(raw, narrated);
74
75
 
75
76
  const recipe = buildRecipe({ url, mode, steps, meta });
@@ -88,7 +89,7 @@ async function produceDemo(session, { url, mode = 'auto', steps = null, workDir,
88
89
 
89
90
  // Re-produce the video from frames already on disk: no browser, no re-driving
90
91
  // the page. Used to swap the narration voice or speed after the fact.
91
- async function rerender(workDir, { voice = null, speed = 1, narrate = true, onStatus = () => {} } = {}) {
92
+ async function rerender(workDir, { voice = null, speed = 1, ttsCmd = null, narrate = true, onStatus = () => {} } = {}) {
92
93
  const metaPath = path.join(workDir, 'meta.json');
93
94
  if (!fs.existsSync(metaPath)) throw new Error(`no meta.json in ${workDir}`);
94
95
  const meta = JSON.parse(fs.readFileSync(metaPath, 'utf8'));
@@ -102,7 +103,7 @@ async function rerender(workDir, { voice = null, speed = 1, narrate = true, onSt
102
103
  await render(meta, raw, { onStatus });
103
104
 
104
105
  let narration = { narrated: false };
105
- if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, onStatus });
106
+ if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, ttsCmd, onStatus });
106
107
  else fs.copyFileSync(raw, narrated);
107
108
 
108
109
  const recipe = JSON.parse(fs.readFileSync(path.join(workDir, 'recipe.json'), 'utf8'));
package/recorder.js CHANGED
@@ -40,8 +40,11 @@ class Timeline {
40
40
  this.zoom = 1;
41
41
  this.center = null;
42
42
  }
43
- recordSegment(caption, narration, dur = null) {
44
- this.segments.push({ t: Date.now(), caption: caption || null, narration: narration || null, dur });
43
+ recordSegment(caption, narration, dur = null, voice = null, audio = null) {
44
+ this.segments.push({
45
+ t: Date.now(), caption: caption || null, narration: narration || null, dur,
46
+ ...(voice ? { voice } : {}), ...(audio ? { audio } : {}),
47
+ });
45
48
  }
46
49
  recordMove(to, dur) {
47
50
  this.moves.push({ t: Date.now(), from: { ...this.pos }, to: { ...to }, dur });
@@ -0,0 +1,10 @@
1
+ // CI fixture: a stand-in "external TTS engine". Writes a short tone to {out}
2
+ // so the --tts-cmd plumbing can be exercised on every platform, including
3
+ // ones with no system voices installed.
4
+ const { execFileSync } = require('child_process');
5
+ const ffmpeg = require('ffmpeg-static');
6
+ const out = process.argv[2];
7
+ const text = process.argv.slice(3).join(' ');
8
+ const seconds = Math.max(1, Math.min(8, text.split(/\s+/).length / 3)).toFixed(2);
9
+ execFileSync(ffmpeg, ['-y', '-f', 'lavfi', '-i', `sine=frequency=340:duration=${seconds}`, '-ac', '1', out], { stdio: 'pipe' });
10
+ console.log(`fixture wrote ${seconds}s to ${out}`);
@@ -64,7 +64,14 @@ Reference: https://voila.anzalabidi.dev/llms.txt · https://github.com/anzal1/vo
64
64
  for them in chat.
65
65
  - Prefer `zoom` with a `selector` over a bare `level`: voila measures the
66
66
  element and picks the level and camera centre, so nothing is cropped.
67
- - Voices: `voila voices` lists 28 English voices with quality grades. af_heart
67
+ - Languages: narration voice is per step, so demos can mix languages. Kokoro
68
+ is ENGLISH ONLY (af_heart A, af_bella A-, bf_emma British). For other
69
+ languages use a macOS system voice (`voice: "say:Monica"`, `voila voices
70
+ --all` lists ~180 across ~50 languages), or `--tts-cmd` with any engine on
71
+ any platform (placeholders {out} {text} {voice}), or point a step at a
72
+ ready-made clip with `audio: file.mp3`. On Linux/Windows non-English needs
73
+ --tts-cmd: tell the user rather than silently narrating in English.
74
+ - Voices: `voila voices` lists the English voices with quality grades. af_heart
68
75
  (A) default, af_bella (A-), af_nicole (B-), bf_emma (B-, British). `--speed`
69
76
  or the speed param (0.5-1.6) changes pace; 0.9 reads calmer.
70
77
  - A failing step is retried once automatically; warnings appear in the result.
package/tour.js CHANGED
@@ -242,7 +242,7 @@ async function runSteps(page, tl, steps, opts) {
242
242
  const step = steps[si];
243
243
  if (step.caption || step.narration) {
244
244
  await finishSegment();
245
- tl.recordSegment(step.caption, step.narration, step._narrDurMs || null);
245
+ tl.recordSegment(step.caption, step.narration, step._narrDurMs || null, step.voice || null, step.audio || null);
246
246
  segStart = Date.now();
247
247
  segMinMs = (step._narrDurMs || 0) + 600;
248
248
  }
package/voices.js CHANGED
@@ -55,4 +55,52 @@ function format() {
55
55
  return lines.join('\n');
56
56
  }
57
57
 
58
- module.exports = { allVoices, ranked, isValid, suggest, format };
58
+ // --- system voices (macOS `say`) ---------------------------------------------
59
+ // Kokoro is English-only in JS (its other voice files ship without a
60
+ // grapheme-to-phoneme stage for those languages). Every Mac already carries
61
+ // ~180 voices across ~50 languages, so those cover non-English narration.
62
+
63
+ let sysCache = null;
64
+ function systemVoices() {
65
+ if (sysCache) return sysCache;
66
+ sysCache = [];
67
+ if (process.platform !== 'darwin') return sysCache;
68
+ try {
69
+ const { execFileSync } = require('child_process');
70
+ const out = String(execFileSync('say', ['-v', '?'], { maxBuffer: 4e6 }));
71
+ for (const line of out.split('\n')) {
72
+ const m = /^(.+?)\s{2,}([a-z]{2}_[A-Z]{2})\s/.exec(line);
73
+ if (m) sysCache.push({ id: `say:${m[1].trim()}`, name: m[1].trim(), language: m[2].replace('_', '-'), gender: '', grade: 'system' });
74
+ }
75
+ } catch { /* no say binary */ }
76
+ return sysCache;
77
+ }
78
+
79
+ function systemLanguages() {
80
+ const langs = {};
81
+ for (const v of systemVoices()) (langs[v.language] = langs[v.language] || []).push(v.name);
82
+ return langs;
83
+ }
84
+
85
+ function isSystemVoice(id) {
86
+ if (!id) return false;
87
+ const name = String(id).replace(/^say:/, '').toLowerCase();
88
+ return systemVoices().some(v => v.name.toLowerCase() === name);
89
+ }
90
+
91
+ function formatAll() {
92
+ const lines = [format()];
93
+ const langs = systemLanguages();
94
+ const codes = Object.keys(langs).sort();
95
+ if (!codes.length) {
96
+ lines.push('\nSystem voices: none found (macOS only). For other languages use --tts-cmd.');
97
+ return lines.join('\n');
98
+ }
99
+ lines.push(`\nSystem voices (macOS, ${systemVoices().length} across ${codes.length} languages)`);
100
+ for (const c of codes) lines.push(` ${c.padEnd(7)} ${langs[c].slice(0, 6).join(', ')}${langs[c].length > 6 ? ` +${langs[c].length - 6}` : ''}`);
101
+ lines.push('\nUse a system voice for non-English narration: --voice "say:Monica"');
102
+ lines.push('Any other engine: --tts-cmd \'piper --model es.onnx -f {out} -- "{text}"\'');
103
+ return lines.join('\n');
104
+ }
105
+
106
+ module.exports = { allVoices, ranked, isValid, suggest, format, formatAll, systemVoices, systemLanguages, isSystemVoice };