voila-recorder 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/AGENTS.md CHANGED
@@ -24,6 +24,9 @@ npx -y voila-recorder record <url> --steps steps.yaml [--device mobile]
24
24
  npx -y voila-recorder review demo.mp4
25
25
  npx -y voila-recorder voices
26
26
  npx -y voila-recorder login <url>
27
+ npx -y voila-recorder doctor
28
+ npx -y voila-recorder fork demo.mp4 [--url other] [--print]
29
+ npx -y voila-recorder rerender <dir> --voice bf_emma
27
30
  ```
28
31
 
29
32
  ## The loop — always follow it
@@ -55,8 +58,12 @@ npx -y voila-recorder login <url>
55
58
  - Narration style: short sentences, product language, 8-15 words per beat.
56
59
  - Prefer `zoom` with a `selector` over a raw `level`: voila measures the element
57
60
  and picks the level and camera centre so nothing gets cropped.
58
- - Voices: 28 English (US/UK). af_heart (A) is the default, af_bella (A-) and
59
- bf_emma (British) are the other good ones. `speed` 0.5-1.6 sets pace.
61
+ - Voices: Kokoro is English only (af_heart A, af_bella A-, bf_emma British).
62
+ `speed` 0.5-1.6 sets pace.
63
+ - Other languages: set `voice:` per narration step, so one demo can mix them.
64
+ `say:Monica` uses a macOS system voice (~50 languages, `voila voices --all`);
65
+ `--tts-cmd 'engine -o {out} "{text}"'` plugs in any engine on any platform;
66
+ `audio: clip.mp3` on a step uses a file you already have.
60
67
  - Sign-in walls: recording refuses to film a login page. Run `voila login <url>`
61
68
  (or the voila_login tool), let the HUMAN sign in in the window that opens, and
62
69
  the session persists in a local profile for every later recording.
package/README.md CHANGED
@@ -28,9 +28,18 @@ voila record <url> --device mobile # iPhone-class viewport (portrait)
28
28
  voila outline <url> # page structure for planning
29
29
  voila review demo.mp4 --frames 12 # frames + recipe for self-review
30
30
  voila serve # web UI
31
+ voila doctor # check + pre-download chromium and the voice model
32
+ voila fork demo.mp4 --url https://other.app # rebuild any voila demo from the recipe inside it
33
+ voila rerender <dir> --voice bf_emma # new voice, no re-recording (needs --keep-frames)
34
+ voila login https://app.example.com # you sign in; session saved locally
35
+ voila voices # 28 narration voices, graded
31
36
  voila mcp # stdio MCP server
32
37
  ```
33
38
 
39
+ First run downloads Chromium (~150MB) and the voice model (~90MB) into
40
+ `~/.cache/voila`, so upgrades do not re-download. Consent banners are dismissed
41
+ before recording, preferring "reject" over "accept".
42
+
34
43
  Devices: `desktop` (1280×800), `mobile` (390×844, touch + mobile UA),
35
44
  `tablet` (834×1112). Cross-platform: verified on macOS and Linux (arm64
36
45
  container); recording is headless-safe for CI.
@@ -64,6 +73,28 @@ on-device, no cloud, no API keys ([audio.js](audio.js)). Falls back to macOS
64
73
  `say` if Kokoro can't load. Voices: `af_heart` (default), `af_bella`,
65
74
  `am_adam`, … (`voice` param). Disable with `narrate: false` / `--no-narrate`.
66
75
 
76
+ **Other languages, and mixing them.** Kokoro's JS port speaks English only, so
77
+ non-English narration comes from one of two other engines, chosen per step:
78
+
79
+ ```yaml
80
+ - action: hover
81
+ selector: h1
82
+ narration: "This part is English." # Kokoro
83
+ - action: scroll_to
84
+ selector: "#pricing"
85
+ voice: "say:Mónica" # macOS system voice
86
+ narration: "Esta parte está en español."
87
+ - action: wait
88
+ ms: 500
89
+ audio: ./clips/intro-ja.mp3 # a clip you already have
90
+ ```
91
+
92
+ `voila voices --all` lists the ~180 system voices across ~50 languages on
93
+ macOS. On Linux and Windows use any engine you like:
94
+ `--tts-cmd 'piper -m es.onnx -f {out} -- "{text}"'` ({out}, {text}, {voice}).
95
+ Every step is still paced to its own spoken clip, so mixed-language demos stay
96
+ in sync.
97
+
67
98
  ## Recipes — demos as code
68
99
 
69
100
  Every video ships with its source: `recipe.json` (URL + steps + narration +
package/RECIPE.md CHANGED
@@ -51,6 +51,10 @@ Each step: `{ action, ...params, caption?, narration?, optional?, pause? }`
51
51
  | `zoom` | `level` (1-3) or `selector`, `ms` | camera zoom; with a selector it frames that element |
52
52
  | `wait` | `ms` | hold (cursor keeps breathing on long holds) |
53
53
 
54
+ Any step may also carry `voice:` (Kokoro id, `say:Name`, or an id for your
55
+ `--tts-cmd` engine) and `audio:` (a ready-made clip, skipping TTS entirely).
56
+ Because voice is per step, a recipe can mix languages.
57
+
54
58
  `caption` renders as a lower-third; `narration` is spoken by on-device TTS and
55
59
  **paces the segment** — the recording holds until the clip finishes, so a
56
60
  recreated demo re-times itself to whatever voice regenerates it.
package/audio.js CHANGED
@@ -22,6 +22,7 @@ async function getKokoro() {
22
22
  // Must load the CJS build (exports map: require → dist/kokoro.cjs): it
23
23
  // resolves bundled voice files via __dirname, while the ESM build loses
24
24
  // __dirname and breaks when cwd isn't the package root.
25
+ require('./modelcache').useStableCache();
25
26
  const { KokoroTTS } = require('kokoro-js');
26
27
  kokoroInstance = await KokoroTTS.from_pretrained(
27
28
  'onnx-community/Kokoro-82M-v1.0-ONNX', { dtype: 'q8' }
@@ -30,7 +31,7 @@ async function getKokoro() {
30
31
  return kokoroInstance;
31
32
  }
32
33
 
33
- async function synthKokoro(texts, dir, voice, onStatus, speed = 1) {
34
+ async function _unusedSynthKokoro(texts, dir, voice, onStatus, speed = 1) {
34
35
  // Fail loudly on a bad voice name rather than silently using the default.
35
36
  if (voice && /^[a-z]{2}_/.test(voice) && !voices.isValid(voice)) {
36
37
  throw new Error(`unknown voice "${voice}". Try: ${voices.suggest(voice).join(', ')} (run \`voila voices\` for all ${voices.ranked().length})`);
@@ -76,7 +77,7 @@ async function pickSayVoice(preferred) {
76
77
  return 'Samantha';
77
78
  }
78
79
 
79
- async function synthSay(texts, dir, voice, onStatus) {
80
+ async function _unusedSynthSay(texts, dir, voice, onStatus) {
80
81
  const v = await pickSayVoice(voice);
81
82
  onStatus(`narrating with say (${v})`);
82
83
  const clips = [];
@@ -89,25 +90,113 @@ async function synthSay(texts, dir, voice, onStatus) {
89
90
  }
90
91
 
91
92
  // --- pipeline entry ----------------------------------------------------------
93
+ // Each narration item is {text, voice?, audio?}. Voice decides the engine:
94
+ // af_heart -> Kokoro (English, best quality, cross platform)
95
+ // say:Monica -> a system voice (macOS; ~50 languages)
96
+ // (--tts-cmd set) -> any external engine, any language
97
+ // A step can also point at a ready-made file with `audio:`, which bypasses TTS.
98
+ // Because the voice is per item, one demo can mix languages.
99
+
100
+ function engineFor(voice, ttsCmd) {
101
+ // Supplying --tts-cmd means "use my engine", unless a step names a specific
102
+ // Kokoro or system voice, which then wins for that step.
103
+ if (!voice) {
104
+ if (ttsCmd) return 'cmd';
105
+ return process.env.VOILA_TTS === 'say' ? 'say' : 'kokoro';
106
+ }
107
+ if (String(voice).startsWith('cmd:') || (ttsCmd && String(voice).startsWith('cmd'))) return 'cmd';
108
+ if (String(voice).startsWith('say:')) return 'say';
109
+ if (voices.isValid(voice)) return 'kokoro';
110
+ if (voices.isSystemVoice(voice)) return 'say';
111
+ if (ttsCmd) return 'cmd';
112
+ throw new Error(
113
+ `unknown voice "${voice}". Kokoro (English): ${voices.suggest(voice).join(', ')}. ` +
114
+ `For other languages use a system voice like "say:Monica" (see \`voila voices --all\`) ` +
115
+ `or supply --tts-cmd for your own engine.`
116
+ );
117
+ }
118
+
119
+ async function sayOne(text, file, voiceName, speed) {
120
+ const rate = Math.round(185 * (speed || 1));
121
+ await run('say', ['-v', voiceName, '-r', String(rate), '-o', file, text]);
122
+ }
123
+
124
+ async function cmdOne(text, file, voiceName, ttsCmd) {
125
+ const cmd = ttsCmd
126
+ .replaceAll('{text}', text.replace(/"/g, '\\"'))
127
+ .replaceAll('{out}', file)
128
+ .replaceAll('{voice}', voiceName || '');
129
+ await new Promise((res, rej) => {
130
+ require('child_process').exec(cmd, { maxBuffer: 1e7 }, (err, _o, se) =>
131
+ err ? rej(new Error(`tts-cmd failed: ${String(se || err.message).slice(0, 200)}`)) : res());
132
+ });
133
+ if (!fs.existsSync(file)) throw new Error(`tts-cmd produced no file at ${file}`);
134
+ }
92
135
 
93
- // Synthesize narration clips up front so the recorder can pace segments to the
94
- // spoken durations. Returns {clips: [{file, durMs}], voice, backend}.
95
- async function prepareNarration(texts, dir, voice, onStatus = () => {}, speed = 1) {
136
+ // Synthesize every narration clip up front so the recorder can pace segments
137
+ // to real spoken durations. Returns {clips:[{file,durMs}], voice, backend}.
138
+ async function prepareNarration(items, dir, defaultVoice, onStatus = () => {}, speed = 1, ttsCmd = null) {
96
139
  fs.mkdirSync(dir, { recursive: true });
97
- const backend = process.env.VOILA_TTS || 'kokoro';
98
- if (backend === 'kokoro') {
99
- try {
100
- return await synthKokoro(texts, dir, voice, onStatus, speed);
101
- } catch (e) {
102
- if (/unknown voice/.test(e.message)) throw e; // user error, not a fallback case
103
- onStatus(`kokoro unavailable (${e.message.slice(0, 80)})`);
140
+ const norm = items.map(it => (typeof it === 'string' ? { text: it } : it));
141
+ const clips = [];
142
+ const used = new Set();
143
+ let kokoro = null;
144
+
145
+ for (let i = 0; i < norm.length; i++) {
146
+ const it = norm[i];
147
+ const voice = it.voice || defaultVoice || null;
148
+
149
+ // A pre-made audio file wins over any engine.
150
+ if (it.audio) {
151
+ if (!fs.existsSync(it.audio)) throw new Error(`audio file not found: ${it.audio}`);
152
+ clips.push({ file: it.audio, durMs: await ffDurationMs(it.audio) });
153
+ used.add('file');
154
+ continue;
155
+ }
156
+
157
+ const engine = engineFor(voice, ttsCmd);
158
+ const ext = engine === 'kokoro' ? 'wav' : engine === 'say' ? 'aiff' : 'wav';
159
+ const file = path.join(dir, `seg${i}.${ext}`);
160
+
161
+ if (engine === 'kokoro') {
162
+ if (!kokoro) {
163
+ onStatus('loading Kokoro TTS');
164
+ require('./modelcache').useStableCache();
165
+ const { KokoroTTS } = require('kokoro-js');
166
+ kokoro = await KokoroTTS.from_pretrained('onnx-community/Kokoro-82M-v1.0-ONNX', { dtype: 'q8' });
167
+ }
168
+ const v = voice && voices.isValid(voice) ? voice : 'af_heart';
169
+ onStatus(`narrating ${i + 1}/${norm.length} with Kokoro (${v})`);
170
+ const audio = await kokoro.generate(it.text, { voice: v, speed });
171
+ await audio.save(file);
172
+ clips.push({
173
+ file,
174
+ durMs: audio.audio && audio.sampling_rate
175
+ ? Math.round((audio.audio.length / audio.sampling_rate) * 1000)
176
+ : await ffDurationMs(file),
177
+ });
178
+ used.add(`kokoro:${v}`);
179
+ } else if (engine === 'say') {
180
+ if (process.platform !== 'darwin') {
181
+ throw new Error(`system voices need macOS. Use a Kokoro voice for English, or --tts-cmd on this platform.`);
182
+ }
183
+ const name = String(voice).replace(/^say:/, '');
184
+ onStatus(`narrating ${i + 1}/${norm.length} with system voice (${name})`);
185
+ await sayOne(it.text, file, name, speed);
186
+ clips.push({ file, durMs: await ffDurationMs(file) });
187
+ used.add(`say:${name}`);
188
+ } else {
189
+ onStatus(`narrating ${i + 1}/${norm.length} with tts-cmd`);
190
+ await cmdOne(it.text, file, String(voice || '').replace(/^cmd:/, ''), ttsCmd);
191
+ clips.push({ file, durMs: await ffDurationMs(file) });
192
+ used.add('cmd');
104
193
  }
105
194
  }
106
- if (process.platform === 'darwin') return synthSay(texts, dir, voice, onStatus);
107
- throw new Error('no TTS backend available');
195
+
196
+ return { clips, voice: [...used].join(', ') || 'none', backend: [...used].join(', ') };
108
197
  }
109
198
 
110
- async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1, prepared = null, onStatus = () => {} } = {}) {
199
+ async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1, ttsCmd = null, prepared = null, onStatus = () => {} } = {}) {
111
200
  const segs = (meta.segments || []).filter(s => s.narration);
112
201
  if (!segs.length) {
113
202
  fs.copyFileSync(videoIn, videoOut);
@@ -119,7 +208,9 @@ async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1,
119
208
  let synth = prepared && prepared.clips.length === segs.length ? prepared : null;
120
209
  if (!synth) {
121
210
  try {
122
- synth = await prepareNarration(segs.map(s => s.narration), dir, voice, onStatus, speed);
211
+ synth = await prepareNarration(
212
+ segs.map(s => ({ text: s.narration, voice: s.voice, audio: s.audio })),
213
+ dir, voice, onStatus, speed, ttsCmd);
123
214
  } catch (e) {
124
215
  onStatus(`narration skipped: ${e.message}`);
125
216
  fs.copyFileSync(videoIn, videoOut);
@@ -153,7 +244,7 @@ async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1,
153
244
  ff.on('error', rej);
154
245
  });
155
246
 
156
- fs.rmSync(dir, { recursive: true, force: true });
247
+ fs.rmSync(dir, { recursive: true, force: true }); // only generated clips live here
157
248
  return { narrated: true, voice: synth.voice, backend: synth.backend, segments: clips.length };
158
249
  }
159
250
 
package/cli.js CHANGED
@@ -18,11 +18,14 @@ function arg(name, fallback = null) {
18
18
  }
19
19
 
20
20
  const USAGE = `usage:
21
+ voila doctor (check + download everything voila needs)
21
22
  voila outline <url> [--device desktop|mobile|tablet]
22
- voila record <url> [--steps f.yaml] [--device mobile] [--voice name] [--speed 1] [--no-narrate] [--headful] [--out dir] [--profile dir]
23
+ voila record <url> [--steps f.yaml] [--device mobile] [--voice name] [--speed 1] [--tts-cmd tmpl] [--no-narrate] [--headful] [--keep-frames] [--no-dismiss] [--dismiss sel] [--out dir] [--profile dir]
23
24
  voila review <video.mp4> [--frames 12] [--out dir]
24
25
  voila login <url> [--profile dir] (sign in yourself; session is saved locally)
25
- voila voices (list every narration voice, best first)
26
+ voila voices [--all] (Kokoro voices; --all adds system voices for other languages)
27
+ voila fork <video.mp4> [--url u] [--voice v] [--print] [--out dir]
28
+ voila rerender <dir> [--voice v] [--speed n] (needs --keep-frames on the original)
26
29
  voila skill (install the voila skill into ~/.claude/skills)
27
30
  voila serve (web UI, PORT env or --port)
28
31
  voila mcp (stdio MCP server)`;
@@ -39,8 +42,16 @@ const USAGE = `usage:
39
42
  require('./mcp');
40
43
  return;
41
44
  }
45
+ if (cmd === 'doctor' || cmd === '--version' || cmd === '-v') {
46
+ if (cmd !== 'doctor') { console.log(require('./package.json').version); return; }
47
+ console.error(`voila ${require('./package.json').version}\n`);
48
+ const r = await require('./doctor').doctor({ fix: !process.argv.includes('--check') });
49
+ process.exit(r.ready ? 0 : 1);
50
+ }
42
51
  if (cmd === 'voices') {
43
- console.log(require('./voices').format());
52
+ const v = require('./voices');
53
+ console.log(process.argv.includes('--all') ? v.formatAll() : v.format());
54
+ if (!process.argv.includes('--all')) console.log('\nNon-English? run: voila voices --all');
44
55
  return;
45
56
  }
46
57
  if (cmd === 'skill') {
@@ -68,6 +79,61 @@ const USAGE = `usage:
68
79
  return;
69
80
  }
70
81
 
82
+ if (cmd === 'fork') {
83
+ // The recipe inside a voila MP4 is its source. Turn it back into a script
84
+ // and, unless --print, record it again (optionally against another URL).
85
+ const video = process.argv[3];
86
+ if (!video) { console.error(USAGE); process.exit(1); }
87
+ const { reviewDemo } = require('./review');
88
+ const info = await reviewDemo(video, { count: 3 });
89
+ if (!info.recipe || !info.recipe.steps) {
90
+ console.error('FAILED: no voila recipe found in that file (was it made by voila?)');
91
+ process.exit(1);
92
+ }
93
+ const stepsYaml = yaml.dump(info.recipe.steps);
94
+ if (process.argv.includes('--print')) { console.log(stepsYaml); return; }
95
+
96
+ const target = arg('--url', info.recipe.url);
97
+ const workDir = arg('--out', path.join(__dirname, 'recordings', `fork-${Date.now()}`));
98
+ fs.mkdirSync(workDir, { recursive: true });
99
+ fs.writeFileSync(path.join(workDir, 'steps.yaml'), stepsYaml);
100
+ console.error(`[voila] forking ${info.recipe.steps.length} steps onto ${target}`);
101
+
102
+ const { VoilaSession } = require('./recorder');
103
+ const { produceDemo } = require('./pipeline');
104
+ const s2 = new VoilaSession({
105
+ headless: !process.argv.includes('--headful'),
106
+ device: arg('--device', 'desktop'),
107
+ profileDir: arg('--profile', path.join(__dirname, 'profile')),
108
+ });
109
+ try {
110
+ const r = await produceDemo(s2, {
111
+ url: target, steps: yaml.load(stepsYaml), workDir,
112
+ narrate: !process.argv.includes('--no-narrate'),
113
+ voice: arg('--voice'), speed: Number(arg('--speed', '1')) || 1,
114
+ keepFrames: process.argv.includes('--keep-frames'),
115
+ onStatus: m => console.error('[voila]', m),
116
+ });
117
+ console.log(r.video);
118
+ } finally { await s2.close(); }
119
+ return;
120
+ }
121
+
122
+ if (cmd === 'rerender') {
123
+ const dir = process.argv[3];
124
+ if (!dir) { console.error(USAGE); process.exit(1); }
125
+ const { rerender } = require('./pipeline');
126
+ const r = await rerender(dir, {
127
+ voice: arg('--voice'),
128
+ speed: Number(arg('--speed', '1')) || 1,
129
+ ttsCmd: arg('--tts-cmd'),
130
+ narrate: !process.argv.includes('--no-narrate'),
131
+ onStatus: m => console.error('[voila]', m),
132
+ });
133
+ console.log(r.video);
134
+ return;
135
+ }
136
+
71
137
  const url = process.argv[3];
72
138
  if (!cmd || !url || !['outline', 'record', 'login'].includes(cmd)) {
73
139
  console.error(USAGE);
@@ -80,6 +146,8 @@ const USAGE = `usage:
80
146
  headless: !process.argv.includes('--headful'),
81
147
  device: arg('--device', 'desktop'),
82
148
  profileDir: arg('--profile', path.join(__dirname, 'profile')),
149
+ dismiss: !process.argv.includes('--no-dismiss'),
150
+ dismissSelector: arg('--dismiss'),
83
151
  });
84
152
 
85
153
  try {
@@ -101,6 +169,8 @@ const USAGE = `usage:
101
169
  narrate: !process.argv.includes('--no-narrate'),
102
170
  voice: arg('--voice'),
103
171
  speed: Number(arg('--speed', '1')) || 1,
172
+ ttsCmd: arg('--tts-cmd'),
173
+ keepFrames: process.argv.includes('--keep-frames'),
104
174
  onStatus: s => console.error('[voila]', s),
105
175
  });
106
176
  console.log(result.video);
package/consent.js ADDED
@@ -0,0 +1,97 @@
1
+ // Cookie and consent banners ruin a demo: they sit in frame for the whole
2
+ // recording. Dismiss them before the camera rolls.
3
+ //
4
+ // Privacy first: we look for "reject" / "necessary only" before "accept", so
5
+ // the recorded session declines non-essential cookies wherever that choice
6
+ // exists. Accepting is only a last resort to clear the overlay.
7
+
8
+ const REJECT = [
9
+ /^(reject|decline|refuse)( all)?$/i,
10
+ /necessary (cookies )?only/i,
11
+ /^only (essential|necessary|required)/i,
12
+ /^(essential|required) (cookies )?only/i,
13
+ /continue without accepting/i,
14
+ /^reject non-essential/i,
15
+ ];
16
+
17
+ const ACCEPT = [
18
+ /^(accept|allow|agree)( all| cookies)?$/i,
19
+ /^(ok|got it|i understand|understood)$/i,
20
+ /^(dismiss|close)$/i,
21
+ ];
22
+
23
+ // Runs in the page. Returns the label it clicked, or null.
24
+ const dismissInPage = ([rejectSrc, acceptSrc]) => {
25
+ const toRe = arr => arr.map(([s, f]) => new RegExp(s, f));
26
+ const reject = toRe(rejectSrc), accept = toRe(acceptSrc);
27
+
28
+ const visible = el => {
29
+ const r = el.getBoundingClientRect();
30
+ const cs = getComputedStyle(el);
31
+ return r.width > 20 && r.height > 12 && cs.visibility !== 'hidden' && cs.opacity !== '0';
32
+ };
33
+
34
+ // Only consider controls that live inside something banner-shaped, so we
35
+ // never click an "Accept" button that is part of the product itself.
36
+ const looksLikeBanner = el => {
37
+ const box = el.closest('[class*="cookie" i],[id*="cookie" i],[class*="consent" i],[id*="consent" i],[class*="gdpr" i],[id*="gdpr" i],[aria-label*="cookie" i],[role="dialog"],[class*="banner" i]');
38
+ if (box) return true;
39
+ // Or a fixed-position bar pinned to an edge of the viewport.
40
+ let n = el;
41
+ for (let i = 0; i < 6 && n; i++, n = n.parentElement) {
42
+ const cs = getComputedStyle(n);
43
+ if (cs.position === 'fixed' || cs.position === 'sticky') {
44
+ const r = n.getBoundingClientRect();
45
+ if (r.width > innerWidth * 0.5 && (r.bottom > innerHeight * 0.6 || r.top < innerHeight * 0.4)) return true;
46
+ }
47
+ }
48
+ return false;
49
+ };
50
+
51
+ const controls = [...document.querySelectorAll('button,a[role="button"],[role="button"],input[type="button"],input[type="submit"]')]
52
+ .filter(visible).filter(looksLikeBanner);
53
+
54
+ for (const patterns of [reject, accept]) {
55
+ for (const el of controls) {
56
+ const label = (el.innerText || el.value || el.getAttribute('aria-label') || '').trim();
57
+ if (!label || label.length > 40) continue;
58
+ if (patterns.some(re => re.test(label))) { el.click(); return label; }
59
+ }
60
+ }
61
+ return null;
62
+ };
63
+
64
+ async function dismissConsent(page, { selector = null, timeout = 2500 } = {}) {
65
+ const results = [];
66
+ if (selector) {
67
+ const el = page.locator(selector).first();
68
+ try {
69
+ await el.waitFor({ state: 'visible', timeout });
70
+ await el.click({ timeout: 2000 });
71
+ results.push(`custom: ${selector}`);
72
+ } catch { /* nothing matched the override */ }
73
+ }
74
+
75
+ const src = [
76
+ REJECT.map(r => [r.source, r.flags]),
77
+ ACCEPT.map(r => [r.source, r.flags]),
78
+ ];
79
+
80
+ // Banners often mount late, and some sites stack two of them.
81
+ for (let attempt = 0; attempt < 3; attempt++) {
82
+ let clicked = null;
83
+ try {
84
+ clicked = await page.evaluate(dismissInPage, src);
85
+ for (const frame of page.frames()) {
86
+ if (clicked || frame === page.mainFrame()) continue;
87
+ clicked = await frame.evaluate(dismissInPage, src).catch(() => null);
88
+ }
89
+ } catch { /* page navigated mid-check */ }
90
+ if (clicked) results.push(clicked);
91
+ await page.waitForTimeout(attempt === 0 ? 700 : 500);
92
+ if (!clicked && attempt > 0) break;
93
+ }
94
+ return results;
95
+ }
96
+
97
+ module.exports = { dismissConsent };
package/doctor.js ADDED
@@ -0,0 +1,118 @@
1
+ // Pre-flight: tell people what voila needs, what is already on disk, and
2
+ // download the rest with visible progress. The first run used to be several
3
+ // silent minutes, which reads as a hang.
4
+
5
+ const fs = require('fs');
6
+ const path = require('path');
7
+ const os = require('os');
8
+ const { execFile, execFileSync } = require('child_process');
9
+
10
+ const MB = n => `${(n / 1e6).toFixed(0)}MB`;
11
+
12
+ function chromiumPath() {
13
+ try {
14
+ const { chromium } = require('playwright');
15
+ return chromium.executablePath();
16
+ } catch { return null; }
17
+ }
18
+
19
+ function chromiumReady() {
20
+ const p = chromiumPath();
21
+ return !!(p && fs.existsSync(p));
22
+ }
23
+
24
+ // transformers.js caches models under ~/.cache/huggingface by default.
25
+ function kokoroCacheDir() {
26
+ return require('./modelcache').CACHE_DIR;
27
+ }
28
+
29
+ function kokoroReady() {
30
+ const dir = kokoroCacheDir();
31
+ if (!fs.existsSync(dir)) return false;
32
+ const hit = [];
33
+ const walk = (d, depth = 0) => {
34
+ if (depth > 4 || hit.length) return;
35
+ for (const e of fs.readdirSync(d, { withFileTypes: true })) {
36
+ if (hit.length) return;
37
+ const full = path.join(d, e.name);
38
+ if (e.isDirectory()) walk(full, depth + 1);
39
+ else if (/\.onnx(_data)?$/.test(e.name) && fs.statSync(full).size > 5e6) hit.push(full);
40
+ }
41
+ };
42
+ try { walk(dir); } catch { /* unreadable cache */ }
43
+ return hit.length > 0;
44
+ }
45
+
46
+ function ffmpegReady() {
47
+ try { return fs.existsSync(require('ffmpeg-static')); } catch { return false; }
48
+ }
49
+
50
+ // Install Chromium with its progress bar visible instead of swallowed.
51
+ function installChromium({ quiet = false } = {}) {
52
+ let cliPath;
53
+ try { cliPath = require.resolve('playwright/cli'); }
54
+ catch { cliPath = path.join(path.dirname(require.resolve('playwright')), 'cli.js'); }
55
+ execFileSync(process.execPath, [cliPath, 'install', 'chromium'], {
56
+ stdio: quiet ? 'pipe' : ['ignore', 'inherit', 'inherit'],
57
+ timeout: 900000,
58
+ });
59
+ }
60
+
61
+ async function warmKokoro(onStatus = () => {}) {
62
+ require('./modelcache').useStableCache();
63
+ const { KokoroTTS } = require('kokoro-js');
64
+ let lastPct = -5;
65
+ const tts = await KokoroTTS.from_pretrained('onnx-community/Kokoro-82M-v1.0-ONNX', {
66
+ dtype: 'q8',
67
+ progress_callback: p => {
68
+ if (p.status === 'download') return onStatus(`fetching ${p.file}`);
69
+ if (p.status !== 'progress') return;
70
+ // Hugging Face omits content-length on some files, so fall back to bytes.
71
+ if (typeof p.total === 'number' && p.total > 0) {
72
+ const pct = Math.floor((p.progress || 0) / 5) * 5;
73
+ if (pct > lastPct) { lastPct = pct; onStatus(`${p.file || 'model'} ${pct}% of ${MB(p.total)}`); }
74
+ } else if (typeof p.loaded === 'number') {
75
+ const step = Math.floor(p.loaded / 1e7);
76
+ if (step > lastPct) { lastPct = step; onStatus(`${p.file || 'model'} ${MB(p.loaded)} downloaded`); }
77
+ }
78
+ },
79
+ });
80
+ // Force one tiny synthesis so the voice files are fetched too.
81
+ await tts.generate('Ready.', { voice: 'af_heart' });
82
+ return true;
83
+ }
84
+
85
+ async function doctor({ fix = true, log = console.error } = {}) {
86
+ const nodeOk = Number(process.versions.node.split('.')[0]) >= 20;
87
+ log(`node ${process.versions.node} ${nodeOk ? 'ok' : 'TOO OLD, voila needs >= 20'}`);
88
+ log(`ffmpeg ${ffmpegReady() ? 'bundled, ok' : 'MISSING (reinstall voila-recorder)'}`);
89
+
90
+ let chrome = chromiumReady();
91
+ log(`chromium ${chrome ? 'installed' : 'not installed (~150MB download)'}`);
92
+ if (!chrome && fix) {
93
+ log('\ndownloading chromium...');
94
+ installChromium();
95
+ chrome = chromiumReady();
96
+ log(`chromium ${chrome ? 'installed' : 'FAILED'}`);
97
+ }
98
+
99
+ let voice = kokoroReady();
100
+ log(`voice model ${voice ? `cached in ${kokoroCacheDir()}` : 'not cached (~90MB download, first narration only)'}`);
101
+ if (!voice && fix) {
102
+ log('\nfetching the voice model...');
103
+ try {
104
+ await warmKokoro(m => log(` ${m}`));
105
+ voice = true;
106
+ log('voice model cached');
107
+ } catch (e) {
108
+ log(`voice model FAILED: ${e.message.slice(0, 120)}`);
109
+ log('(recording still works with --no-narrate)');
110
+ }
111
+ }
112
+
113
+ const ready = nodeOk && ffmpegReady() && chrome;
114
+ log(`\n${ready ? 'voila is ready. Try: voila record https://example.com' : 'voila is not ready yet, see above.'}`);
115
+ return { nodeOk, ffmpeg: ffmpegReady(), chromium: chrome, voice, ready };
116
+ }
117
+
118
+ module.exports = { doctor, chromiumReady, kokoroReady, installChromium, warmKokoro };
package/mcp.js CHANGED
@@ -37,7 +37,7 @@ function getSession(device) {
37
37
  return sessions.get(key);
38
38
  }
39
39
 
40
- const server = new McpServer({ name: 'voila', version: '0.5.0' });
40
+ const server = new McpServer({ name: 'voila', version: '0.7.0' });
41
41
  const deviceParam = z.enum(['desktop', 'mobile', 'tablet']).optional().default('desktop');
42
42
 
43
43
  server.tool(
@@ -61,6 +61,7 @@ server.tool(
61
61
  'caption is burned into the video as a lower-third; narration is spoken via on-device TTS (Kokoro) at that step, ' +
62
62
  'and segment pacing automatically stretches to fit each narration clip — no need to pad waits. ' +
63
63
  'Steps marked optional:true are skipped on failure instead of aborting. ' +
64
+ 'A step may set its own voice: (mixing languages within one demo) or audio: (a ready-made clip). ' +
64
65
  'device selects the recorded viewport (mobile emulates an iPhone-class device). ' +
65
66
  'On failure the error names the failing step and includes the live page outline — patch the steps and retry. ' +
66
67
  'Returns the MP4 path, the recipe path, and any warnings.',
@@ -68,16 +69,17 @@ server.tool(
68
69
  url: z.string().url(),
69
70
  steps_yaml: z.string().optional(),
70
71
  narrate: z.boolean().optional().default(true),
71
- voice: z.string().optional().describe('narration voice id, e.g. af_heart (A), af_bella (A-), bf_emma (B-, British). Call voila_voices for the full list.'),
72
+ voice: z.string().optional().describe('default narration voice. Kokoro is ENGLISH ONLY (af_heart, af_bella, bf_emma). For other languages use a macOS system voice ("say:Monica") or set tts_cmd. Per-step `voice:` overrides this, so one demo can mix languages.'),
72
73
  speed: z.number().min(0.5).max(1.6).optional().default(1).describe('narration speed; 0.9 reads calmer'),
74
+ tts_cmd: z.string().optional().describe('external TTS engine template for any language/platform, e.g. \'piper -m es.onnx -f {out} -- "{text}"\'. Placeholders: {out} {text} {voice}.'),
73
75
  device: deviceParam,
74
76
  },
75
- async ({ url, steps_yaml, narrate, voice, speed, device }) => enqueue(async () => {
77
+ async ({ url, steps_yaml, narrate, voice, speed, tts_cmd, device }) => enqueue(async () => {
76
78
  const steps = steps_yaml ? yaml.load(steps_yaml) : null;
77
79
  const workDir = path.join(__dirname, 'recordings', `mcp-${Date.now()}`);
78
80
  fs.mkdirSync(workDir, { recursive: true });
79
81
  const result = await produceDemo(getSession(device), {
80
- url, steps, workDir, narrate, voice: voice || null, speed,
82
+ url, steps, workDir, narrate, voice: voice || null, speed, ttsCmd: tts_cmd || null,
81
83
  onStatus: () => {},
82
84
  });
83
85
  return {
@@ -116,11 +118,19 @@ server.tool(
116
118
 
117
119
  server.tool(
118
120
  'voila_voices',
119
- 'List every narration voice with its quality grade, best first. Use before voila_record when the ' +
120
- 'user asks for a different voice, an accent, or a male or female narrator.',
121
- {},
122
- async () => ({
123
- content: [{ type: 'text', text: JSON.stringify(voiceCatalogue.ranked(), null, 2) }],
121
+ 'List narration voices. Kokoro voices are English only, with quality grades. Pass system:true to ' +
122
+ 'also get the machine\'s system voices, which is how you narrate other languages (macOS only; on ' +
123
+ 'Linux or Windows use tts_cmd instead). Use before voila_record when the user asks for a different ' +
124
+ 'voice, an accent, a male or female narrator, or a non-English language.',
125
+ { system: z.boolean().optional().default(false) },
126
+ async ({ system }) => ({
127
+ content: [{ type: 'text', text: JSON.stringify({
128
+ kokoro: voiceCatalogue.ranked(),
129
+ englishOnly: true,
130
+ system: system ? voiceCatalogue.systemVoices() : undefined,
131
+ systemLanguages: system ? Object.keys(voiceCatalogue.systemLanguages()) : undefined,
132
+ note: 'Non-English: use a system voice (say:Name) on macOS, or tts_cmd on any platform.',
133
+ }, null, 2) }],
124
134
  })
125
135
  );
126
136
 
package/modelcache.js ADDED
@@ -0,0 +1,22 @@
1
+ // transformers.js caches models inside its own node_modules folder by default,
2
+ // so every voila upgrade (and every fresh npx hash) would re-download ~90MB.
3
+ // Pin the cache to a stable per-user directory instead.
4
+
5
+ const os = require('os');
6
+ const path = require('path');
7
+
8
+ const CACHE_DIR = process.env.VOILA_MODEL_DIR
9
+ || path.join(os.homedir(), '.cache', 'voila', 'models');
10
+
11
+ let applied = false;
12
+ function useStableCache() {
13
+ if (applied) return CACHE_DIR;
14
+ try {
15
+ const { env } = require('@huggingface/transformers');
16
+ env.cacheDir = CACHE_DIR;
17
+ applied = true;
18
+ } catch { /* transformers not resolvable; kokoro will use its default */ }
19
+ return CACHE_DIR;
20
+ }
21
+
22
+ module.exports = { useStableCache, CACHE_DIR };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "voila-recorder",
3
- "version": "0.5.0",
3
+ "version": "0.7.0",
4
4
  "description": "Permission-free product demo recorder: URL in, narrated auto-zoomed MP4 out — with the recipe embedded in the video. Agent-native (MCP), fully on-device.",
5
5
  "license": "MIT",
6
6
  "main": "pipeline.js",
@@ -18,13 +18,17 @@
18
18
  "render.js",
19
19
  "audio.js",
20
20
  "review.js",
21
+ "modelcache.js",
22
+ "consent.js",
23
+ "doctor.js",
21
24
  "voices.js",
22
25
  "auth.js",
23
26
  "public/",
24
27
  "skills/",
25
28
  "AGENTS.md",
26
29
  "README.md",
27
- "RECIPE.md"
30
+ "RECIPE.md",
31
+ "scripts/"
28
32
  ],
29
33
  "repository": {
30
34
  "type": "git",
package/pipeline.js CHANGED
@@ -44,15 +44,16 @@ function embedRecipe(videoIn, videoOut, recipe) {
44
44
  });
45
45
  }
46
46
 
47
- async function produceDemo(session, { url, mode = 'auto', steps = null, workDir, voice = null, speed = 1, narrate = true, onStatus = () => {} }) {
47
+ async function produceDemo(session, { url, mode = 'auto', steps = null, workDir, voice = null, speed = 1, ttsCmd = null, narrate = true, keepFrames = false, onStatus = () => {} }) {
48
48
  // Steps mode: synthesize narration BEFORE recording so segment pacing and
49
49
  // caption lifetimes match the spoken clip durations exactly.
50
50
  let prepared = null;
51
51
  if (steps && narrate) {
52
- const texts = steps.filter(s => s.narration).map(s => s.narration);
53
- if (texts.length) {
52
+ const items = steps.filter(s => s.narration)
53
+ .map(s => ({ text: s.narration, voice: s.voice, audio: s.audio }));
54
+ if (items.length) {
54
55
  try {
55
- prepared = await prepareNarration(texts, path.join(workDir, 'tts'), voice, onStatus, speed);
56
+ prepared = await prepareNarration(items, path.join(workDir, 'tts'), voice, onStatus, speed, ttsCmd);
56
57
  let i = 0;
57
58
  for (const s of steps) if (s.narration) s._narrDurMs = prepared.clips[i++].durMs;
58
59
  } catch (e) {
@@ -69,7 +70,7 @@ async function produceDemo(session, { url, mode = 'auto', steps = null, workDir,
69
70
  await render(meta, raw, { onStatus });
70
71
 
71
72
  let narration = { narrated: false };
72
- if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, prepared, onStatus });
73
+ if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, ttsCmd, prepared, onStatus });
73
74
  else fs.copyFileSync(raw, narrated);
74
75
 
75
76
  const recipe = buildRecipe({ url, mode, steps, meta });
@@ -77,10 +78,40 @@ async function produceDemo(session, { url, mode = 'auto', steps = null, workDir,
77
78
  onStatus('embedding recipe');
78
79
  await embedRecipe(narrated, out, recipe);
79
80
 
80
- fs.rmSync(meta.framesDir, { recursive: true, force: true });
81
+ // Frames are the expensive part to recreate: keeping them lets `voila
82
+ // rerender` change the voice, speed or captions in seconds instead of
83
+ // re-driving the browser. They are large, so it is opt-in.
84
+ if (!keepFrames) fs.rmSync(meta.framesDir, { recursive: true, force: true });
81
85
  fs.rmSync(raw, { force: true });
82
86
  fs.rmSync(narrated, { force: true });
83
- return { video: out, recipe: path.join(workDir, 'recipe.json'), meta, narration };
87
+ return { video: out, recipe: path.join(workDir, 'recipe.json'), meta, narration, framesKept: keepFrames };
88
+ }
89
+
90
+ // Re-produce the video from frames already on disk: no browser, no re-driving
91
+ // the page. Used to swap the narration voice or speed after the fact.
92
+ async function rerender(workDir, { voice = null, speed = 1, ttsCmd = null, narrate = true, onStatus = () => {} } = {}) {
93
+ const metaPath = path.join(workDir, 'meta.json');
94
+ if (!fs.existsSync(metaPath)) throw new Error(`no meta.json in ${workDir}`);
95
+ const meta = JSON.parse(fs.readFileSync(metaPath, 'utf8'));
96
+ if (!fs.existsSync(meta.framesDir) || !fs.readdirSync(meta.framesDir).length) {
97
+ throw new Error(`frames were not kept for this recording. Re-record with --keep-frames to enable rerender.`);
98
+ }
99
+
100
+ const raw = path.join(workDir, 'raw.mp4');
101
+ const narrated = path.join(workDir, 'narrated.mp4');
102
+ const out = path.join(workDir, 'demo.mp4');
103
+ await render(meta, raw, { onStatus });
104
+
105
+ let narration = { narrated: false };
106
+ if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, ttsCmd, onStatus });
107
+ else fs.copyFileSync(raw, narrated);
108
+
109
+ const recipe = JSON.parse(fs.readFileSync(path.join(workDir, 'recipe.json'), 'utf8'));
110
+ onStatus('embedding recipe');
111
+ await embedRecipe(narrated, out, recipe);
112
+ fs.rmSync(raw, { force: true });
113
+ fs.rmSync(narrated, { force: true });
114
+ return { video: out, meta, narration };
84
115
  }
85
116
 
86
117
  async function outline(session, url) {
@@ -88,4 +119,4 @@ async function outline(session, url) {
88
119
  return extractOutline(page);
89
120
  }
90
121
 
91
- module.exports = { produceDemo, outline };
122
+ module.exports = { produceDemo, rerender, outline };
package/recorder.js CHANGED
@@ -40,8 +40,11 @@ class Timeline {
40
40
  this.zoom = 1;
41
41
  this.center = null;
42
42
  }
43
- recordSegment(caption, narration, dur = null) {
44
- this.segments.push({ t: Date.now(), caption: caption || null, narration: narration || null, dur });
43
+ recordSegment(caption, narration, dur = null, voice = null, audio = null) {
44
+ this.segments.push({
45
+ t: Date.now(), caption: caption || null, narration: narration || null, dur,
46
+ ...(voice ? { voice } : {}), ...(audio ? { audio } : {}),
47
+ });
45
48
  }
46
49
  recordMove(to, dur) {
47
50
  this.moves.push({ t: Date.now(), from: { ...this.pos }, to: { ...to }, dur });
@@ -59,7 +62,9 @@ class Timeline {
59
62
  }
60
63
 
61
64
  class VoilaSession {
62
- constructor({ profileDir, headless = false, device = 'desktop' } = {}) {
65
+ constructor({ profileDir, headless = false, device = 'desktop', dismiss = true, dismissSelector = null } = {}) {
66
+ this.dismiss = dismiss;
67
+ this.dismissSelector = dismissSelector;
63
68
  this.profileDir = profileDir || path.join(__dirname, 'profile');
64
69
  this.headless = headless;
65
70
  this.device = DEVICES[device] ? device : 'desktop';
@@ -85,11 +90,9 @@ class VoilaSession {
85
90
  // Zero-install path: fetch Chromium on first use instead of making the
86
91
  // user run `npx playwright install` themselves.
87
92
  if (!/Executable doesn't exist|missing dependencies|browser.*not found/i.test(String(e.message))) throw e;
88
- const { execFileSync } = require('child_process');
89
- let cliPath;
90
- try { cliPath = require.resolve('playwright/cli'); }
91
- catch { cliPath = path.join(path.dirname(require.resolve('playwright/package.json')), 'cli.js'); }
92
- execFileSync(process.execPath, [cliPath, 'install', 'chromium'], { stdio: 'pipe', timeout: 600000 });
93
+ // Visible progress: a silent 150MB download reads as a hang.
94
+ process.stderr.write('[voila] first run: downloading Chromium (~150MB, one time)\n');
95
+ require('./doctor').installChromium();
93
96
  this.context = await launch();
94
97
  }
95
98
  await this.context.addInitScript(OVERLAY_SOURCE);
@@ -99,6 +102,10 @@ class VoilaSession {
99
102
  this.page.on('close', () => { this.page = null; });
100
103
  await this.page.goto(url, { waitUntil: 'domcontentloaded', timeout: 45000 });
101
104
  await this.page.waitForLoadState('networkidle', { timeout: 8000 }).catch(() => {});
105
+ if (this.dismiss) {
106
+ const { dismissConsent } = require('./consent');
107
+ this.dismissed = await dismissConsent(this.page, { selector: this.dismissSelector });
108
+ }
102
109
  return this.page;
103
110
  }
104
111
 
@@ -173,6 +180,7 @@ class VoilaSession {
173
180
  frames: frames.sort((a, b) => a.t - b.t),
174
181
  moves: tl.moves, zooms: tl.zooms, segments: tl.segments,
175
182
  warnings: tl.warnings || [],
183
+ dismissed: this.dismissed || [],
176
184
  framesDir,
177
185
  };
178
186
  fs.writeFileSync(path.join(workDir, 'meta.json'), JSON.stringify(meta));
@@ -0,0 +1,14 @@
1
+ // CI helper: assert the rendered demo actually carries a narration track.
2
+ const { execFileSync } = require('child_process');
3
+ const ffmpeg = require('ffmpeg-static');
4
+
5
+ const file = process.argv[2];
6
+ let out = '';
7
+ try { execFileSync(ffmpeg, ['-i', file], { stdio: 'pipe' }); }
8
+ catch (e) { out = String(e.stderr || ''); }
9
+
10
+ if (!/Audio: aac/.test(out)) {
11
+ console.error(out || '(no ffmpeg output)');
12
+ throw new Error(`no narration track in ${file}`);
13
+ }
14
+ console.log(`narration track OK in ${file}`);
@@ -0,0 +1,10 @@
1
+ // CI fixture: a stand-in "external TTS engine". Writes a short tone to {out}
2
+ // so the --tts-cmd plumbing can be exercised on every platform, including
3
+ // ones with no system voices installed.
4
+ const { execFileSync } = require('child_process');
5
+ const ffmpeg = require('ffmpeg-static');
6
+ const out = process.argv[2];
7
+ const text = process.argv.slice(3).join(' ');
8
+ const seconds = Math.max(1, Math.min(8, text.split(/\s+/).length / 3)).toFixed(2);
9
+ execFileSync(ffmpeg, ['-y', '-f', 'lavfi', '-i', `sine=frequency=340:duration=${seconds}`, '-ac', '1', out], { stdio: 'pipe' });
10
+ console.log(`fixture wrote ${seconds}s to ${out}`);
@@ -14,9 +14,12 @@ Prefer the MCP tools if registered (`voila_outline`, `voila_record`,
14
14
  `voila_review`); otherwise use the CLI via npx:
15
15
 
16
16
  ```bash
17
+ npx -y voila-recorder doctor # first run: pre-downloads chromium + voice model
17
18
  npx -y voila-recorder outline <url>
18
19
  npx -y voila-recorder record <url> --steps steps.yaml [--device mobile]
19
20
  npx -y voila-recorder review demo.mp4
21
+ npx -y voila-recorder fork demo.mp4 --url https://other.example
22
+ npx -y voila-recorder rerender <dir> --voice bf_emma
20
23
  ```
21
24
 
22
25
  Register the MCP server once with: `claude mcp add voila -- npx -y voila-recorder mcp`
@@ -61,9 +64,23 @@ Reference: https://voila.anzalabidi.dev/llms.txt · https://github.com/anzal1/vo
61
64
  for them in chat.
62
65
  - Prefer `zoom` with a `selector` over a bare `level`: voila measures the
63
66
  element and picks the level and camera centre, so nothing is cropped.
64
- - Voices: `voila voices` lists 28 English voices with quality grades. af_heart
67
+ - Languages: narration voice is per step, so demos can mix languages. Kokoro
68
+ is ENGLISH ONLY (af_heart A, af_bella A-, bf_emma British). For other
69
+ languages use a macOS system voice (`voice: "say:Monica"`, `voila voices
70
+ --all` lists ~180 across ~50 languages), or `--tts-cmd` with any engine on
71
+ any platform (placeholders {out} {text} {voice}), or point a step at a
72
+ ready-made clip with `audio: file.mp3`. On Linux/Windows non-English needs
73
+ --tts-cmd: tell the user rather than silently narrating in English.
74
+ - Voices: `voila voices` lists the English voices with quality grades. af_heart
65
75
  (A) default, af_bella (A-), af_nicole (B-), bf_emma (B-, British). `--speed`
66
76
  or the speed param (0.5-1.6) changes pace; 0.9 reads calmer.
67
77
  - A failing step is retried once automatically; warnings appear in the result.
78
+ - Iterating on narration? Record once with `--keep-frames`, then `rerender <dir>
79
+ --voice x --speed n`. It skips the browser entirely and finishes in seconds.
80
+ - Consent banners are auto-dismissed (reject preferred over accept). Use
81
+ `--dismiss <selector>` for an unusual one, `--no-dismiss` to leave it alone.
82
+ - First run on a new machine downloads ~240MB. Run `voila doctor` first and tell
83
+ the user it is downloading, so it does not look frozen.
84
+ - `fork <video.mp4>` rebuilds any voila demo from the recipe inside the file.
68
85
  - Narration style: short sentences, product language, no "as you can see".
69
86
  8–15 words per beat reads best at Kokoro's pace.
package/tour.js CHANGED
@@ -242,7 +242,7 @@ async function runSteps(page, tl, steps, opts) {
242
242
  const step = steps[si];
243
243
  if (step.caption || step.narration) {
244
244
  await finishSegment();
245
- tl.recordSegment(step.caption, step.narration, step._narrDurMs || null);
245
+ tl.recordSegment(step.caption, step.narration, step._narrDurMs || null, step.voice || null, step.audio || null);
246
246
  segStart = Date.now();
247
247
  segMinMs = (step._narrDurMs || 0) + 600;
248
248
  }
package/voices.js CHANGED
@@ -55,4 +55,52 @@ function format() {
55
55
  return lines.join('\n');
56
56
  }
57
57
 
58
- module.exports = { allVoices, ranked, isValid, suggest, format };
58
+ // --- system voices (macOS `say`) ---------------------------------------------
59
+ // Kokoro is English-only in JS (its other voice files ship without a
60
+ // grapheme-to-phoneme stage for those languages). Every Mac already carries
61
+ // ~180 voices across ~50 languages, so those cover non-English narration.
62
+
63
+ let sysCache = null;
64
+ function systemVoices() {
65
+ if (sysCache) return sysCache;
66
+ sysCache = [];
67
+ if (process.platform !== 'darwin') return sysCache;
68
+ try {
69
+ const { execFileSync } = require('child_process');
70
+ const out = String(execFileSync('say', ['-v', '?'], { maxBuffer: 4e6 }));
71
+ for (const line of out.split('\n')) {
72
+ const m = /^(.+?)\s{2,}([a-z]{2}_[A-Z]{2})\s/.exec(line);
73
+ if (m) sysCache.push({ id: `say:${m[1].trim()}`, name: m[1].trim(), language: m[2].replace('_', '-'), gender: '', grade: 'system' });
74
+ }
75
+ } catch { /* no say binary */ }
76
+ return sysCache;
77
+ }
78
+
79
+ function systemLanguages() {
80
+ const langs = {};
81
+ for (const v of systemVoices()) (langs[v.language] = langs[v.language] || []).push(v.name);
82
+ return langs;
83
+ }
84
+
85
+ function isSystemVoice(id) {
86
+ if (!id) return false;
87
+ const name = String(id).replace(/^say:/, '').toLowerCase();
88
+ return systemVoices().some(v => v.name.toLowerCase() === name);
89
+ }
90
+
91
+ function formatAll() {
92
+ const lines = [format()];
93
+ const langs = systemLanguages();
94
+ const codes = Object.keys(langs).sort();
95
+ if (!codes.length) {
96
+ lines.push('\nSystem voices: none found (macOS only). For other languages use --tts-cmd.');
97
+ return lines.join('\n');
98
+ }
99
+ lines.push(`\nSystem voices (macOS, ${systemVoices().length} across ${codes.length} languages)`);
100
+ for (const c of codes) lines.push(` ${c.padEnd(7)} ${langs[c].slice(0, 6).join(', ')}${langs[c].length > 6 ? ` +${langs[c].length - 6}` : ''}`);
101
+ lines.push('\nUse a system voice for non-English narration: --voice "say:Monica"');
102
+ lines.push('Any other engine: --tts-cmd \'piper --model es.onnx -f {out} -- "{text}"\'');
103
+ return lines.join('\n');
104
+ }
105
+
106
+ module.exports = { allVoices, ranked, isValid, suggest, format, formatAll, systemVoices, systemLanguages, isSystemVoice };