voila-recorder 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/AGENTS.md CHANGED
@@ -58,12 +58,12 @@ npx -y voila-recorder rerender <dir> --voice bf_emma
58
58
  - Narration style: short sentences, product language, 8-15 words per beat.
59
59
  - Prefer `zoom` with a `selector` over a raw `level`: voila measures the element
60
60
  and picks the level and camera centre so nothing gets cropped.
61
- - Voices: Kokoro is English only (af_heart A, af_bella A-, bf_emma British).
62
- `speed` 0.5-1.6 sets pace.
63
- - Other languages: set `voice:` per narration step, so one demo can mix them.
64
- `say:Monica` uses a macOS system voice (~50 languages, `voila voices --all`);
65
- `--tts-cmd 'engine -o {out} "{text}"'` plugs in any engine on any platform;
66
- `audio: clip.mp3` on a step uses a file you already have.
61
+ - Voices: Kokoro covers English, Spanish, French, Italian, Portuguese and
62
+ Hindi on every platform (af_heart, ef_dora, ff_siwis, if_sara, pf_dora,
63
+ hf_alpha). `speed` 0.5-1.6 sets pace. `voila voices` lists all 41.
64
+ - Mixing languages: set `voice:` per narration step.
65
+ - Japanese/Mandarin are gated (espeak mispronounces them). Use `--tts-cmd`
66
+ with a dedicated engine, a macOS `say:` voice, or `audio: clip.mp3`.
67
67
  - Sign-in walls: recording refuses to film a login page. Run `voila login <url>`
68
68
  (or the voila_login tool), let the HUMAN sign in in the window that opens, and
69
69
  the session persists in a local profile for every later recording.
package/README.md CHANGED
@@ -73,27 +73,27 @@ on-device, no cloud, no API keys ([audio.js](audio.js)). Falls back to macOS
73
73
  `say` if Kokoro can't load. Voices: `af_heart` (default), `af_bella`,
74
74
  `am_adam`, … (`voice` param). Disable with `narrate: false` / `--no-narrate`.
75
75
 
76
- **Other languages, and mixing them.** Kokoro's JS port speaks English only, so
77
- non-English narration comes from one of two other engines, chosen per step:
76
+ **Six languages, mixable in one demo.** Kokoro speaks English (US/UK),
77
+ Spanish, French, Italian, Portuguese and Hindi on every platform, on-device.
78
+ Set `voice:` per step:
78
79
 
79
80
  ```yaml
80
81
  - action: hover
81
82
  selector: h1
82
- narration: "This part is English." # Kokoro
83
+ narration: "This part is English." # af_heart
83
84
  - action: scroll_to
84
85
  selector: "#pricing"
85
- voice: "say:Mónica" # macOS system voice
86
+ voice: ef_dora # Spanish, same model
86
87
  narration: "Esta parte está en español."
87
- - action: wait
88
- ms: 500
89
- audio: ./clips/intro-ja.mp3 # a clip you already have
90
88
  ```
91
89
 
92
- `voila voices --all` lists the ~180 system voices across ~50 languages on
93
- macOS. On Linux and Windows use any engine you like:
94
- `--tts-cmd 'piper -m es.onnx -f {out} -- "{text}"'` ({out}, {text}, {voice}).
95
- Every step is still paced to its own spoken clip, so mixed-language demos stay
96
- in sync.
90
+ `voila voices` lists all 41. Every step is paced to its own spoken clip, so
91
+ mixed-language demos stay in sync. Japanese and Mandarin voices ship with the
92
+ model but are gated: espeak mispronounces them badly. For those, and for
93
+ anything else, plug in your own engine with
94
+ `--tts-cmd 'piper -m ja.onnx -f {out} -- "{text}"'`, use a macOS system voice
95
+ (`voice: "say:Kyoko"`, `voila voices --all`), or hand a step a ready-made clip
96
+ with `audio: intro.mp3`.
97
97
 
98
98
  ## Recipes — demos as code
99
99
 
package/audio.js CHANGED
@@ -166,8 +166,27 @@ async function prepareNarration(items, dir, defaultVoice, onStatus = () => {}, s
166
166
  kokoro = await KokoroTTS.from_pretrained('onnx-community/Kokoro-82M-v1.0-ONNX', { dtype: 'q8' });
167
167
  }
168
168
  const v = voice && voices.isValid(voice) ? voice : 'af_heart';
169
- onStatus(`narrating ${i + 1}/${norm.length} with Kokoro (${v})`);
170
- const audio = await kokoro.generate(it.text, { voice: v, speed });
169
+ const { langOf, phonemize } = require('./phonemes');
170
+ const lang = langOf(v);
171
+ onStatus(`narrating ${i + 1}/${norm.length} with Kokoro (${v}, ${lang ? lang.name : 'English'})`);
172
+
173
+ let audio;
174
+ if (lang && lang.tier !== 'native') {
175
+ // kokoro-js only phonemizes English, so do it ourselves with espeak-ng
176
+ // (WASM, every platform) and feed the model token ids directly.
177
+ if (lang.tier === 'experimental' && !process.env.VOILA_EXPERIMENTAL_LANGS) {
178
+ throw new Error(
179
+ `${lang.name} voices are experimental: espeak mispronounces them badly ` +
180
+ `(Japanese leaks English words, Mandarin emits numeric tones Kokoro never saw). ` +
181
+ `Set VOILA_EXPERIMENTAL_LANGS=1 to try anyway, or use --tts-cmd with a ${lang.name} engine.`
182
+ );
183
+ }
184
+ const ipa = await phonemize(it.text, v);
185
+ const enc = kokoro.tokenizer(ipa, { truncation: true });
186
+ audio = await kokoro.generate_from_ids(enc.input_ids, { voice: v, speed });
187
+ } else {
188
+ audio = await kokoro.generate(it.text, { voice: v, speed });
189
+ }
171
190
  await audio.save(file);
172
191
  clips.push({
173
192
  file,
package/mcp.js CHANGED
@@ -37,7 +37,7 @@ function getSession(device) {
37
37
  return sessions.get(key);
38
38
  }
39
39
 
40
- const server = new McpServer({ name: 'voila', version: '0.7.0' });
40
+ const server = new McpServer({ name: 'voila', version: '0.8.0' });
41
41
  const deviceParam = z.enum(['desktop', 'mobile', 'tablet']).optional().default('desktop');
42
42
 
43
43
  server.tool(
@@ -69,7 +69,7 @@ server.tool(
69
69
  url: z.string().url(),
70
70
  steps_yaml: z.string().optional(),
71
71
  narrate: z.boolean().optional().default(true),
72
- voice: z.string().optional().describe('default narration voice. Kokoro is ENGLISH ONLY (af_heart, af_bella, bf_emma). For other languages use a macOS system voice ("say:Monica") or set tts_cmd. Per-step `voice:` overrides this, so one demo can mix languages.'),
72
+ voice: z.string().optional().describe('default narration voice. Kokoro speaks English (af_heart, af_bella, bf_emma), Spanish (ef_dora), French (ff_siwis), Italian (if_sara), Portuguese (pf_dora) and Hindi (hf_alpha) on every platform. Per-step `voice:` overrides this, so one demo can mix languages. Japanese/Mandarin are gated (mispronounced) - use tts_cmd for those.'),
73
73
  speed: z.number().min(0.5).max(1.6).optional().default(1).describe('narration speed; 0.9 reads calmer'),
74
74
  tts_cmd: z.string().optional().describe('external TTS engine template for any language/platform, e.g. \'piper -m es.onnx -f {out} -- "{text}"\'. Placeholders: {out} {text} {voice}.'),
75
75
  device: deviceParam,
@@ -118,18 +118,20 @@ server.tool(
118
118
 
119
119
  server.tool(
120
120
  'voila_voices',
121
- 'List narration voices. Kokoro voices are English only, with quality grades. Pass system:true to ' +
122
- 'also get the machine\'s system voices, which is how you narrate other languages (macOS only; on ' +
123
- 'Linux or Windows use tts_cmd instead). Use before voila_record when the user asks for a different ' +
124
- 'voice, an accent, a male or female narrator, or a non-English language.',
121
+ 'List narration voices. Kokoro covers English, Spanish, French, Italian, Portuguese and Hindi on ' +
122
+ 'every platform, on-device; English voices carry quality grades. Japanese and Mandarin voices exist ' +
123
+ 'but are gated because espeak mispronounces them. Pass system:true to also list the machine\'s own ' +
124
+ 'voices (macOS). Use before voila_record when the user asks for a different voice, an accent, a male ' +
125
+ 'or female narrator, or a non-English language.',
125
126
  { system: z.boolean().optional().default(false) },
126
127
  async ({ system }) => ({
127
128
  content: [{ type: 'text', text: JSON.stringify({
128
129
  kokoro: voiceCatalogue.ranked(),
129
- englishOnly: true,
130
+ languages: 'English (US/UK), Spanish, French, Italian, Portuguese (BR), Hindi',
131
+ gated: 'Japanese and Mandarin: espeak mispronounces them; use tts_cmd or a system voice',
130
132
  system: system ? voiceCatalogue.systemVoices() : undefined,
131
133
  systemLanguages: system ? Object.keys(voiceCatalogue.systemLanguages()) : undefined,
132
- note: 'Non-English: use a system voice (say:Name) on macOS, or tts_cmd on any platform.',
134
+ note: 'Set voice: per step to mix languages in one demo.',
133
135
  }, null, 2) }],
134
136
  })
135
137
  );
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "voila-recorder",
3
- "version": "0.7.0",
3
+ "version": "0.8.0",
4
4
  "description": "Permission-free product demo recorder: URL in, narrated auto-zoomed MP4 out — with the recipe embedded in the video. Agent-native (MCP), fully on-device.",
5
5
  "license": "MIT",
6
6
  "main": "pipeline.js",
@@ -21,6 +21,7 @@
21
21
  "modelcache.js",
22
22
  "consent.js",
23
23
  "doctor.js",
24
+ "phonemes.js",
24
25
  "voices.js",
25
26
  "auth.js",
26
27
  "public/",
@@ -54,6 +55,7 @@
54
55
  "demos-as-code"
55
56
  ],
56
57
  "dependencies": {
58
+ "@echogarden/espeak-ng-emscripten": "^0.3.5",
57
59
  "@modelcontextprotocol/sdk": "^1.30.0",
58
60
  "express": "^4.19.2",
59
61
  "ffmpeg-static": "^5.2.0",
package/phonemes.js ADDED
@@ -0,0 +1,51 @@
1
+ // Grapheme-to-phoneme for Kokoro's non-English voices.
2
+ //
3
+ // kokoro-js only ships an English phonemizer, which is why its 26 non-English
4
+ // voice files sat unusable. espeak-ng compiled to WASM carries the full
5
+ // language data, runs on every platform, and emits the IPA Kokoro was trained
6
+ // on. That combination gives real multilingual narration with no OS-specific
7
+ // dependency.
8
+
9
+ // Kokoro voice ids are prefixed by language: af_/am_ = American English,
10
+ // bf_/bm_ = British, ef_/em_ = Spanish, and so on.
11
+ const LANGS = {
12
+ a: { espeak: 'en-us', name: 'English (US)', tier: 'native' },
13
+ b: { espeak: 'en-gb', name: 'English (UK)', tier: 'native' },
14
+ e: { espeak: 'es', name: 'Spanish', tier: 'good' },
15
+ f: { espeak: 'fr-fr', name: 'French', tier: 'good' },
16
+ h: { espeak: 'hi', name: 'Hindi', tier: 'good' },
17
+ i: { espeak: 'it', name: 'Italian', tier: 'good' },
18
+ p: { espeak: 'pt-br', name: 'Portuguese (BR)', tier: 'good' },
19
+ // espeak leaks English words into Japanese kanji, and emits numeric tones
20
+ // for Mandarin that Kokoro was not trained on. Both need a dedicated G2P
21
+ // (the Python release uses one); until then they are off by default.
22
+ j: { espeak: 'ja', name: 'Japanese', tier: 'experimental' },
23
+ z: { espeak: 'cmn', name: 'Mandarin', tier: 'experimental' },
24
+ };
25
+
26
+ const langOf = voiceId => LANGS[String(voiceId || '')[0]] || null;
27
+
28
+ let worker = null;
29
+ async function getWorker() {
30
+ if (!worker) {
31
+ const mod = await require('@echogarden/espeak-ng-emscripten').default();
32
+ worker = await new mod.eSpeakNGWorker();
33
+ }
34
+ return worker;
35
+ }
36
+
37
+ // espeak separates phonemes with underscores and sentences with newlines.
38
+ // Kokoro wants a plain IPA string.
39
+ function tidy(ipa) {
40
+ return ipa.replace(/_/g, '').replace(/\s*\n\s*/g, ' ').replace(/\s+/g, ' ').trim();
41
+ }
42
+
43
+ async function phonemize(text, voiceId) {
44
+ const lang = langOf(voiceId);
45
+ if (!lang) throw new Error(`no language mapping for voice "${voiceId}"`);
46
+ const w = await getWorker();
47
+ w.set_voice(lang.espeak);
48
+ return tidy(w.synthesize_ipa(text).ipa);
49
+ }
50
+
51
+ module.exports = { phonemize, langOf, LANGS };
@@ -64,13 +64,13 @@ Reference: https://voila.anzalabidi.dev/llms.txt · https://github.com/anzal1/vo
64
64
  for them in chat.
65
65
  - Prefer `zoom` with a `selector` over a bare `level`: voila measures the
66
66
  element and picks the level and camera centre, so nothing is cropped.
67
- - Languages: narration voice is per step, so demos can mix languages. Kokoro
68
- is ENGLISH ONLY (af_heart A, af_bella A-, bf_emma British). For other
69
- languages use a macOS system voice (`voice: "say:Monica"`, `voila voices
70
- --all` lists ~180 across ~50 languages), or `--tts-cmd` with any engine on
71
- any platform (placeholders {out} {text} {voice}), or point a step at a
72
- ready-made clip with `audio: file.mp3`. On Linux/Windows non-English needs
73
- --tts-cmd: tell the user rather than silently narrating in English.
67
+ - Languages: narration voice is per step, so demos can mix languages freely.
68
+ Kokoro covers English (af_heart A, af_bella A-, bf_emma British), Spanish
69
+ (ef_dora), French (ff_siwis), Italian (if_sara), Portuguese (pf_dora) and
70
+ Hindi (hf_alpha) on EVERY platform, on-device. `voila voices` lists all 41.
71
+ Japanese and Mandarin voices are gated because espeak mispronounces them;
72
+ for those use `--tts-cmd` with a dedicated engine, a macOS system voice
73
+ (`voice: "say:Kyoko"`, see `voila voices --all`), or `audio: clip.mp3`.
74
74
  - Voices: `voila voices` lists the English voices with quality grades. af_heart
75
75
  (A) default, af_bella (A-), af_nicole (B-), bf_emma (B-, British). `--speed`
76
76
  or the speed param (0.5-1.6) changes pace; 0.9 reads calmer.
package/voices.js CHANGED
@@ -3,6 +3,15 @@
3
3
 
4
4
  let cache = null;
5
5
 
6
+ // Voices the model ships, including the non-English ones kokoro-js leaves out
7
+ // of its metadata. Their language comes from the id prefix.
8
+ function shippedVoiceIds() {
9
+ const fs = require('fs'), path = require('path');
10
+ const dir = path.join(path.dirname(require.resolve('kokoro-js')), '..', 'voices');
11
+ try { return fs.readdirSync(dir).filter(f => f.endsWith('.bin')).map(f => f.replace('.bin', '')); }
12
+ catch { return []; }
13
+ }
14
+
6
15
  function allVoices() {
7
16
  if (cache) return cache;
8
17
  const fs = require('fs');
@@ -16,11 +25,30 @@ function allVoices() {
16
25
  while ((m = re.exec(src))) {
17
26
  out.push({ id: m[1], name: m[2], language: m[3], gender: m[4], grade: m[6] });
18
27
  }
28
+ // Fold in the non-English voices, which ship as files but carry no metadata.
29
+ const { langOf } = require('./phonemes');
30
+ const known = new Set(out.map(v => v.id));
31
+ for (const id of shippedVoiceIds()) {
32
+ if (known.has(id)) continue;
33
+ const lang = langOf(id);
34
+ if (!lang || lang.tier === 'native') continue;
35
+ out.push({
36
+ id,
37
+ name: (id.split('_')[1] || id).replace(/^./, c => c.toUpperCase()),
38
+ language: lang.name,
39
+ gender: id[1] === 'f' ? 'Female' : id[1] === 'm' ? 'Male' : '',
40
+ grade: lang.tier === 'experimental' ? 'experimental' : 'unrated',
41
+ tier: lang.tier,
42
+ });
43
+ }
44
+ for (const v of out) if (!v.tier) v.tier = 'native';
19
45
  cache = out;
20
46
  return out;
21
47
  }
22
48
 
23
49
  const gradeRank = g => {
50
+ if (g === 'unrated') return 6;
51
+ if (g === 'experimental') return 9;
24
52
  const base = { A: 0, B: 1, C: 2, D: 3, F: 4 }[g[0]] ?? 5;
25
53
  const mod = g[1] === '+' ? -0.3 : g[1] === '-' ? 0.3 : 0;
26
54
  return base + mod;
@@ -46,12 +74,15 @@ function format() {
46
74
  for (const v of rows) (byLang[v.language] = byLang[v.language] || []).push(v);
47
75
  const lines = [];
48
76
  for (const [lang, vs] of Object.entries(byLang)) {
49
- lines.push(`\n${lang} (${vs.length} voices, best first)`);
77
+ const tier = vs[0].tier;
78
+ const note = tier === 'experimental' ? ' [experimental: pronunciation is unreliable]' : '';
79
+ lines.push(`\n${lang} (${vs.length} voices)${note}`);
50
80
  for (const v of vs) {
51
- lines.push(` ${v.id.padEnd(13)} ${v.grade.padEnd(3)} ${v.gender.padEnd(7)} ${v.name}`);
81
+ lines.push(` ${v.id.padEnd(13)} ${String(v.grade).padEnd(12)} ${v.gender.padEnd(7)} ${v.name}`);
52
82
  }
53
83
  }
54
- lines.push('\nUse with: --voice af_bella (also --speed 0.9 to slow the delivery)');
84
+ lines.push('\nAll of these run on-device on every platform.');
85
+ lines.push('Use with: --voice ef_dora (--speed 0.9 slows the delivery)');
55
86
  return lines.join('\n');
56
87
  }
57
88