voila-recorder 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +6 -6
- package/README.md +12 -12
- package/audio.js +21 -2
- package/mcp.js +10 -8
- package/package.json +3 -1
- package/phonemes.js +51 -0
- package/skills/voila/SKILL.md +7 -7
- package/voices.js +34 -3
package/AGENTS.md
CHANGED
|
@@ -58,12 +58,12 @@ npx -y voila-recorder rerender <dir> --voice bf_emma
|
|
|
58
58
|
- Narration style: short sentences, product language, 8-15 words per beat.
|
|
59
59
|
- Prefer `zoom` with a `selector` over a raw `level`: voila measures the element
|
|
60
60
|
and picks the level and camera centre so nothing gets cropped.
|
|
61
|
-
- Voices: Kokoro
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
61
|
+
- Voices: Kokoro covers English, Spanish, French, Italian, Portuguese and
|
|
62
|
+
Hindi on every platform (af_heart, ef_dora, ff_siwis, if_sara, pf_dora,
|
|
63
|
+
hf_alpha). `speed` 0.5-1.6 sets pace. `voila voices` lists all 41.
|
|
64
|
+
- Mixing languages: set `voice:` per narration step.
|
|
65
|
+
- Japanese/Mandarin are gated (espeak mispronounces them). Use `--tts-cmd`
|
|
66
|
+
with a dedicated engine, a macOS `say:` voice, or `audio: clip.mp3`.
|
|
67
67
|
- Sign-in walls: recording refuses to film a login page. Run `voila login <url>`
|
|
68
68
|
(or the voila_login tool), let the HUMAN sign in in the window that opens, and
|
|
69
69
|
the session persists in a local profile for every later recording.
|
package/README.md
CHANGED
|
@@ -73,27 +73,27 @@ on-device, no cloud, no API keys ([audio.js](audio.js)). Falls back to macOS
|
|
|
73
73
|
`say` if Kokoro can't load. Voices: `af_heart` (default), `af_bella`,
|
|
74
74
|
`am_adam`, … (`voice` param). Disable with `narrate: false` / `--no-narrate`.
|
|
75
75
|
|
|
76
|
-
**
|
|
77
|
-
|
|
76
|
+
**Six languages, mixable in one demo.** Kokoro speaks English (US/UK),
|
|
77
|
+
Spanish, French, Italian, Portuguese and Hindi on every platform, on-device.
|
|
78
|
+
Set `voice:` per step:
|
|
78
79
|
|
|
79
80
|
```yaml
|
|
80
81
|
- action: hover
|
|
81
82
|
selector: h1
|
|
82
|
-
narration: "This part is English." #
|
|
83
|
+
narration: "This part is English." # af_heart
|
|
83
84
|
- action: scroll_to
|
|
84
85
|
selector: "#pricing"
|
|
85
|
-
voice:
|
|
86
|
+
voice: ef_dora # Spanish, same model
|
|
86
87
|
narration: "Esta parte está en español."
|
|
87
|
-
- action: wait
|
|
88
|
-
ms: 500
|
|
89
|
-
audio: ./clips/intro-ja.mp3 # a clip you already have
|
|
90
88
|
```
|
|
91
89
|
|
|
92
|
-
`voila voices
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
90
|
+
`voila voices` lists all 41. Every step is paced to its own spoken clip, so
|
|
91
|
+
mixed-language demos stay in sync. Japanese and Mandarin voices ship with the
|
|
92
|
+
model but are gated: espeak mispronounces them badly. For those, and for
|
|
93
|
+
anything else, plug in your own engine with
|
|
94
|
+
`--tts-cmd 'piper -m ja.onnx -f {out} -- "{text}"'`, use a macOS system voice
|
|
95
|
+
(`voice: "say:Kyoko"`, `voila voices --all`), or hand a step a ready-made clip
|
|
96
|
+
with `audio: intro.mp3`.
|
|
97
97
|
|
|
98
98
|
## Recipes — demos as code
|
|
99
99
|
|
package/audio.js
CHANGED
|
@@ -166,8 +166,27 @@ async function prepareNarration(items, dir, defaultVoice, onStatus = () => {}, s
|
|
|
166
166
|
kokoro = await KokoroTTS.from_pretrained('onnx-community/Kokoro-82M-v1.0-ONNX', { dtype: 'q8' });
|
|
167
167
|
}
|
|
168
168
|
const v = voice && voices.isValid(voice) ? voice : 'af_heart';
|
|
169
|
-
|
|
170
|
-
const
|
|
169
|
+
const { langOf, phonemize } = require('./phonemes');
|
|
170
|
+
const lang = langOf(v);
|
|
171
|
+
onStatus(`narrating ${i + 1}/${norm.length} with Kokoro (${v}, ${lang ? lang.name : 'English'})`);
|
|
172
|
+
|
|
173
|
+
let audio;
|
|
174
|
+
if (lang && lang.tier !== 'native') {
|
|
175
|
+
// kokoro-js only phonemizes English, so do it ourselves with espeak-ng
|
|
176
|
+
// (WASM, every platform) and feed the model token ids directly.
|
|
177
|
+
if (lang.tier === 'experimental' && !process.env.VOILA_EXPERIMENTAL_LANGS) {
|
|
178
|
+
throw new Error(
|
|
179
|
+
`${lang.name} voices are experimental: espeak mispronounces them badly ` +
|
|
180
|
+
`(Japanese leaks English words, Mandarin emits numeric tones Kokoro never saw). ` +
|
|
181
|
+
`Set VOILA_EXPERIMENTAL_LANGS=1 to try anyway, or use --tts-cmd with a ${lang.name} engine.`
|
|
182
|
+
);
|
|
183
|
+
}
|
|
184
|
+
const ipa = await phonemize(it.text, v);
|
|
185
|
+
const enc = kokoro.tokenizer(ipa, { truncation: true });
|
|
186
|
+
audio = await kokoro.generate_from_ids(enc.input_ids, { voice: v, speed });
|
|
187
|
+
} else {
|
|
188
|
+
audio = await kokoro.generate(it.text, { voice: v, speed });
|
|
189
|
+
}
|
|
171
190
|
await audio.save(file);
|
|
172
191
|
clips.push({
|
|
173
192
|
file,
|
package/mcp.js
CHANGED
|
@@ -37,7 +37,7 @@ function getSession(device) {
|
|
|
37
37
|
return sessions.get(key);
|
|
38
38
|
}
|
|
39
39
|
|
|
40
|
-
const server = new McpServer({ name: 'voila', version: '0.
|
|
40
|
+
const server = new McpServer({ name: 'voila', version: '0.8.0' });
|
|
41
41
|
const deviceParam = z.enum(['desktop', 'mobile', 'tablet']).optional().default('desktop');
|
|
42
42
|
|
|
43
43
|
server.tool(
|
|
@@ -69,7 +69,7 @@ server.tool(
|
|
|
69
69
|
url: z.string().url(),
|
|
70
70
|
steps_yaml: z.string().optional(),
|
|
71
71
|
narrate: z.boolean().optional().default(true),
|
|
72
|
-
voice: z.string().optional().describe('default narration voice. Kokoro
|
|
72
|
+
voice: z.string().optional().describe('default narration voice. Kokoro speaks English (af_heart, af_bella, bf_emma), Spanish (ef_dora), French (ff_siwis), Italian (if_sara), Portuguese (pf_dora) and Hindi (hf_alpha) on every platform. Per-step `voice:` overrides this, so one demo can mix languages. Japanese/Mandarin are gated (mispronounced) - use tts_cmd for those.'),
|
|
73
73
|
speed: z.number().min(0.5).max(1.6).optional().default(1).describe('narration speed; 0.9 reads calmer'),
|
|
74
74
|
tts_cmd: z.string().optional().describe('external TTS engine template for any language/platform, e.g. \'piper -m es.onnx -f {out} -- "{text}"\'. Placeholders: {out} {text} {voice}.'),
|
|
75
75
|
device: deviceParam,
|
|
@@ -118,18 +118,20 @@ server.tool(
|
|
|
118
118
|
|
|
119
119
|
server.tool(
|
|
120
120
|
'voila_voices',
|
|
121
|
-
'List narration voices. Kokoro
|
|
122
|
-
'
|
|
123
|
-
'
|
|
124
|
-
'
|
|
121
|
+
'List narration voices. Kokoro covers English, Spanish, French, Italian, Portuguese and Hindi on ' +
|
|
122
|
+
'every platform, on-device; English voices carry quality grades. Japanese and Mandarin voices exist ' +
|
|
123
|
+
'but are gated because espeak mispronounces them. Pass system:true to also list the machine\'s own ' +
|
|
124
|
+
'voices (macOS). Use before voila_record when the user asks for a different voice, an accent, a male ' +
|
|
125
|
+
'or female narrator, or a non-English language.',
|
|
125
126
|
{ system: z.boolean().optional().default(false) },
|
|
126
127
|
async ({ system }) => ({
|
|
127
128
|
content: [{ type: 'text', text: JSON.stringify({
|
|
128
129
|
kokoro: voiceCatalogue.ranked(),
|
|
129
|
-
|
|
130
|
+
languages: 'English (US/UK), Spanish, French, Italian, Portuguese (BR), Hindi',
|
|
131
|
+
gated: 'Japanese and Mandarin: espeak mispronounces them; use tts_cmd or a system voice',
|
|
130
132
|
system: system ? voiceCatalogue.systemVoices() : undefined,
|
|
131
133
|
systemLanguages: system ? Object.keys(voiceCatalogue.systemLanguages()) : undefined,
|
|
132
|
-
note: '
|
|
134
|
+
note: 'Set voice: per step to mix languages in one demo.',
|
|
133
135
|
}, null, 2) }],
|
|
134
136
|
})
|
|
135
137
|
);
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "voila-recorder",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.8.0",
|
|
4
4
|
"description": "Permission-free product demo recorder: URL in, narrated auto-zoomed MP4 out — with the recipe embedded in the video. Agent-native (MCP), fully on-device.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"main": "pipeline.js",
|
|
@@ -21,6 +21,7 @@
|
|
|
21
21
|
"modelcache.js",
|
|
22
22
|
"consent.js",
|
|
23
23
|
"doctor.js",
|
|
24
|
+
"phonemes.js",
|
|
24
25
|
"voices.js",
|
|
25
26
|
"auth.js",
|
|
26
27
|
"public/",
|
|
@@ -54,6 +55,7 @@
|
|
|
54
55
|
"demos-as-code"
|
|
55
56
|
],
|
|
56
57
|
"dependencies": {
|
|
58
|
+
"@echogarden/espeak-ng-emscripten": "^0.3.5",
|
|
57
59
|
"@modelcontextprotocol/sdk": "^1.30.0",
|
|
58
60
|
"express": "^4.19.2",
|
|
59
61
|
"ffmpeg-static": "^5.2.0",
|
package/phonemes.js
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
// Grapheme-to-phoneme for Kokoro's non-English voices.
|
|
2
|
+
//
|
|
3
|
+
// kokoro-js only ships an English phonemizer, which is why its 26 non-English
|
|
4
|
+
// voice files sat unusable. espeak-ng compiled to WASM carries the full
|
|
5
|
+
// language data, runs on every platform, and emits the IPA Kokoro was trained
|
|
6
|
+
// on. That combination gives real multilingual narration with no OS-specific
|
|
7
|
+
// dependency.
|
|
8
|
+
|
|
9
|
+
// Kokoro voice ids are prefixed by language: af_/am_ = American English,
|
|
10
|
+
// bf_/bm_ = British, ef_/em_ = Spanish, and so on.
|
|
11
|
+
const LANGS = {
|
|
12
|
+
a: { espeak: 'en-us', name: 'English (US)', tier: 'native' },
|
|
13
|
+
b: { espeak: 'en-gb', name: 'English (UK)', tier: 'native' },
|
|
14
|
+
e: { espeak: 'es', name: 'Spanish', tier: 'good' },
|
|
15
|
+
f: { espeak: 'fr-fr', name: 'French', tier: 'good' },
|
|
16
|
+
h: { espeak: 'hi', name: 'Hindi', tier: 'good' },
|
|
17
|
+
i: { espeak: 'it', name: 'Italian', tier: 'good' },
|
|
18
|
+
p: { espeak: 'pt-br', name: 'Portuguese (BR)', tier: 'good' },
|
|
19
|
+
// espeak leaks English words into Japanese kanji, and emits numeric tones
|
|
20
|
+
// for Mandarin that Kokoro was not trained on. Both need a dedicated G2P
|
|
21
|
+
// (the Python release uses one); until then they are off by default.
|
|
22
|
+
j: { espeak: 'ja', name: 'Japanese', tier: 'experimental' },
|
|
23
|
+
z: { espeak: 'cmn', name: 'Mandarin', tier: 'experimental' },
|
|
24
|
+
};
|
|
25
|
+
|
|
26
|
+
const langOf = voiceId => LANGS[String(voiceId || '')[0]] || null;
|
|
27
|
+
|
|
28
|
+
let worker = null;
|
|
29
|
+
async function getWorker() {
|
|
30
|
+
if (!worker) {
|
|
31
|
+
const mod = await require('@echogarden/espeak-ng-emscripten').default();
|
|
32
|
+
worker = await new mod.eSpeakNGWorker();
|
|
33
|
+
}
|
|
34
|
+
return worker;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
// espeak separates phonemes with underscores and sentences with newlines.
|
|
38
|
+
// Kokoro wants a plain IPA string.
|
|
39
|
+
function tidy(ipa) {
|
|
40
|
+
return ipa.replace(/_/g, '').replace(/\s*\n\s*/g, ' ').replace(/\s+/g, ' ').trim();
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
async function phonemize(text, voiceId) {
|
|
44
|
+
const lang = langOf(voiceId);
|
|
45
|
+
if (!lang) throw new Error(`no language mapping for voice "${voiceId}"`);
|
|
46
|
+
const w = await getWorker();
|
|
47
|
+
w.set_voice(lang.espeak);
|
|
48
|
+
return tidy(w.synthesize_ipa(text).ipa);
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
module.exports = { phonemize, langOf, LANGS };
|
package/skills/voila/SKILL.md
CHANGED
|
@@ -64,13 +64,13 @@ Reference: https://voila.anzalabidi.dev/llms.txt · https://github.com/anzal1/vo
|
|
|
64
64
|
for them in chat.
|
|
65
65
|
- Prefer `zoom` with a `selector` over a bare `level`: voila measures the
|
|
66
66
|
element and picks the level and camera centre, so nothing is cropped.
|
|
67
|
-
- Languages: narration voice is per step, so demos can mix languages.
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
67
|
+
- Languages: narration voice is per step, so demos can mix languages freely.
|
|
68
|
+
Kokoro covers English (af_heart A, af_bella A-, bf_emma British), Spanish
|
|
69
|
+
(ef_dora), French (ff_siwis), Italian (if_sara), Portuguese (pf_dora) and
|
|
70
|
+
Hindi (hf_alpha) on EVERY platform, on-device. `voila voices` lists all 41.
|
|
71
|
+
Japanese and Mandarin voices are gated because espeak mispronounces them;
|
|
72
|
+
for those use `--tts-cmd` with a dedicated engine, a macOS system voice
|
|
73
|
+
(`voice: "say:Kyoko"`, see `voila voices --all`), or `audio: clip.mp3`.
|
|
74
74
|
- Voices: `voila voices` lists the English voices with quality grades. af_heart
|
|
75
75
|
(A) default, af_bella (A-), af_nicole (B-), bf_emma (B-, British). `--speed`
|
|
76
76
|
or the speed param (0.5-1.6) changes pace; 0.9 reads calmer.
|
package/voices.js
CHANGED
|
@@ -3,6 +3,15 @@
|
|
|
3
3
|
|
|
4
4
|
let cache = null;
|
|
5
5
|
|
|
6
|
+
// Voices the model ships, including the non-English ones kokoro-js leaves out
|
|
7
|
+
// of its metadata. Their language comes from the id prefix.
|
|
8
|
+
function shippedVoiceIds() {
|
|
9
|
+
const fs = require('fs'), path = require('path');
|
|
10
|
+
const dir = path.join(path.dirname(require.resolve('kokoro-js')), '..', 'voices');
|
|
11
|
+
try { return fs.readdirSync(dir).filter(f => f.endsWith('.bin')).map(f => f.replace('.bin', '')); }
|
|
12
|
+
catch { return []; }
|
|
13
|
+
}
|
|
14
|
+
|
|
6
15
|
function allVoices() {
|
|
7
16
|
if (cache) return cache;
|
|
8
17
|
const fs = require('fs');
|
|
@@ -16,11 +25,30 @@ function allVoices() {
|
|
|
16
25
|
while ((m = re.exec(src))) {
|
|
17
26
|
out.push({ id: m[1], name: m[2], language: m[3], gender: m[4], grade: m[6] });
|
|
18
27
|
}
|
|
28
|
+
// Fold in the non-English voices, which ship as files but carry no metadata.
|
|
29
|
+
const { langOf } = require('./phonemes');
|
|
30
|
+
const known = new Set(out.map(v => v.id));
|
|
31
|
+
for (const id of shippedVoiceIds()) {
|
|
32
|
+
if (known.has(id)) continue;
|
|
33
|
+
const lang = langOf(id);
|
|
34
|
+
if (!lang || lang.tier === 'native') continue;
|
|
35
|
+
out.push({
|
|
36
|
+
id,
|
|
37
|
+
name: (id.split('_')[1] || id).replace(/^./, c => c.toUpperCase()),
|
|
38
|
+
language: lang.name,
|
|
39
|
+
gender: id[1] === 'f' ? 'Female' : id[1] === 'm' ? 'Male' : '',
|
|
40
|
+
grade: lang.tier === 'experimental' ? 'experimental' : 'unrated',
|
|
41
|
+
tier: lang.tier,
|
|
42
|
+
});
|
|
43
|
+
}
|
|
44
|
+
for (const v of out) if (!v.tier) v.tier = 'native';
|
|
19
45
|
cache = out;
|
|
20
46
|
return out;
|
|
21
47
|
}
|
|
22
48
|
|
|
23
49
|
const gradeRank = g => {
|
|
50
|
+
if (g === 'unrated') return 6;
|
|
51
|
+
if (g === 'experimental') return 9;
|
|
24
52
|
const base = { A: 0, B: 1, C: 2, D: 3, F: 4 }[g[0]] ?? 5;
|
|
25
53
|
const mod = g[1] === '+' ? -0.3 : g[1] === '-' ? 0.3 : 0;
|
|
26
54
|
return base + mod;
|
|
@@ -46,12 +74,15 @@ function format() {
|
|
|
46
74
|
for (const v of rows) (byLang[v.language] = byLang[v.language] || []).push(v);
|
|
47
75
|
const lines = [];
|
|
48
76
|
for (const [lang, vs] of Object.entries(byLang)) {
|
|
49
|
-
|
|
77
|
+
const tier = vs[0].tier;
|
|
78
|
+
const note = tier === 'experimental' ? ' [experimental: pronunciation is unreliable]' : '';
|
|
79
|
+
lines.push(`\n${lang} (${vs.length} voices)${note}`);
|
|
50
80
|
for (const v of vs) {
|
|
51
|
-
lines.push(` ${v.id.padEnd(13)} ${v.grade.padEnd(
|
|
81
|
+
lines.push(` ${v.id.padEnd(13)} ${String(v.grade).padEnd(12)} ${v.gender.padEnd(7)} ${v.name}`);
|
|
52
82
|
}
|
|
53
83
|
}
|
|
54
|
-
lines.push('\
|
|
84
|
+
lines.push('\nAll of these run on-device on every platform.');
|
|
85
|
+
lines.push('Use with: --voice ef_dora (--speed 0.9 slows the delivery)');
|
|
55
86
|
return lines.join('\n');
|
|
56
87
|
}
|
|
57
88
|
|