voila-recorder 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +6 -2
- package/README.md +22 -0
- package/RECIPE.md +4 -0
- package/audio.js +107 -17
- package/cli.js +7 -3
- package/mcp.js +19 -9
- package/package.json +3 -3
- package/pipeline.js +8 -7
- package/recorder.js +5 -2
- package/scripts/tts-cmd-fixture.js +10 -0
- package/skills/voila/SKILL.md +8 -1
- package/tour.js +1 -1
- package/voices.js +49 -1
package/AGENTS.md
CHANGED
|
@@ -58,8 +58,12 @@ npx -y voila-recorder rerender <dir> --voice bf_emma
|
|
|
58
58
|
- Narration style: short sentences, product language, 8-15 words per beat.
|
|
59
59
|
- Prefer `zoom` with a `selector` over a raw `level`: voila measures the element
|
|
60
60
|
and picks the level and camera centre so nothing gets cropped.
|
|
61
|
-
- Voices:
|
|
62
|
-
|
|
61
|
+
- Voices: Kokoro is English only (af_heart A, af_bella A-, bf_emma British).
|
|
62
|
+
`speed` 0.5-1.6 sets pace.
|
|
63
|
+
- Other languages: set `voice:` per narration step, so one demo can mix them.
|
|
64
|
+
`say:Monica` uses a macOS system voice (~50 languages, `voila voices --all`);
|
|
65
|
+
`--tts-cmd 'engine -o {out} "{text}"'` plugs in any engine on any platform;
|
|
66
|
+
`audio: clip.mp3` on a step uses a file you already have.
|
|
63
67
|
- Sign-in walls: recording refuses to film a login page. Run `voila login <url>`
|
|
64
68
|
(or the voila_login tool), let the HUMAN sign in in the window that opens, and
|
|
65
69
|
the session persists in a local profile for every later recording.
|
package/README.md
CHANGED
|
@@ -73,6 +73,28 @@ on-device, no cloud, no API keys ([audio.js](audio.js)). Falls back to macOS
|
|
|
73
73
|
`say` if Kokoro can't load. Voices: `af_heart` (default), `af_bella`,
|
|
74
74
|
`am_adam`, … (`voice` param). Disable with `narrate: false` / `--no-narrate`.
|
|
75
75
|
|
|
76
|
+
**Other languages, and mixing them.** Kokoro's JS port speaks English only, so
|
|
77
|
+
non-English narration comes from one of two other engines, chosen per step:
|
|
78
|
+
|
|
79
|
+
```yaml
|
|
80
|
+
- action: hover
|
|
81
|
+
selector: h1
|
|
82
|
+
narration: "This part is English." # Kokoro
|
|
83
|
+
- action: scroll_to
|
|
84
|
+
selector: "#pricing"
|
|
85
|
+
voice: "say:Mónica" # macOS system voice
|
|
86
|
+
narration: "Esta parte está en español."
|
|
87
|
+
- action: wait
|
|
88
|
+
ms: 500
|
|
89
|
+
audio: ./clips/intro-ja.mp3 # a clip you already have
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
`voila voices --all` lists the ~180 system voices across ~50 languages on
|
|
93
|
+
macOS. On Linux and Windows use any engine you like:
|
|
94
|
+
`--tts-cmd 'piper -m es.onnx -f {out} -- "{text}"'` ({out}, {text}, {voice}).
|
|
95
|
+
Every step is still paced to its own spoken clip, so mixed-language demos stay
|
|
96
|
+
in sync.
|
|
97
|
+
|
|
76
98
|
## Recipes — demos as code
|
|
77
99
|
|
|
78
100
|
Every video ships with its source: `recipe.json` (URL + steps + narration +
|
package/RECIPE.md
CHANGED
|
@@ -51,6 +51,10 @@ Each step: `{ action, ...params, caption?, narration?, optional?, pause? }`
|
|
|
51
51
|
| `zoom` | `level` (1-3) or `selector`, `ms` | camera zoom; with a selector it frames that element |
|
|
52
52
|
| `wait` | `ms` | hold (cursor keeps breathing on long holds) |
|
|
53
53
|
|
|
54
|
+
Any step may also carry `voice:` (Kokoro id, `say:Name`, or an id for your
|
|
55
|
+
`--tts-cmd` engine) and `audio:` (a ready-made clip, skipping TTS entirely).
|
|
56
|
+
Because voice is per step, a recipe can mix languages.
|
|
57
|
+
|
|
54
58
|
`caption` renders as a lower-third; `narration` is spoken by on-device TTS and
|
|
55
59
|
**paces the segment** — the recording holds until the clip finishes, so a
|
|
56
60
|
recreated demo re-times itself to whatever voice regenerates it.
|
package/audio.js
CHANGED
|
@@ -31,7 +31,7 @@ async function getKokoro() {
|
|
|
31
31
|
return kokoroInstance;
|
|
32
32
|
}
|
|
33
33
|
|
|
34
|
-
async function
|
|
34
|
+
async function _unusedSynthKokoro(texts, dir, voice, onStatus, speed = 1) {
|
|
35
35
|
// Fail loudly on a bad voice name rather than silently using the default.
|
|
36
36
|
if (voice && /^[a-z]{2}_/.test(voice) && !voices.isValid(voice)) {
|
|
37
37
|
throw new Error(`unknown voice "${voice}". Try: ${voices.suggest(voice).join(', ')} (run \`voila voices\` for all ${voices.ranked().length})`);
|
|
@@ -77,7 +77,7 @@ async function pickSayVoice(preferred) {
|
|
|
77
77
|
return 'Samantha';
|
|
78
78
|
}
|
|
79
79
|
|
|
80
|
-
async function
|
|
80
|
+
async function _unusedSynthSay(texts, dir, voice, onStatus) {
|
|
81
81
|
const v = await pickSayVoice(voice);
|
|
82
82
|
onStatus(`narrating with say (${v})`);
|
|
83
83
|
const clips = [];
|
|
@@ -90,25 +90,113 @@ async function synthSay(texts, dir, voice, onStatus) {
|
|
|
90
90
|
}
|
|
91
91
|
|
|
92
92
|
// --- pipeline entry ----------------------------------------------------------
|
|
93
|
+
// Each narration item is {text, voice?, audio?}. Voice decides the engine:
|
|
94
|
+
// af_heart -> Kokoro (English, best quality, cross platform)
|
|
95
|
+
// say:Monica -> a system voice (macOS; ~50 languages)
|
|
96
|
+
// (--tts-cmd set) -> any external engine, any language
|
|
97
|
+
// A step can also point at a ready-made file with `audio:`, which bypasses TTS.
|
|
98
|
+
// Because the voice is per item, one demo can mix languages.
|
|
99
|
+
|
|
100
|
+
function engineFor(voice, ttsCmd) {
|
|
101
|
+
// Supplying --tts-cmd means "use my engine", unless a step names a specific
|
|
102
|
+
// Kokoro or system voice, which then wins for that step.
|
|
103
|
+
if (!voice) {
|
|
104
|
+
if (ttsCmd) return 'cmd';
|
|
105
|
+
return process.env.VOILA_TTS === 'say' ? 'say' : 'kokoro';
|
|
106
|
+
}
|
|
107
|
+
if (String(voice).startsWith('cmd:') || (ttsCmd && String(voice).startsWith('cmd'))) return 'cmd';
|
|
108
|
+
if (String(voice).startsWith('say:')) return 'say';
|
|
109
|
+
if (voices.isValid(voice)) return 'kokoro';
|
|
110
|
+
if (voices.isSystemVoice(voice)) return 'say';
|
|
111
|
+
if (ttsCmd) return 'cmd';
|
|
112
|
+
throw new Error(
|
|
113
|
+
`unknown voice "${voice}". Kokoro (English): ${voices.suggest(voice).join(', ')}. ` +
|
|
114
|
+
`For other languages use a system voice like "say:Monica" (see \`voila voices --all\`) ` +
|
|
115
|
+
`or supply --tts-cmd for your own engine.`
|
|
116
|
+
);
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
async function sayOne(text, file, voiceName, speed) {
|
|
120
|
+
const rate = Math.round(185 * (speed || 1));
|
|
121
|
+
await run('say', ['-v', voiceName, '-r', String(rate), '-o', file, text]);
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
async function cmdOne(text, file, voiceName, ttsCmd) {
|
|
125
|
+
const cmd = ttsCmd
|
|
126
|
+
.replaceAll('{text}', text.replace(/"/g, '\\"'))
|
|
127
|
+
.replaceAll('{out}', file)
|
|
128
|
+
.replaceAll('{voice}', voiceName || '');
|
|
129
|
+
await new Promise((res, rej) => {
|
|
130
|
+
require('child_process').exec(cmd, { maxBuffer: 1e7 }, (err, _o, se) =>
|
|
131
|
+
err ? rej(new Error(`tts-cmd failed: ${String(se || err.message).slice(0, 200)}`)) : res());
|
|
132
|
+
});
|
|
133
|
+
if (!fs.existsSync(file)) throw new Error(`tts-cmd produced no file at ${file}`);
|
|
134
|
+
}
|
|
93
135
|
|
|
94
|
-
// Synthesize narration
|
|
95
|
-
// spoken durations. Returns {clips:
|
|
96
|
-
async function prepareNarration(
|
|
136
|
+
// Synthesize every narration clip up front so the recorder can pace segments
|
|
137
|
+
// to real spoken durations. Returns {clips:[{file,durMs}], voice, backend}.
|
|
138
|
+
async function prepareNarration(items, dir, defaultVoice, onStatus = () => {}, speed = 1, ttsCmd = null) {
|
|
97
139
|
fs.mkdirSync(dir, { recursive: true });
|
|
98
|
-
const
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
140
|
+
const norm = items.map(it => (typeof it === 'string' ? { text: it } : it));
|
|
141
|
+
const clips = [];
|
|
142
|
+
const used = new Set();
|
|
143
|
+
let kokoro = null;
|
|
144
|
+
|
|
145
|
+
for (let i = 0; i < norm.length; i++) {
|
|
146
|
+
const it = norm[i];
|
|
147
|
+
const voice = it.voice || defaultVoice || null;
|
|
148
|
+
|
|
149
|
+
// A pre-made audio file wins over any engine.
|
|
150
|
+
if (it.audio) {
|
|
151
|
+
if (!fs.existsSync(it.audio)) throw new Error(`audio file not found: ${it.audio}`);
|
|
152
|
+
clips.push({ file: it.audio, durMs: await ffDurationMs(it.audio) });
|
|
153
|
+
used.add('file');
|
|
154
|
+
continue;
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
const engine = engineFor(voice, ttsCmd);
|
|
158
|
+
const ext = engine === 'kokoro' ? 'wav' : engine === 'say' ? 'aiff' : 'wav';
|
|
159
|
+
const file = path.join(dir, `seg${i}.${ext}`);
|
|
160
|
+
|
|
161
|
+
if (engine === 'kokoro') {
|
|
162
|
+
if (!kokoro) {
|
|
163
|
+
onStatus('loading Kokoro TTS');
|
|
164
|
+
require('./modelcache').useStableCache();
|
|
165
|
+
const { KokoroTTS } = require('kokoro-js');
|
|
166
|
+
kokoro = await KokoroTTS.from_pretrained('onnx-community/Kokoro-82M-v1.0-ONNX', { dtype: 'q8' });
|
|
167
|
+
}
|
|
168
|
+
const v = voice && voices.isValid(voice) ? voice : 'af_heart';
|
|
169
|
+
onStatus(`narrating ${i + 1}/${norm.length} with Kokoro (${v})`);
|
|
170
|
+
const audio = await kokoro.generate(it.text, { voice: v, speed });
|
|
171
|
+
await audio.save(file);
|
|
172
|
+
clips.push({
|
|
173
|
+
file,
|
|
174
|
+
durMs: audio.audio && audio.sampling_rate
|
|
175
|
+
? Math.round((audio.audio.length / audio.sampling_rate) * 1000)
|
|
176
|
+
: await ffDurationMs(file),
|
|
177
|
+
});
|
|
178
|
+
used.add(`kokoro:${v}`);
|
|
179
|
+
} else if (engine === 'say') {
|
|
180
|
+
if (process.platform !== 'darwin') {
|
|
181
|
+
throw new Error(`system voices need macOS. Use a Kokoro voice for English, or --tts-cmd on this platform.`);
|
|
182
|
+
}
|
|
183
|
+
const name = String(voice).replace(/^say:/, '');
|
|
184
|
+
onStatus(`narrating ${i + 1}/${norm.length} with system voice (${name})`);
|
|
185
|
+
await sayOne(it.text, file, name, speed);
|
|
186
|
+
clips.push({ file, durMs: await ffDurationMs(file) });
|
|
187
|
+
used.add(`say:${name}`);
|
|
188
|
+
} else {
|
|
189
|
+
onStatus(`narrating ${i + 1}/${norm.length} with tts-cmd`);
|
|
190
|
+
await cmdOne(it.text, file, String(voice || '').replace(/^cmd:/, ''), ttsCmd);
|
|
191
|
+
clips.push({ file, durMs: await ffDurationMs(file) });
|
|
192
|
+
used.add('cmd');
|
|
105
193
|
}
|
|
106
194
|
}
|
|
107
|
-
|
|
108
|
-
|
|
195
|
+
|
|
196
|
+
return { clips, voice: [...used].join(', ') || 'none', backend: [...used].join(', ') };
|
|
109
197
|
}
|
|
110
198
|
|
|
111
|
-
async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1, prepared = null, onStatus = () => {} } = {}) {
|
|
199
|
+
async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1, ttsCmd = null, prepared = null, onStatus = () => {} } = {}) {
|
|
112
200
|
const segs = (meta.segments || []).filter(s => s.narration);
|
|
113
201
|
if (!segs.length) {
|
|
114
202
|
fs.copyFileSync(videoIn, videoOut);
|
|
@@ -120,7 +208,9 @@ async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1,
|
|
|
120
208
|
let synth = prepared && prepared.clips.length === segs.length ? prepared : null;
|
|
121
209
|
if (!synth) {
|
|
122
210
|
try {
|
|
123
|
-
synth = await prepareNarration(
|
|
211
|
+
synth = await prepareNarration(
|
|
212
|
+
segs.map(s => ({ text: s.narration, voice: s.voice, audio: s.audio })),
|
|
213
|
+
dir, voice, onStatus, speed, ttsCmd);
|
|
124
214
|
} catch (e) {
|
|
125
215
|
onStatus(`narration skipped: ${e.message}`);
|
|
126
216
|
fs.copyFileSync(videoIn, videoOut);
|
|
@@ -154,7 +244,7 @@ async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1,
|
|
|
154
244
|
ff.on('error', rej);
|
|
155
245
|
});
|
|
156
246
|
|
|
157
|
-
fs.rmSync(dir, { recursive: true, force: true });
|
|
247
|
+
fs.rmSync(dir, { recursive: true, force: true }); // only generated clips live here
|
|
158
248
|
return { narrated: true, voice: synth.voice, backend: synth.backend, segments: clips.length };
|
|
159
249
|
}
|
|
160
250
|
|
package/cli.js
CHANGED
|
@@ -20,10 +20,10 @@ function arg(name, fallback = null) {
|
|
|
20
20
|
const USAGE = `usage:
|
|
21
21
|
voila doctor (check + download everything voila needs)
|
|
22
22
|
voila outline <url> [--device desktop|mobile|tablet]
|
|
23
|
-
voila record <url> [--steps f.yaml] [--device mobile] [--voice name] [--speed 1] [--no-narrate] [--headful] [--keep-frames] [--no-dismiss] [--dismiss sel] [--out dir] [--profile dir]
|
|
23
|
+
voila record <url> [--steps f.yaml] [--device mobile] [--voice name] [--speed 1] [--tts-cmd tmpl] [--no-narrate] [--headful] [--keep-frames] [--no-dismiss] [--dismiss sel] [--out dir] [--profile dir]
|
|
24
24
|
voila review <video.mp4> [--frames 12] [--out dir]
|
|
25
25
|
voila login <url> [--profile dir] (sign in yourself; session is saved locally)
|
|
26
|
-
voila voices
|
|
26
|
+
voila voices [--all] (Kokoro voices; --all adds system voices for other languages)
|
|
27
27
|
voila fork <video.mp4> [--url u] [--voice v] [--print] [--out dir]
|
|
28
28
|
voila rerender <dir> [--voice v] [--speed n] (needs --keep-frames on the original)
|
|
29
29
|
voila skill (install the voila skill into ~/.claude/skills)
|
|
@@ -49,7 +49,9 @@ const USAGE = `usage:
|
|
|
49
49
|
process.exit(r.ready ? 0 : 1);
|
|
50
50
|
}
|
|
51
51
|
if (cmd === 'voices') {
|
|
52
|
-
|
|
52
|
+
const v = require('./voices');
|
|
53
|
+
console.log(process.argv.includes('--all') ? v.formatAll() : v.format());
|
|
54
|
+
if (!process.argv.includes('--all')) console.log('\nNon-English? run: voila voices --all');
|
|
53
55
|
return;
|
|
54
56
|
}
|
|
55
57
|
if (cmd === 'skill') {
|
|
@@ -124,6 +126,7 @@ const USAGE = `usage:
|
|
|
124
126
|
const r = await rerender(dir, {
|
|
125
127
|
voice: arg('--voice'),
|
|
126
128
|
speed: Number(arg('--speed', '1')) || 1,
|
|
129
|
+
ttsCmd: arg('--tts-cmd'),
|
|
127
130
|
narrate: !process.argv.includes('--no-narrate'),
|
|
128
131
|
onStatus: m => console.error('[voila]', m),
|
|
129
132
|
});
|
|
@@ -166,6 +169,7 @@ const USAGE = `usage:
|
|
|
166
169
|
narrate: !process.argv.includes('--no-narrate'),
|
|
167
170
|
voice: arg('--voice'),
|
|
168
171
|
speed: Number(arg('--speed', '1')) || 1,
|
|
172
|
+
ttsCmd: arg('--tts-cmd'),
|
|
169
173
|
keepFrames: process.argv.includes('--keep-frames'),
|
|
170
174
|
onStatus: s => console.error('[voila]', s),
|
|
171
175
|
});
|
package/mcp.js
CHANGED
|
@@ -37,7 +37,7 @@ function getSession(device) {
|
|
|
37
37
|
return sessions.get(key);
|
|
38
38
|
}
|
|
39
39
|
|
|
40
|
-
const server = new McpServer({ name: 'voila', version: '0.
|
|
40
|
+
const server = new McpServer({ name: 'voila', version: '0.7.0' });
|
|
41
41
|
const deviceParam = z.enum(['desktop', 'mobile', 'tablet']).optional().default('desktop');
|
|
42
42
|
|
|
43
43
|
server.tool(
|
|
@@ -61,6 +61,7 @@ server.tool(
|
|
|
61
61
|
'caption is burned into the video as a lower-third; narration is spoken via on-device TTS (Kokoro) at that step, ' +
|
|
62
62
|
'and segment pacing automatically stretches to fit each narration clip — no need to pad waits. ' +
|
|
63
63
|
'Steps marked optional:true are skipped on failure instead of aborting. ' +
|
|
64
|
+
'A step may set its own voice: (mixing languages within one demo) or audio: (a ready-made clip). ' +
|
|
64
65
|
'device selects the recorded viewport (mobile emulates an iPhone-class device). ' +
|
|
65
66
|
'On failure the error names the failing step and includes the live page outline — patch the steps and retry. ' +
|
|
66
67
|
'Returns the MP4 path, the recipe path, and any warnings.',
|
|
@@ -68,16 +69,17 @@ server.tool(
|
|
|
68
69
|
url: z.string().url(),
|
|
69
70
|
steps_yaml: z.string().optional(),
|
|
70
71
|
narrate: z.boolean().optional().default(true),
|
|
71
|
-
voice: z.string().optional().describe('narration voice
|
|
72
|
+
voice: z.string().optional().describe('default narration voice. Kokoro is ENGLISH ONLY (af_heart, af_bella, bf_emma). For other languages use a macOS system voice ("say:Monica") or set tts_cmd. Per-step `voice:` overrides this, so one demo can mix languages.'),
|
|
72
73
|
speed: z.number().min(0.5).max(1.6).optional().default(1).describe('narration speed; 0.9 reads calmer'),
|
|
74
|
+
tts_cmd: z.string().optional().describe('external TTS engine template for any language/platform, e.g. \'piper -m es.onnx -f {out} -- "{text}"\'. Placeholders: {out} {text} {voice}.'),
|
|
73
75
|
device: deviceParam,
|
|
74
76
|
},
|
|
75
|
-
async ({ url, steps_yaml, narrate, voice, speed, device }) => enqueue(async () => {
|
|
77
|
+
async ({ url, steps_yaml, narrate, voice, speed, tts_cmd, device }) => enqueue(async () => {
|
|
76
78
|
const steps = steps_yaml ? yaml.load(steps_yaml) : null;
|
|
77
79
|
const workDir = path.join(__dirname, 'recordings', `mcp-${Date.now()}`);
|
|
78
80
|
fs.mkdirSync(workDir, { recursive: true });
|
|
79
81
|
const result = await produceDemo(getSession(device), {
|
|
80
|
-
url, steps, workDir, narrate, voice: voice || null, speed,
|
|
82
|
+
url, steps, workDir, narrate, voice: voice || null, speed, ttsCmd: tts_cmd || null,
|
|
81
83
|
onStatus: () => {},
|
|
82
84
|
});
|
|
83
85
|
return {
|
|
@@ -116,11 +118,19 @@ server.tool(
|
|
|
116
118
|
|
|
117
119
|
server.tool(
|
|
118
120
|
'voila_voices',
|
|
119
|
-
'List
|
|
120
|
-
'
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
121
|
+
'List narration voices. Kokoro voices are English only, with quality grades. Pass system:true to ' +
|
|
122
|
+
'also get the machine\'s system voices, which is how you narrate other languages (macOS only; on ' +
|
|
123
|
+
'Linux or Windows use tts_cmd instead). Use before voila_record when the user asks for a different ' +
|
|
124
|
+
'voice, an accent, a male or female narrator, or a non-English language.',
|
|
125
|
+
{ system: z.boolean().optional().default(false) },
|
|
126
|
+
async ({ system }) => ({
|
|
127
|
+
content: [{ type: 'text', text: JSON.stringify({
|
|
128
|
+
kokoro: voiceCatalogue.ranked(),
|
|
129
|
+
englishOnly: true,
|
|
130
|
+
system: system ? voiceCatalogue.systemVoices() : undefined,
|
|
131
|
+
systemLanguages: system ? Object.keys(voiceCatalogue.systemLanguages()) : undefined,
|
|
132
|
+
note: 'Non-English: use a system voice (say:Name) on macOS, or tts_cmd on any platform.',
|
|
133
|
+
}, null, 2) }],
|
|
124
134
|
})
|
|
125
135
|
);
|
|
126
136
|
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "voila-recorder",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "Permission-free product demo recorder: URL in, narrated auto-zoomed MP4 out
|
|
3
|
+
"version": "0.7.0",
|
|
4
|
+
"description": "Permission-free product demo recorder: URL in, narrated auto-zoomed MP4 out — with the recipe embedded in the video. Agent-native (MCP), fully on-device.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"main": "pipeline.js",
|
|
7
7
|
"bin": {
|
|
@@ -62,4 +62,4 @@
|
|
|
62
62
|
"playwright": "^1.48.0",
|
|
63
63
|
"sharp": "^0.33.5"
|
|
64
64
|
}
|
|
65
|
-
}
|
|
65
|
+
}
|
package/pipeline.js
CHANGED
|
@@ -44,15 +44,16 @@ function embedRecipe(videoIn, videoOut, recipe) {
|
|
|
44
44
|
});
|
|
45
45
|
}
|
|
46
46
|
|
|
47
|
-
async function produceDemo(session, { url, mode = 'auto', steps = null, workDir, voice = null, speed = 1, narrate = true, keepFrames = false, onStatus = () => {} }) {
|
|
47
|
+
async function produceDemo(session, { url, mode = 'auto', steps = null, workDir, voice = null, speed = 1, ttsCmd = null, narrate = true, keepFrames = false, onStatus = () => {} }) {
|
|
48
48
|
// Steps mode: synthesize narration BEFORE recording so segment pacing and
|
|
49
49
|
// caption lifetimes match the spoken clip durations exactly.
|
|
50
50
|
let prepared = null;
|
|
51
51
|
if (steps && narrate) {
|
|
52
|
-
const
|
|
53
|
-
|
|
52
|
+
const items = steps.filter(s => s.narration)
|
|
53
|
+
.map(s => ({ text: s.narration, voice: s.voice, audio: s.audio }));
|
|
54
|
+
if (items.length) {
|
|
54
55
|
try {
|
|
55
|
-
prepared = await prepareNarration(
|
|
56
|
+
prepared = await prepareNarration(items, path.join(workDir, 'tts'), voice, onStatus, speed, ttsCmd);
|
|
56
57
|
let i = 0;
|
|
57
58
|
for (const s of steps) if (s.narration) s._narrDurMs = prepared.clips[i++].durMs;
|
|
58
59
|
} catch (e) {
|
|
@@ -69,7 +70,7 @@ async function produceDemo(session, { url, mode = 'auto', steps = null, workDir,
|
|
|
69
70
|
await render(meta, raw, { onStatus });
|
|
70
71
|
|
|
71
72
|
let narration = { narrated: false };
|
|
72
|
-
if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, prepared, onStatus });
|
|
73
|
+
if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, ttsCmd, prepared, onStatus });
|
|
73
74
|
else fs.copyFileSync(raw, narrated);
|
|
74
75
|
|
|
75
76
|
const recipe = buildRecipe({ url, mode, steps, meta });
|
|
@@ -88,7 +89,7 @@ async function produceDemo(session, { url, mode = 'auto', steps = null, workDir,
|
|
|
88
89
|
|
|
89
90
|
// Re-produce the video from frames already on disk: no browser, no re-driving
|
|
90
91
|
// the page. Used to swap the narration voice or speed after the fact.
|
|
91
|
-
async function rerender(workDir, { voice = null, speed = 1, narrate = true, onStatus = () => {} } = {}) {
|
|
92
|
+
async function rerender(workDir, { voice = null, speed = 1, ttsCmd = null, narrate = true, onStatus = () => {} } = {}) {
|
|
92
93
|
const metaPath = path.join(workDir, 'meta.json');
|
|
93
94
|
if (!fs.existsSync(metaPath)) throw new Error(`no meta.json in ${workDir}`);
|
|
94
95
|
const meta = JSON.parse(fs.readFileSync(metaPath, 'utf8'));
|
|
@@ -102,7 +103,7 @@ async function rerender(workDir, { voice = null, speed = 1, narrate = true, onSt
|
|
|
102
103
|
await render(meta, raw, { onStatus });
|
|
103
104
|
|
|
104
105
|
let narration = { narrated: false };
|
|
105
|
-
if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, onStatus });
|
|
106
|
+
if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, ttsCmd, onStatus });
|
|
106
107
|
else fs.copyFileSync(raw, narrated);
|
|
107
108
|
|
|
108
109
|
const recipe = JSON.parse(fs.readFileSync(path.join(workDir, 'recipe.json'), 'utf8'));
|
package/recorder.js
CHANGED
|
@@ -40,8 +40,11 @@ class Timeline {
|
|
|
40
40
|
this.zoom = 1;
|
|
41
41
|
this.center = null;
|
|
42
42
|
}
|
|
43
|
-
recordSegment(caption, narration, dur = null) {
|
|
44
|
-
this.segments.push({
|
|
43
|
+
recordSegment(caption, narration, dur = null, voice = null, audio = null) {
|
|
44
|
+
this.segments.push({
|
|
45
|
+
t: Date.now(), caption: caption || null, narration: narration || null, dur,
|
|
46
|
+
...(voice ? { voice } : {}), ...(audio ? { audio } : {}),
|
|
47
|
+
});
|
|
45
48
|
}
|
|
46
49
|
recordMove(to, dur) {
|
|
47
50
|
this.moves.push({ t: Date.now(), from: { ...this.pos }, to: { ...to }, dur });
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
// CI fixture: a stand-in "external TTS engine". Writes a short tone to {out}
|
|
2
|
+
// so the --tts-cmd plumbing can be exercised on every platform, including
|
|
3
|
+
// ones with no system voices installed.
|
|
4
|
+
const { execFileSync } = require('child_process');
|
|
5
|
+
const ffmpeg = require('ffmpeg-static');
|
|
6
|
+
const out = process.argv[2];
|
|
7
|
+
const text = process.argv.slice(3).join(' ');
|
|
8
|
+
const seconds = Math.max(1, Math.min(8, text.split(/\s+/).length / 3)).toFixed(2);
|
|
9
|
+
execFileSync(ffmpeg, ['-y', '-f', 'lavfi', '-i', `sine=frequency=340:duration=${seconds}`, '-ac', '1', out], { stdio: 'pipe' });
|
|
10
|
+
console.log(`fixture wrote ${seconds}s to ${out}`);
|
package/skills/voila/SKILL.md
CHANGED
|
@@ -64,7 +64,14 @@ Reference: https://voila.anzalabidi.dev/llms.txt · https://github.com/anzal1/vo
|
|
|
64
64
|
for them in chat.
|
|
65
65
|
- Prefer `zoom` with a `selector` over a bare `level`: voila measures the
|
|
66
66
|
element and picks the level and camera centre, so nothing is cropped.
|
|
67
|
-
-
|
|
67
|
+
- Languages: narration voice is per step, so demos can mix languages. Kokoro
|
|
68
|
+
is ENGLISH ONLY (af_heart A, af_bella A-, bf_emma British). For other
|
|
69
|
+
languages use a macOS system voice (`voice: "say:Monica"`, `voila voices
|
|
70
|
+
--all` lists ~180 across ~50 languages), or `--tts-cmd` with any engine on
|
|
71
|
+
any platform (placeholders {out} {text} {voice}), or point a step at a
|
|
72
|
+
ready-made clip with `audio: file.mp3`. On Linux/Windows non-English needs
|
|
73
|
+
--tts-cmd: tell the user rather than silently narrating in English.
|
|
74
|
+
- Voices: `voila voices` lists the English voices with quality grades. af_heart
|
|
68
75
|
(A) default, af_bella (A-), af_nicole (B-), bf_emma (B-, British). `--speed`
|
|
69
76
|
or the speed param (0.5-1.6) changes pace; 0.9 reads calmer.
|
|
70
77
|
- A failing step is retried once automatically; warnings appear in the result.
|
package/tour.js
CHANGED
|
@@ -242,7 +242,7 @@ async function runSteps(page, tl, steps, opts) {
|
|
|
242
242
|
const step = steps[si];
|
|
243
243
|
if (step.caption || step.narration) {
|
|
244
244
|
await finishSegment();
|
|
245
|
-
tl.recordSegment(step.caption, step.narration, step._narrDurMs || null);
|
|
245
|
+
tl.recordSegment(step.caption, step.narration, step._narrDurMs || null, step.voice || null, step.audio || null);
|
|
246
246
|
segStart = Date.now();
|
|
247
247
|
segMinMs = (step._narrDurMs || 0) + 600;
|
|
248
248
|
}
|
package/voices.js
CHANGED
|
@@ -55,4 +55,52 @@ function format() {
|
|
|
55
55
|
return lines.join('\n');
|
|
56
56
|
}
|
|
57
57
|
|
|
58
|
-
|
|
58
|
+
// --- system voices (macOS `say`) ---------------------------------------------
|
|
59
|
+
// Kokoro is English-only in JS (its other voice files ship without a
|
|
60
|
+
// grapheme-to-phoneme stage for those languages). Every Mac already carries
|
|
61
|
+
// ~180 voices across ~50 languages, so those cover non-English narration.
|
|
62
|
+
|
|
63
|
+
let sysCache = null;
|
|
64
|
+
function systemVoices() {
|
|
65
|
+
if (sysCache) return sysCache;
|
|
66
|
+
sysCache = [];
|
|
67
|
+
if (process.platform !== 'darwin') return sysCache;
|
|
68
|
+
try {
|
|
69
|
+
const { execFileSync } = require('child_process');
|
|
70
|
+
const out = String(execFileSync('say', ['-v', '?'], { maxBuffer: 4e6 }));
|
|
71
|
+
for (const line of out.split('\n')) {
|
|
72
|
+
const m = /^(.+?)\s{2,}([a-z]{2}_[A-Z]{2})\s/.exec(line);
|
|
73
|
+
if (m) sysCache.push({ id: `say:${m[1].trim()}`, name: m[1].trim(), language: m[2].replace('_', '-'), gender: '', grade: 'system' });
|
|
74
|
+
}
|
|
75
|
+
} catch { /* no say binary */ }
|
|
76
|
+
return sysCache;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
function systemLanguages() {
|
|
80
|
+
const langs = {};
|
|
81
|
+
for (const v of systemVoices()) (langs[v.language] = langs[v.language] || []).push(v.name);
|
|
82
|
+
return langs;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
function isSystemVoice(id) {
|
|
86
|
+
if (!id) return false;
|
|
87
|
+
const name = String(id).replace(/^say:/, '').toLowerCase();
|
|
88
|
+
return systemVoices().some(v => v.name.toLowerCase() === name);
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
function formatAll() {
|
|
92
|
+
const lines = [format()];
|
|
93
|
+
const langs = systemLanguages();
|
|
94
|
+
const codes = Object.keys(langs).sort();
|
|
95
|
+
if (!codes.length) {
|
|
96
|
+
lines.push('\nSystem voices: none found (macOS only). For other languages use --tts-cmd.');
|
|
97
|
+
return lines.join('\n');
|
|
98
|
+
}
|
|
99
|
+
lines.push(`\nSystem voices (macOS, ${systemVoices().length} across ${codes.length} languages)`);
|
|
100
|
+
for (const c of codes) lines.push(` ${c.padEnd(7)} ${langs[c].slice(0, 6).join(', ')}${langs[c].length > 6 ? ` +${langs[c].length - 6}` : ''}`);
|
|
101
|
+
lines.push('\nUse a system voice for non-English narration: --voice "say:Monica"');
|
|
102
|
+
lines.push('Any other engine: --tts-cmd \'piper --model es.onnx -f {out} -- "{text}"\'');
|
|
103
|
+
return lines.join('\n');
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
module.exports = { allVoices, ranked, isValid, suggest, format, formatAll, systemVoices, systemLanguages, isSystemVoice };
|