voila-recorder 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +9 -2
- package/README.md +31 -0
- package/RECIPE.md +4 -0
- package/audio.js +108 -17
- package/cli.js +73 -3
- package/consent.js +97 -0
- package/doctor.js +118 -0
- package/mcp.js +19 -9
- package/modelcache.js +22 -0
- package/package.json +6 -2
- package/pipeline.js +39 -8
- package/recorder.js +16 -8
- package/scripts/check-audio.js +14 -0
- package/scripts/tts-cmd-fixture.js +10 -0
- package/skills/voila/SKILL.md +18 -1
- package/tour.js +1 -1
- package/voices.js +49 -1
package/AGENTS.md
CHANGED
|
@@ -24,6 +24,9 @@ npx -y voila-recorder record <url> --steps steps.yaml [--device mobile]
|
|
|
24
24
|
npx -y voila-recorder review demo.mp4
|
|
25
25
|
npx -y voila-recorder voices
|
|
26
26
|
npx -y voila-recorder login <url>
|
|
27
|
+
npx -y voila-recorder doctor
|
|
28
|
+
npx -y voila-recorder fork demo.mp4 [--url other] [--print]
|
|
29
|
+
npx -y voila-recorder rerender <dir> --voice bf_emma
|
|
27
30
|
```
|
|
28
31
|
|
|
29
32
|
## The loop — always follow it
|
|
@@ -55,8 +58,12 @@ npx -y voila-recorder login <url>
|
|
|
55
58
|
- Narration style: short sentences, product language, 8-15 words per beat.
|
|
56
59
|
- Prefer `zoom` with a `selector` over a raw `level`: voila measures the element
|
|
57
60
|
and picks the level and camera centre so nothing gets cropped.
|
|
58
|
-
- Voices:
|
|
59
|
-
|
|
61
|
+
- Voices: Kokoro is English only (af_heart A, af_bella A-, bf_emma British).
|
|
62
|
+
`speed` 0.5-1.6 sets pace.
|
|
63
|
+
- Other languages: set `voice:` per narration step, so one demo can mix them.
|
|
64
|
+
`say:Monica` uses a macOS system voice (~50 languages, `voila voices --all`);
|
|
65
|
+
`--tts-cmd 'engine -o {out} "{text}"'` plugs in any engine on any platform;
|
|
66
|
+
`audio: clip.mp3` on a step uses a file you already have.
|
|
60
67
|
- Sign-in walls: recording refuses to film a login page. Run `voila login <url>`
|
|
61
68
|
(or the voila_login tool), let the HUMAN sign in in the window that opens, and
|
|
62
69
|
the session persists in a local profile for every later recording.
|
package/README.md
CHANGED
|
@@ -28,9 +28,18 @@ voila record <url> --device mobile # iPhone-class viewport (portrait)
|
|
|
28
28
|
voila outline <url> # page structure for planning
|
|
29
29
|
voila review demo.mp4 --frames 12 # frames + recipe for self-review
|
|
30
30
|
voila serve # web UI
|
|
31
|
+
voila doctor # check + pre-download chromium and the voice model
|
|
32
|
+
voila fork demo.mp4 --url https://other.app # rebuild any voila demo from the recipe inside it
|
|
33
|
+
voila rerender <dir> --voice bf_emma # new voice, no re-recording (needs --keep-frames)
|
|
34
|
+
voila login https://app.example.com # you sign in; session saved locally
|
|
35
|
+
voila voices # 28 narration voices, graded
|
|
31
36
|
voila mcp # stdio MCP server
|
|
32
37
|
```
|
|
33
38
|
|
|
39
|
+
First run downloads Chromium (~150MB) and the voice model (~90MB) into
|
|
40
|
+
`~/.cache/voila`, so upgrades do not re-download. Consent banners are dismissed
|
|
41
|
+
before recording, preferring "reject" over "accept".
|
|
42
|
+
|
|
34
43
|
Devices: `desktop` (1280×800), `mobile` (390×844, touch + mobile UA),
|
|
35
44
|
`tablet` (834×1112). Cross-platform: verified on macOS and Linux (arm64
|
|
36
45
|
container); recording is headless-safe for CI.
|
|
@@ -64,6 +73,28 @@ on-device, no cloud, no API keys ([audio.js](audio.js)). Falls back to macOS
|
|
|
64
73
|
`say` if Kokoro can't load. Voices: `af_heart` (default), `af_bella`,
|
|
65
74
|
`am_adam`, … (`voice` param). Disable with `narrate: false` / `--no-narrate`.
|
|
66
75
|
|
|
76
|
+
**Other languages, and mixing them.** Kokoro's JS port speaks English only, so
|
|
77
|
+
non-English narration comes from one of two other engines, chosen per step:
|
|
78
|
+
|
|
79
|
+
```yaml
|
|
80
|
+
- action: hover
|
|
81
|
+
selector: h1
|
|
82
|
+
narration: "This part is English." # Kokoro
|
|
83
|
+
- action: scroll_to
|
|
84
|
+
selector: "#pricing"
|
|
85
|
+
voice: "say:Mónica" # macOS system voice
|
|
86
|
+
narration: "Esta parte está en español."
|
|
87
|
+
- action: wait
|
|
88
|
+
ms: 500
|
|
89
|
+
audio: ./clips/intro-ja.mp3 # a clip you already have
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
`voila voices --all` lists the ~180 system voices across ~50 languages on
|
|
93
|
+
macOS. On Linux and Windows use any engine you like:
|
|
94
|
+
`--tts-cmd 'piper -m es.onnx -f {out} -- "{text}"'` ({out}, {text}, {voice}).
|
|
95
|
+
Every step is still paced to its own spoken clip, so mixed-language demos stay
|
|
96
|
+
in sync.
|
|
97
|
+
|
|
67
98
|
## Recipes — demos as code
|
|
68
99
|
|
|
69
100
|
Every video ships with its source: `recipe.json` (URL + steps + narration +
|
package/RECIPE.md
CHANGED
|
@@ -51,6 +51,10 @@ Each step: `{ action, ...params, caption?, narration?, optional?, pause? }`
|
|
|
51
51
|
| `zoom` | `level` (1-3) or `selector`, `ms` | camera zoom; with a selector it frames that element |
|
|
52
52
|
| `wait` | `ms` | hold (cursor keeps breathing on long holds) |
|
|
53
53
|
|
|
54
|
+
Any step may also carry `voice:` (Kokoro id, `say:Name`, or an id for your
|
|
55
|
+
`--tts-cmd` engine) and `audio:` (a ready-made clip, skipping TTS entirely).
|
|
56
|
+
Because voice is per step, a recipe can mix languages.
|
|
57
|
+
|
|
54
58
|
`caption` renders as a lower-third; `narration` is spoken by on-device TTS and
|
|
55
59
|
**paces the segment** — the recording holds until the clip finishes, so a
|
|
56
60
|
recreated demo re-times itself to whatever voice regenerates it.
|
package/audio.js
CHANGED
|
@@ -22,6 +22,7 @@ async function getKokoro() {
|
|
|
22
22
|
// Must load the CJS build (exports map: require → dist/kokoro.cjs): it
|
|
23
23
|
// resolves bundled voice files via __dirname, while the ESM build loses
|
|
24
24
|
// __dirname and breaks when cwd isn't the package root.
|
|
25
|
+
require('./modelcache').useStableCache();
|
|
25
26
|
const { KokoroTTS } = require('kokoro-js');
|
|
26
27
|
kokoroInstance = await KokoroTTS.from_pretrained(
|
|
27
28
|
'onnx-community/Kokoro-82M-v1.0-ONNX', { dtype: 'q8' }
|
|
@@ -30,7 +31,7 @@ async function getKokoro() {
|
|
|
30
31
|
return kokoroInstance;
|
|
31
32
|
}
|
|
32
33
|
|
|
33
|
-
async function
|
|
34
|
+
async function _unusedSynthKokoro(texts, dir, voice, onStatus, speed = 1) {
|
|
34
35
|
// Fail loudly on a bad voice name rather than silently using the default.
|
|
35
36
|
if (voice && /^[a-z]{2}_/.test(voice) && !voices.isValid(voice)) {
|
|
36
37
|
throw new Error(`unknown voice "${voice}". Try: ${voices.suggest(voice).join(', ')} (run \`voila voices\` for all ${voices.ranked().length})`);
|
|
@@ -76,7 +77,7 @@ async function pickSayVoice(preferred) {
|
|
|
76
77
|
return 'Samantha';
|
|
77
78
|
}
|
|
78
79
|
|
|
79
|
-
async function
|
|
80
|
+
async function _unusedSynthSay(texts, dir, voice, onStatus) {
|
|
80
81
|
const v = await pickSayVoice(voice);
|
|
81
82
|
onStatus(`narrating with say (${v})`);
|
|
82
83
|
const clips = [];
|
|
@@ -89,25 +90,113 @@ async function synthSay(texts, dir, voice, onStatus) {
|
|
|
89
90
|
}
|
|
90
91
|
|
|
91
92
|
// --- pipeline entry ----------------------------------------------------------
|
|
93
|
+
// Each narration item is {text, voice?, audio?}. Voice decides the engine:
|
|
94
|
+
// af_heart -> Kokoro (English, best quality, cross platform)
|
|
95
|
+
// say:Monica -> a system voice (macOS; ~50 languages)
|
|
96
|
+
// (--tts-cmd set) -> any external engine, any language
|
|
97
|
+
// A step can also point at a ready-made file with `audio:`, which bypasses TTS.
|
|
98
|
+
// Because the voice is per item, one demo can mix languages.
|
|
99
|
+
|
|
100
|
+
function engineFor(voice, ttsCmd) {
|
|
101
|
+
// Supplying --tts-cmd means "use my engine", unless a step names a specific
|
|
102
|
+
// Kokoro or system voice, which then wins for that step.
|
|
103
|
+
if (!voice) {
|
|
104
|
+
if (ttsCmd) return 'cmd';
|
|
105
|
+
return process.env.VOILA_TTS === 'say' ? 'say' : 'kokoro';
|
|
106
|
+
}
|
|
107
|
+
if (String(voice).startsWith('cmd:') || (ttsCmd && String(voice).startsWith('cmd'))) return 'cmd';
|
|
108
|
+
if (String(voice).startsWith('say:')) return 'say';
|
|
109
|
+
if (voices.isValid(voice)) return 'kokoro';
|
|
110
|
+
if (voices.isSystemVoice(voice)) return 'say';
|
|
111
|
+
if (ttsCmd) return 'cmd';
|
|
112
|
+
throw new Error(
|
|
113
|
+
`unknown voice "${voice}". Kokoro (English): ${voices.suggest(voice).join(', ')}. ` +
|
|
114
|
+
`For other languages use a system voice like "say:Monica" (see \`voila voices --all\`) ` +
|
|
115
|
+
`or supply --tts-cmd for your own engine.`
|
|
116
|
+
);
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
async function sayOne(text, file, voiceName, speed) {
|
|
120
|
+
const rate = Math.round(185 * (speed || 1));
|
|
121
|
+
await run('say', ['-v', voiceName, '-r', String(rate), '-o', file, text]);
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
async function cmdOne(text, file, voiceName, ttsCmd) {
|
|
125
|
+
const cmd = ttsCmd
|
|
126
|
+
.replaceAll('{text}', text.replace(/"/g, '\\"'))
|
|
127
|
+
.replaceAll('{out}', file)
|
|
128
|
+
.replaceAll('{voice}', voiceName || '');
|
|
129
|
+
await new Promise((res, rej) => {
|
|
130
|
+
require('child_process').exec(cmd, { maxBuffer: 1e7 }, (err, _o, se) =>
|
|
131
|
+
err ? rej(new Error(`tts-cmd failed: ${String(se || err.message).slice(0, 200)}`)) : res());
|
|
132
|
+
});
|
|
133
|
+
if (!fs.existsSync(file)) throw new Error(`tts-cmd produced no file at ${file}`);
|
|
134
|
+
}
|
|
92
135
|
|
|
93
|
-
// Synthesize narration
|
|
94
|
-
// spoken durations. Returns {clips:
|
|
95
|
-
async function prepareNarration(
|
|
136
|
+
// Synthesize every narration clip up front so the recorder can pace segments
|
|
137
|
+
// to real spoken durations. Returns {clips:[{file,durMs}], voice, backend}.
|
|
138
|
+
async function prepareNarration(items, dir, defaultVoice, onStatus = () => {}, speed = 1, ttsCmd = null) {
|
|
96
139
|
fs.mkdirSync(dir, { recursive: true });
|
|
97
|
-
const
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
140
|
+
const norm = items.map(it => (typeof it === 'string' ? { text: it } : it));
|
|
141
|
+
const clips = [];
|
|
142
|
+
const used = new Set();
|
|
143
|
+
let kokoro = null;
|
|
144
|
+
|
|
145
|
+
for (let i = 0; i < norm.length; i++) {
|
|
146
|
+
const it = norm[i];
|
|
147
|
+
const voice = it.voice || defaultVoice || null;
|
|
148
|
+
|
|
149
|
+
// A pre-made audio file wins over any engine.
|
|
150
|
+
if (it.audio) {
|
|
151
|
+
if (!fs.existsSync(it.audio)) throw new Error(`audio file not found: ${it.audio}`);
|
|
152
|
+
clips.push({ file: it.audio, durMs: await ffDurationMs(it.audio) });
|
|
153
|
+
used.add('file');
|
|
154
|
+
continue;
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
const engine = engineFor(voice, ttsCmd);
|
|
158
|
+
const ext = engine === 'kokoro' ? 'wav' : engine === 'say' ? 'aiff' : 'wav';
|
|
159
|
+
const file = path.join(dir, `seg${i}.${ext}`);
|
|
160
|
+
|
|
161
|
+
if (engine === 'kokoro') {
|
|
162
|
+
if (!kokoro) {
|
|
163
|
+
onStatus('loading Kokoro TTS');
|
|
164
|
+
require('./modelcache').useStableCache();
|
|
165
|
+
const { KokoroTTS } = require('kokoro-js');
|
|
166
|
+
kokoro = await KokoroTTS.from_pretrained('onnx-community/Kokoro-82M-v1.0-ONNX', { dtype: 'q8' });
|
|
167
|
+
}
|
|
168
|
+
const v = voice && voices.isValid(voice) ? voice : 'af_heart';
|
|
169
|
+
onStatus(`narrating ${i + 1}/${norm.length} with Kokoro (${v})`);
|
|
170
|
+
const audio = await kokoro.generate(it.text, { voice: v, speed });
|
|
171
|
+
await audio.save(file);
|
|
172
|
+
clips.push({
|
|
173
|
+
file,
|
|
174
|
+
durMs: audio.audio && audio.sampling_rate
|
|
175
|
+
? Math.round((audio.audio.length / audio.sampling_rate) * 1000)
|
|
176
|
+
: await ffDurationMs(file),
|
|
177
|
+
});
|
|
178
|
+
used.add(`kokoro:${v}`);
|
|
179
|
+
} else if (engine === 'say') {
|
|
180
|
+
if (process.platform !== 'darwin') {
|
|
181
|
+
throw new Error(`system voices need macOS. Use a Kokoro voice for English, or --tts-cmd on this platform.`);
|
|
182
|
+
}
|
|
183
|
+
const name = String(voice).replace(/^say:/, '');
|
|
184
|
+
onStatus(`narrating ${i + 1}/${norm.length} with system voice (${name})`);
|
|
185
|
+
await sayOne(it.text, file, name, speed);
|
|
186
|
+
clips.push({ file, durMs: await ffDurationMs(file) });
|
|
187
|
+
used.add(`say:${name}`);
|
|
188
|
+
} else {
|
|
189
|
+
onStatus(`narrating ${i + 1}/${norm.length} with tts-cmd`);
|
|
190
|
+
await cmdOne(it.text, file, String(voice || '').replace(/^cmd:/, ''), ttsCmd);
|
|
191
|
+
clips.push({ file, durMs: await ffDurationMs(file) });
|
|
192
|
+
used.add('cmd');
|
|
104
193
|
}
|
|
105
194
|
}
|
|
106
|
-
|
|
107
|
-
|
|
195
|
+
|
|
196
|
+
return { clips, voice: [...used].join(', ') || 'none', backend: [...used].join(', ') };
|
|
108
197
|
}
|
|
109
198
|
|
|
110
|
-
async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1, prepared = null, onStatus = () => {} } = {}) {
|
|
199
|
+
async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1, ttsCmd = null, prepared = null, onStatus = () => {} } = {}) {
|
|
111
200
|
const segs = (meta.segments || []).filter(s => s.narration);
|
|
112
201
|
if (!segs.length) {
|
|
113
202
|
fs.copyFileSync(videoIn, videoOut);
|
|
@@ -119,7 +208,9 @@ async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1,
|
|
|
119
208
|
let synth = prepared && prepared.clips.length === segs.length ? prepared : null;
|
|
120
209
|
if (!synth) {
|
|
121
210
|
try {
|
|
122
|
-
synth = await prepareNarration(
|
|
211
|
+
synth = await prepareNarration(
|
|
212
|
+
segs.map(s => ({ text: s.narration, voice: s.voice, audio: s.audio })),
|
|
213
|
+
dir, voice, onStatus, speed, ttsCmd);
|
|
123
214
|
} catch (e) {
|
|
124
215
|
onStatus(`narration skipped: ${e.message}`);
|
|
125
216
|
fs.copyFileSync(videoIn, videoOut);
|
|
@@ -153,7 +244,7 @@ async function addNarration(meta, videoIn, videoOut, { voice = null, speed = 1,
|
|
|
153
244
|
ff.on('error', rej);
|
|
154
245
|
});
|
|
155
246
|
|
|
156
|
-
fs.rmSync(dir, { recursive: true, force: true });
|
|
247
|
+
fs.rmSync(dir, { recursive: true, force: true }); // only generated clips live here
|
|
157
248
|
return { narrated: true, voice: synth.voice, backend: synth.backend, segments: clips.length };
|
|
158
249
|
}
|
|
159
250
|
|
package/cli.js
CHANGED
|
@@ -18,11 +18,14 @@ function arg(name, fallback = null) {
|
|
|
18
18
|
}
|
|
19
19
|
|
|
20
20
|
const USAGE = `usage:
|
|
21
|
+
voila doctor (check + download everything voila needs)
|
|
21
22
|
voila outline <url> [--device desktop|mobile|tablet]
|
|
22
|
-
voila record <url> [--steps f.yaml] [--device mobile] [--voice name] [--speed 1] [--no-narrate] [--headful] [--out dir] [--profile dir]
|
|
23
|
+
voila record <url> [--steps f.yaml] [--device mobile] [--voice name] [--speed 1] [--tts-cmd tmpl] [--no-narrate] [--headful] [--keep-frames] [--no-dismiss] [--dismiss sel] [--out dir] [--profile dir]
|
|
23
24
|
voila review <video.mp4> [--frames 12] [--out dir]
|
|
24
25
|
voila login <url> [--profile dir] (sign in yourself; session is saved locally)
|
|
25
|
-
voila voices
|
|
26
|
+
voila voices [--all] (Kokoro voices; --all adds system voices for other languages)
|
|
27
|
+
voila fork <video.mp4> [--url u] [--voice v] [--print] [--out dir]
|
|
28
|
+
voila rerender <dir> [--voice v] [--speed n] (needs --keep-frames on the original)
|
|
26
29
|
voila skill (install the voila skill into ~/.claude/skills)
|
|
27
30
|
voila serve (web UI, PORT env or --port)
|
|
28
31
|
voila mcp (stdio MCP server)`;
|
|
@@ -39,8 +42,16 @@ const USAGE = `usage:
|
|
|
39
42
|
require('./mcp');
|
|
40
43
|
return;
|
|
41
44
|
}
|
|
45
|
+
if (cmd === 'doctor' || cmd === '--version' || cmd === '-v') {
|
|
46
|
+
if (cmd !== 'doctor') { console.log(require('./package.json').version); return; }
|
|
47
|
+
console.error(`voila ${require('./package.json').version}\n`);
|
|
48
|
+
const r = await require('./doctor').doctor({ fix: !process.argv.includes('--check') });
|
|
49
|
+
process.exit(r.ready ? 0 : 1);
|
|
50
|
+
}
|
|
42
51
|
if (cmd === 'voices') {
|
|
43
|
-
|
|
52
|
+
const v = require('./voices');
|
|
53
|
+
console.log(process.argv.includes('--all') ? v.formatAll() : v.format());
|
|
54
|
+
if (!process.argv.includes('--all')) console.log('\nNon-English? run: voila voices --all');
|
|
44
55
|
return;
|
|
45
56
|
}
|
|
46
57
|
if (cmd === 'skill') {
|
|
@@ -68,6 +79,61 @@ const USAGE = `usage:
|
|
|
68
79
|
return;
|
|
69
80
|
}
|
|
70
81
|
|
|
82
|
+
if (cmd === 'fork') {
|
|
83
|
+
// The recipe inside a voila MP4 is its source. Turn it back into a script
|
|
84
|
+
// and, unless --print, record it again (optionally against another URL).
|
|
85
|
+
const video = process.argv[3];
|
|
86
|
+
if (!video) { console.error(USAGE); process.exit(1); }
|
|
87
|
+
const { reviewDemo } = require('./review');
|
|
88
|
+
const info = await reviewDemo(video, { count: 3 });
|
|
89
|
+
if (!info.recipe || !info.recipe.steps) {
|
|
90
|
+
console.error('FAILED: no voila recipe found in that file (was it made by voila?)');
|
|
91
|
+
process.exit(1);
|
|
92
|
+
}
|
|
93
|
+
const stepsYaml = yaml.dump(info.recipe.steps);
|
|
94
|
+
if (process.argv.includes('--print')) { console.log(stepsYaml); return; }
|
|
95
|
+
|
|
96
|
+
const target = arg('--url', info.recipe.url);
|
|
97
|
+
const workDir = arg('--out', path.join(__dirname, 'recordings', `fork-${Date.now()}`));
|
|
98
|
+
fs.mkdirSync(workDir, { recursive: true });
|
|
99
|
+
fs.writeFileSync(path.join(workDir, 'steps.yaml'), stepsYaml);
|
|
100
|
+
console.error(`[voila] forking ${info.recipe.steps.length} steps onto ${target}`);
|
|
101
|
+
|
|
102
|
+
const { VoilaSession } = require('./recorder');
|
|
103
|
+
const { produceDemo } = require('./pipeline');
|
|
104
|
+
const s2 = new VoilaSession({
|
|
105
|
+
headless: !process.argv.includes('--headful'),
|
|
106
|
+
device: arg('--device', 'desktop'),
|
|
107
|
+
profileDir: arg('--profile', path.join(__dirname, 'profile')),
|
|
108
|
+
});
|
|
109
|
+
try {
|
|
110
|
+
const r = await produceDemo(s2, {
|
|
111
|
+
url: target, steps: yaml.load(stepsYaml), workDir,
|
|
112
|
+
narrate: !process.argv.includes('--no-narrate'),
|
|
113
|
+
voice: arg('--voice'), speed: Number(arg('--speed', '1')) || 1,
|
|
114
|
+
keepFrames: process.argv.includes('--keep-frames'),
|
|
115
|
+
onStatus: m => console.error('[voila]', m),
|
|
116
|
+
});
|
|
117
|
+
console.log(r.video);
|
|
118
|
+
} finally { await s2.close(); }
|
|
119
|
+
return;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
if (cmd === 'rerender') {
|
|
123
|
+
const dir = process.argv[3];
|
|
124
|
+
if (!dir) { console.error(USAGE); process.exit(1); }
|
|
125
|
+
const { rerender } = require('./pipeline');
|
|
126
|
+
const r = await rerender(dir, {
|
|
127
|
+
voice: arg('--voice'),
|
|
128
|
+
speed: Number(arg('--speed', '1')) || 1,
|
|
129
|
+
ttsCmd: arg('--tts-cmd'),
|
|
130
|
+
narrate: !process.argv.includes('--no-narrate'),
|
|
131
|
+
onStatus: m => console.error('[voila]', m),
|
|
132
|
+
});
|
|
133
|
+
console.log(r.video);
|
|
134
|
+
return;
|
|
135
|
+
}
|
|
136
|
+
|
|
71
137
|
const url = process.argv[3];
|
|
72
138
|
if (!cmd || !url || !['outline', 'record', 'login'].includes(cmd)) {
|
|
73
139
|
console.error(USAGE);
|
|
@@ -80,6 +146,8 @@ const USAGE = `usage:
|
|
|
80
146
|
headless: !process.argv.includes('--headful'),
|
|
81
147
|
device: arg('--device', 'desktop'),
|
|
82
148
|
profileDir: arg('--profile', path.join(__dirname, 'profile')),
|
|
149
|
+
dismiss: !process.argv.includes('--no-dismiss'),
|
|
150
|
+
dismissSelector: arg('--dismiss'),
|
|
83
151
|
});
|
|
84
152
|
|
|
85
153
|
try {
|
|
@@ -101,6 +169,8 @@ const USAGE = `usage:
|
|
|
101
169
|
narrate: !process.argv.includes('--no-narrate'),
|
|
102
170
|
voice: arg('--voice'),
|
|
103
171
|
speed: Number(arg('--speed', '1')) || 1,
|
|
172
|
+
ttsCmd: arg('--tts-cmd'),
|
|
173
|
+
keepFrames: process.argv.includes('--keep-frames'),
|
|
104
174
|
onStatus: s => console.error('[voila]', s),
|
|
105
175
|
});
|
|
106
176
|
console.log(result.video);
|
package/consent.js
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
// Cookie and consent banners ruin a demo: they sit in frame for the whole
|
|
2
|
+
// recording. Dismiss them before the camera rolls.
|
|
3
|
+
//
|
|
4
|
+
// Privacy first: we look for "reject" / "necessary only" before "accept", so
|
|
5
|
+
// the recorded session declines non-essential cookies wherever that choice
|
|
6
|
+
// exists. Accepting is only a last resort to clear the overlay.
|
|
7
|
+
|
|
8
|
+
const REJECT = [
|
|
9
|
+
/^(reject|decline|refuse)( all)?$/i,
|
|
10
|
+
/necessary (cookies )?only/i,
|
|
11
|
+
/^only (essential|necessary|required)/i,
|
|
12
|
+
/^(essential|required) (cookies )?only/i,
|
|
13
|
+
/continue without accepting/i,
|
|
14
|
+
/^reject non-essential/i,
|
|
15
|
+
];
|
|
16
|
+
|
|
17
|
+
const ACCEPT = [
|
|
18
|
+
/^(accept|allow|agree)( all| cookies)?$/i,
|
|
19
|
+
/^(ok|got it|i understand|understood)$/i,
|
|
20
|
+
/^(dismiss|close)$/i,
|
|
21
|
+
];
|
|
22
|
+
|
|
23
|
+
// Runs in the page. Returns the label it clicked, or null.
|
|
24
|
+
const dismissInPage = ([rejectSrc, acceptSrc]) => {
|
|
25
|
+
const toRe = arr => arr.map(([s, f]) => new RegExp(s, f));
|
|
26
|
+
const reject = toRe(rejectSrc), accept = toRe(acceptSrc);
|
|
27
|
+
|
|
28
|
+
const visible = el => {
|
|
29
|
+
const r = el.getBoundingClientRect();
|
|
30
|
+
const cs = getComputedStyle(el);
|
|
31
|
+
return r.width > 20 && r.height > 12 && cs.visibility !== 'hidden' && cs.opacity !== '0';
|
|
32
|
+
};
|
|
33
|
+
|
|
34
|
+
// Only consider controls that live inside something banner-shaped, so we
|
|
35
|
+
// never click an "Accept" button that is part of the product itself.
|
|
36
|
+
const looksLikeBanner = el => {
|
|
37
|
+
const box = el.closest('[class*="cookie" i],[id*="cookie" i],[class*="consent" i],[id*="consent" i],[class*="gdpr" i],[id*="gdpr" i],[aria-label*="cookie" i],[role="dialog"],[class*="banner" i]');
|
|
38
|
+
if (box) return true;
|
|
39
|
+
// Or a fixed-position bar pinned to an edge of the viewport.
|
|
40
|
+
let n = el;
|
|
41
|
+
for (let i = 0; i < 6 && n; i++, n = n.parentElement) {
|
|
42
|
+
const cs = getComputedStyle(n);
|
|
43
|
+
if (cs.position === 'fixed' || cs.position === 'sticky') {
|
|
44
|
+
const r = n.getBoundingClientRect();
|
|
45
|
+
if (r.width > innerWidth * 0.5 && (r.bottom > innerHeight * 0.6 || r.top < innerHeight * 0.4)) return true;
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
return false;
|
|
49
|
+
};
|
|
50
|
+
|
|
51
|
+
const controls = [...document.querySelectorAll('button,a[role="button"],[role="button"],input[type="button"],input[type="submit"]')]
|
|
52
|
+
.filter(visible).filter(looksLikeBanner);
|
|
53
|
+
|
|
54
|
+
for (const patterns of [reject, accept]) {
|
|
55
|
+
for (const el of controls) {
|
|
56
|
+
const label = (el.innerText || el.value || el.getAttribute('aria-label') || '').trim();
|
|
57
|
+
if (!label || label.length > 40) continue;
|
|
58
|
+
if (patterns.some(re => re.test(label))) { el.click(); return label; }
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
return null;
|
|
62
|
+
};
|
|
63
|
+
|
|
64
|
+
async function dismissConsent(page, { selector = null, timeout = 2500 } = {}) {
|
|
65
|
+
const results = [];
|
|
66
|
+
if (selector) {
|
|
67
|
+
const el = page.locator(selector).first();
|
|
68
|
+
try {
|
|
69
|
+
await el.waitFor({ state: 'visible', timeout });
|
|
70
|
+
await el.click({ timeout: 2000 });
|
|
71
|
+
results.push(`custom: ${selector}`);
|
|
72
|
+
} catch { /* nothing matched the override */ }
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
const src = [
|
|
76
|
+
REJECT.map(r => [r.source, r.flags]),
|
|
77
|
+
ACCEPT.map(r => [r.source, r.flags]),
|
|
78
|
+
];
|
|
79
|
+
|
|
80
|
+
// Banners often mount late, and some sites stack two of them.
|
|
81
|
+
for (let attempt = 0; attempt < 3; attempt++) {
|
|
82
|
+
let clicked = null;
|
|
83
|
+
try {
|
|
84
|
+
clicked = await page.evaluate(dismissInPage, src);
|
|
85
|
+
for (const frame of page.frames()) {
|
|
86
|
+
if (clicked || frame === page.mainFrame()) continue;
|
|
87
|
+
clicked = await frame.evaluate(dismissInPage, src).catch(() => null);
|
|
88
|
+
}
|
|
89
|
+
} catch { /* page navigated mid-check */ }
|
|
90
|
+
if (clicked) results.push(clicked);
|
|
91
|
+
await page.waitForTimeout(attempt === 0 ? 700 : 500);
|
|
92
|
+
if (!clicked && attempt > 0) break;
|
|
93
|
+
}
|
|
94
|
+
return results;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
module.exports = { dismissConsent };
|
package/doctor.js
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
// Pre-flight: tell people what voila needs, what is already on disk, and
|
|
2
|
+
// download the rest with visible progress. The first run used to be several
|
|
3
|
+
// silent minutes, which reads as a hang.
|
|
4
|
+
|
|
5
|
+
const fs = require('fs');
|
|
6
|
+
const path = require('path');
|
|
7
|
+
const os = require('os');
|
|
8
|
+
const { execFile, execFileSync } = require('child_process');
|
|
9
|
+
|
|
10
|
+
const MB = n => `${(n / 1e6).toFixed(0)}MB`;
|
|
11
|
+
|
|
12
|
+
function chromiumPath() {
|
|
13
|
+
try {
|
|
14
|
+
const { chromium } = require('playwright');
|
|
15
|
+
return chromium.executablePath();
|
|
16
|
+
} catch { return null; }
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
function chromiumReady() {
|
|
20
|
+
const p = chromiumPath();
|
|
21
|
+
return !!(p && fs.existsSync(p));
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
// transformers.js caches models under ~/.cache/huggingface by default.
|
|
25
|
+
function kokoroCacheDir() {
|
|
26
|
+
return require('./modelcache').CACHE_DIR;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
function kokoroReady() {
|
|
30
|
+
const dir = kokoroCacheDir();
|
|
31
|
+
if (!fs.existsSync(dir)) return false;
|
|
32
|
+
const hit = [];
|
|
33
|
+
const walk = (d, depth = 0) => {
|
|
34
|
+
if (depth > 4 || hit.length) return;
|
|
35
|
+
for (const e of fs.readdirSync(d, { withFileTypes: true })) {
|
|
36
|
+
if (hit.length) return;
|
|
37
|
+
const full = path.join(d, e.name);
|
|
38
|
+
if (e.isDirectory()) walk(full, depth + 1);
|
|
39
|
+
else if (/\.onnx(_data)?$/.test(e.name) && fs.statSync(full).size > 5e6) hit.push(full);
|
|
40
|
+
}
|
|
41
|
+
};
|
|
42
|
+
try { walk(dir); } catch { /* unreadable cache */ }
|
|
43
|
+
return hit.length > 0;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
function ffmpegReady() {
|
|
47
|
+
try { return fs.existsSync(require('ffmpeg-static')); } catch { return false; }
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
// Install Chromium with its progress bar visible instead of swallowed.
|
|
51
|
+
function installChromium({ quiet = false } = {}) {
|
|
52
|
+
let cliPath;
|
|
53
|
+
try { cliPath = require.resolve('playwright/cli'); }
|
|
54
|
+
catch { cliPath = path.join(path.dirname(require.resolve('playwright')), 'cli.js'); }
|
|
55
|
+
execFileSync(process.execPath, [cliPath, 'install', 'chromium'], {
|
|
56
|
+
stdio: quiet ? 'pipe' : ['ignore', 'inherit', 'inherit'],
|
|
57
|
+
timeout: 900000,
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
async function warmKokoro(onStatus = () => {}) {
|
|
62
|
+
require('./modelcache').useStableCache();
|
|
63
|
+
const { KokoroTTS } = require('kokoro-js');
|
|
64
|
+
let lastPct = -5;
|
|
65
|
+
const tts = await KokoroTTS.from_pretrained('onnx-community/Kokoro-82M-v1.0-ONNX', {
|
|
66
|
+
dtype: 'q8',
|
|
67
|
+
progress_callback: p => {
|
|
68
|
+
if (p.status === 'download') return onStatus(`fetching ${p.file}`);
|
|
69
|
+
if (p.status !== 'progress') return;
|
|
70
|
+
// Hugging Face omits content-length on some files, so fall back to bytes.
|
|
71
|
+
if (typeof p.total === 'number' && p.total > 0) {
|
|
72
|
+
const pct = Math.floor((p.progress || 0) / 5) * 5;
|
|
73
|
+
if (pct > lastPct) { lastPct = pct; onStatus(`${p.file || 'model'} ${pct}% of ${MB(p.total)}`); }
|
|
74
|
+
} else if (typeof p.loaded === 'number') {
|
|
75
|
+
const step = Math.floor(p.loaded / 1e7);
|
|
76
|
+
if (step > lastPct) { lastPct = step; onStatus(`${p.file || 'model'} ${MB(p.loaded)} downloaded`); }
|
|
77
|
+
}
|
|
78
|
+
},
|
|
79
|
+
});
|
|
80
|
+
// Force one tiny synthesis so the voice files are fetched too.
|
|
81
|
+
await tts.generate('Ready.', { voice: 'af_heart' });
|
|
82
|
+
return true;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
async function doctor({ fix = true, log = console.error } = {}) {
|
|
86
|
+
const nodeOk = Number(process.versions.node.split('.')[0]) >= 20;
|
|
87
|
+
log(`node ${process.versions.node} ${nodeOk ? 'ok' : 'TOO OLD, voila needs >= 20'}`);
|
|
88
|
+
log(`ffmpeg ${ffmpegReady() ? 'bundled, ok' : 'MISSING (reinstall voila-recorder)'}`);
|
|
89
|
+
|
|
90
|
+
let chrome = chromiumReady();
|
|
91
|
+
log(`chromium ${chrome ? 'installed' : 'not installed (~150MB download)'}`);
|
|
92
|
+
if (!chrome && fix) {
|
|
93
|
+
log('\ndownloading chromium...');
|
|
94
|
+
installChromium();
|
|
95
|
+
chrome = chromiumReady();
|
|
96
|
+
log(`chromium ${chrome ? 'installed' : 'FAILED'}`);
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
let voice = kokoroReady();
|
|
100
|
+
log(`voice model ${voice ? `cached in ${kokoroCacheDir()}` : 'not cached (~90MB download, first narration only)'}`);
|
|
101
|
+
if (!voice && fix) {
|
|
102
|
+
log('\nfetching the voice model...');
|
|
103
|
+
try {
|
|
104
|
+
await warmKokoro(m => log(` ${m}`));
|
|
105
|
+
voice = true;
|
|
106
|
+
log('voice model cached');
|
|
107
|
+
} catch (e) {
|
|
108
|
+
log(`voice model FAILED: ${e.message.slice(0, 120)}`);
|
|
109
|
+
log('(recording still works with --no-narrate)');
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
const ready = nodeOk && ffmpegReady() && chrome;
|
|
114
|
+
log(`\n${ready ? 'voila is ready. Try: voila record https://example.com' : 'voila is not ready yet, see above.'}`);
|
|
115
|
+
return { nodeOk, ffmpeg: ffmpegReady(), chromium: chrome, voice, ready };
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
module.exports = { doctor, chromiumReady, kokoroReady, installChromium, warmKokoro };
|
package/mcp.js
CHANGED
|
@@ -37,7 +37,7 @@ function getSession(device) {
|
|
|
37
37
|
return sessions.get(key);
|
|
38
38
|
}
|
|
39
39
|
|
|
40
|
-
const server = new McpServer({ name: 'voila', version: '0.
|
|
40
|
+
const server = new McpServer({ name: 'voila', version: '0.7.0' });
|
|
41
41
|
const deviceParam = z.enum(['desktop', 'mobile', 'tablet']).optional().default('desktop');
|
|
42
42
|
|
|
43
43
|
server.tool(
|
|
@@ -61,6 +61,7 @@ server.tool(
|
|
|
61
61
|
'caption is burned into the video as a lower-third; narration is spoken via on-device TTS (Kokoro) at that step, ' +
|
|
62
62
|
'and segment pacing automatically stretches to fit each narration clip — no need to pad waits. ' +
|
|
63
63
|
'Steps marked optional:true are skipped on failure instead of aborting. ' +
|
|
64
|
+
'A step may set its own voice: (mixing languages within one demo) or audio: (a ready-made clip). ' +
|
|
64
65
|
'device selects the recorded viewport (mobile emulates an iPhone-class device). ' +
|
|
65
66
|
'On failure the error names the failing step and includes the live page outline — patch the steps and retry. ' +
|
|
66
67
|
'Returns the MP4 path, the recipe path, and any warnings.',
|
|
@@ -68,16 +69,17 @@ server.tool(
|
|
|
68
69
|
url: z.string().url(),
|
|
69
70
|
steps_yaml: z.string().optional(),
|
|
70
71
|
narrate: z.boolean().optional().default(true),
|
|
71
|
-
voice: z.string().optional().describe('narration voice
|
|
72
|
+
voice: z.string().optional().describe('default narration voice. Kokoro is ENGLISH ONLY (af_heart, af_bella, bf_emma). For other languages use a macOS system voice ("say:Monica") or set tts_cmd. Per-step `voice:` overrides this, so one demo can mix languages.'),
|
|
72
73
|
speed: z.number().min(0.5).max(1.6).optional().default(1).describe('narration speed; 0.9 reads calmer'),
|
|
74
|
+
tts_cmd: z.string().optional().describe('external TTS engine template for any language/platform, e.g. \'piper -m es.onnx -f {out} -- "{text}"\'. Placeholders: {out} {text} {voice}.'),
|
|
73
75
|
device: deviceParam,
|
|
74
76
|
},
|
|
75
|
-
async ({ url, steps_yaml, narrate, voice, speed, device }) => enqueue(async () => {
|
|
77
|
+
async ({ url, steps_yaml, narrate, voice, speed, tts_cmd, device }) => enqueue(async () => {
|
|
76
78
|
const steps = steps_yaml ? yaml.load(steps_yaml) : null;
|
|
77
79
|
const workDir = path.join(__dirname, 'recordings', `mcp-${Date.now()}`);
|
|
78
80
|
fs.mkdirSync(workDir, { recursive: true });
|
|
79
81
|
const result = await produceDemo(getSession(device), {
|
|
80
|
-
url, steps, workDir, narrate, voice: voice || null, speed,
|
|
82
|
+
url, steps, workDir, narrate, voice: voice || null, speed, ttsCmd: tts_cmd || null,
|
|
81
83
|
onStatus: () => {},
|
|
82
84
|
});
|
|
83
85
|
return {
|
|
@@ -116,11 +118,19 @@ server.tool(
|
|
|
116
118
|
|
|
117
119
|
server.tool(
|
|
118
120
|
'voila_voices',
|
|
119
|
-
'List
|
|
120
|
-
'
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
121
|
+
'List narration voices. Kokoro voices are English only, with quality grades. Pass system:true to ' +
|
|
122
|
+
'also get the machine\'s system voices, which is how you narrate other languages (macOS only; on ' +
|
|
123
|
+
'Linux or Windows use tts_cmd instead). Use before voila_record when the user asks for a different ' +
|
|
124
|
+
'voice, an accent, a male or female narrator, or a non-English language.',
|
|
125
|
+
{ system: z.boolean().optional().default(false) },
|
|
126
|
+
async ({ system }) => ({
|
|
127
|
+
content: [{ type: 'text', text: JSON.stringify({
|
|
128
|
+
kokoro: voiceCatalogue.ranked(),
|
|
129
|
+
englishOnly: true,
|
|
130
|
+
system: system ? voiceCatalogue.systemVoices() : undefined,
|
|
131
|
+
systemLanguages: system ? Object.keys(voiceCatalogue.systemLanguages()) : undefined,
|
|
132
|
+
note: 'Non-English: use a system voice (say:Name) on macOS, or tts_cmd on any platform.',
|
|
133
|
+
}, null, 2) }],
|
|
124
134
|
})
|
|
125
135
|
);
|
|
126
136
|
|
package/modelcache.js
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
// transformers.js caches models inside its own node_modules folder by default,
|
|
2
|
+
// so every voila upgrade (and every fresh npx hash) would re-download ~90MB.
|
|
3
|
+
// Pin the cache to a stable per-user directory instead.
|
|
4
|
+
|
|
5
|
+
const os = require('os');
|
|
6
|
+
const path = require('path');
|
|
7
|
+
|
|
8
|
+
const CACHE_DIR = process.env.VOILA_MODEL_DIR
|
|
9
|
+
|| path.join(os.homedir(), '.cache', 'voila', 'models');
|
|
10
|
+
|
|
11
|
+
let applied = false;
|
|
12
|
+
function useStableCache() {
|
|
13
|
+
if (applied) return CACHE_DIR;
|
|
14
|
+
try {
|
|
15
|
+
const { env } = require('@huggingface/transformers');
|
|
16
|
+
env.cacheDir = CACHE_DIR;
|
|
17
|
+
applied = true;
|
|
18
|
+
} catch { /* transformers not resolvable; kokoro will use its default */ }
|
|
19
|
+
return CACHE_DIR;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
module.exports = { useStableCache, CACHE_DIR };
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "voila-recorder",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.7.0",
|
|
4
4
|
"description": "Permission-free product demo recorder: URL in, narrated auto-zoomed MP4 out — with the recipe embedded in the video. Agent-native (MCP), fully on-device.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"main": "pipeline.js",
|
|
@@ -18,13 +18,17 @@
|
|
|
18
18
|
"render.js",
|
|
19
19
|
"audio.js",
|
|
20
20
|
"review.js",
|
|
21
|
+
"modelcache.js",
|
|
22
|
+
"consent.js",
|
|
23
|
+
"doctor.js",
|
|
21
24
|
"voices.js",
|
|
22
25
|
"auth.js",
|
|
23
26
|
"public/",
|
|
24
27
|
"skills/",
|
|
25
28
|
"AGENTS.md",
|
|
26
29
|
"README.md",
|
|
27
|
-
"RECIPE.md"
|
|
30
|
+
"RECIPE.md",
|
|
31
|
+
"scripts/"
|
|
28
32
|
],
|
|
29
33
|
"repository": {
|
|
30
34
|
"type": "git",
|
package/pipeline.js
CHANGED
|
@@ -44,15 +44,16 @@ function embedRecipe(videoIn, videoOut, recipe) {
|
|
|
44
44
|
});
|
|
45
45
|
}
|
|
46
46
|
|
|
47
|
-
async function produceDemo(session, { url, mode = 'auto', steps = null, workDir, voice = null, speed = 1, narrate = true, onStatus = () => {} }) {
|
|
47
|
+
async function produceDemo(session, { url, mode = 'auto', steps = null, workDir, voice = null, speed = 1, ttsCmd = null, narrate = true, keepFrames = false, onStatus = () => {} }) {
|
|
48
48
|
// Steps mode: synthesize narration BEFORE recording so segment pacing and
|
|
49
49
|
// caption lifetimes match the spoken clip durations exactly.
|
|
50
50
|
let prepared = null;
|
|
51
51
|
if (steps && narrate) {
|
|
52
|
-
const
|
|
53
|
-
|
|
52
|
+
const items = steps.filter(s => s.narration)
|
|
53
|
+
.map(s => ({ text: s.narration, voice: s.voice, audio: s.audio }));
|
|
54
|
+
if (items.length) {
|
|
54
55
|
try {
|
|
55
|
-
prepared = await prepareNarration(
|
|
56
|
+
prepared = await prepareNarration(items, path.join(workDir, 'tts'), voice, onStatus, speed, ttsCmd);
|
|
56
57
|
let i = 0;
|
|
57
58
|
for (const s of steps) if (s.narration) s._narrDurMs = prepared.clips[i++].durMs;
|
|
58
59
|
} catch (e) {
|
|
@@ -69,7 +70,7 @@ async function produceDemo(session, { url, mode = 'auto', steps = null, workDir,
|
|
|
69
70
|
await render(meta, raw, { onStatus });
|
|
70
71
|
|
|
71
72
|
let narration = { narrated: false };
|
|
72
|
-
if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, prepared, onStatus });
|
|
73
|
+
if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, ttsCmd, prepared, onStatus });
|
|
73
74
|
else fs.copyFileSync(raw, narrated);
|
|
74
75
|
|
|
75
76
|
const recipe = buildRecipe({ url, mode, steps, meta });
|
|
@@ -77,10 +78,40 @@ async function produceDemo(session, { url, mode = 'auto', steps = null, workDir,
|
|
|
77
78
|
onStatus('embedding recipe');
|
|
78
79
|
await embedRecipe(narrated, out, recipe);
|
|
79
80
|
|
|
80
|
-
|
|
81
|
+
// Frames are the expensive part to recreate: keeping them lets `voila
|
|
82
|
+
// rerender` change the voice, speed or captions in seconds instead of
|
|
83
|
+
// re-driving the browser. They are large, so it is opt-in.
|
|
84
|
+
if (!keepFrames) fs.rmSync(meta.framesDir, { recursive: true, force: true });
|
|
81
85
|
fs.rmSync(raw, { force: true });
|
|
82
86
|
fs.rmSync(narrated, { force: true });
|
|
83
|
-
return { video: out, recipe: path.join(workDir, 'recipe.json'), meta, narration };
|
|
87
|
+
return { video: out, recipe: path.join(workDir, 'recipe.json'), meta, narration, framesKept: keepFrames };
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
// Re-produce the video from frames already on disk: no browser, no re-driving
|
|
91
|
+
// the page. Used to swap the narration voice or speed after the fact.
|
|
92
|
+
async function rerender(workDir, { voice = null, speed = 1, ttsCmd = null, narrate = true, onStatus = () => {} } = {}) {
|
|
93
|
+
const metaPath = path.join(workDir, 'meta.json');
|
|
94
|
+
if (!fs.existsSync(metaPath)) throw new Error(`no meta.json in ${workDir}`);
|
|
95
|
+
const meta = JSON.parse(fs.readFileSync(metaPath, 'utf8'));
|
|
96
|
+
if (!fs.existsSync(meta.framesDir) || !fs.readdirSync(meta.framesDir).length) {
|
|
97
|
+
throw new Error(`frames were not kept for this recording. Re-record with --keep-frames to enable rerender.`);
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
const raw = path.join(workDir, 'raw.mp4');
|
|
101
|
+
const narrated = path.join(workDir, 'narrated.mp4');
|
|
102
|
+
const out = path.join(workDir, 'demo.mp4');
|
|
103
|
+
await render(meta, raw, { onStatus });
|
|
104
|
+
|
|
105
|
+
let narration = { narrated: false };
|
|
106
|
+
if (narrate) narration = await addNarration(meta, raw, narrated, { voice, speed, ttsCmd, onStatus });
|
|
107
|
+
else fs.copyFileSync(raw, narrated);
|
|
108
|
+
|
|
109
|
+
const recipe = JSON.parse(fs.readFileSync(path.join(workDir, 'recipe.json'), 'utf8'));
|
|
110
|
+
onStatus('embedding recipe');
|
|
111
|
+
await embedRecipe(narrated, out, recipe);
|
|
112
|
+
fs.rmSync(raw, { force: true });
|
|
113
|
+
fs.rmSync(narrated, { force: true });
|
|
114
|
+
return { video: out, meta, narration };
|
|
84
115
|
}
|
|
85
116
|
|
|
86
117
|
async function outline(session, url) {
|
|
@@ -88,4 +119,4 @@ async function outline(session, url) {
|
|
|
88
119
|
return extractOutline(page);
|
|
89
120
|
}
|
|
90
121
|
|
|
91
|
-
module.exports = { produceDemo, outline };
|
|
122
|
+
module.exports = { produceDemo, rerender, outline };
|
package/recorder.js
CHANGED
|
@@ -40,8 +40,11 @@ class Timeline {
|
|
|
40
40
|
this.zoom = 1;
|
|
41
41
|
this.center = null;
|
|
42
42
|
}
|
|
43
|
-
recordSegment(caption, narration, dur = null) {
|
|
44
|
-
this.segments.push({
|
|
43
|
+
recordSegment(caption, narration, dur = null, voice = null, audio = null) {
|
|
44
|
+
this.segments.push({
|
|
45
|
+
t: Date.now(), caption: caption || null, narration: narration || null, dur,
|
|
46
|
+
...(voice ? { voice } : {}), ...(audio ? { audio } : {}),
|
|
47
|
+
});
|
|
45
48
|
}
|
|
46
49
|
recordMove(to, dur) {
|
|
47
50
|
this.moves.push({ t: Date.now(), from: { ...this.pos }, to: { ...to }, dur });
|
|
@@ -59,7 +62,9 @@ class Timeline {
|
|
|
59
62
|
}
|
|
60
63
|
|
|
61
64
|
class VoilaSession {
|
|
62
|
-
constructor({ profileDir, headless = false, device = 'desktop' } = {}) {
|
|
65
|
+
constructor({ profileDir, headless = false, device = 'desktop', dismiss = true, dismissSelector = null } = {}) {
|
|
66
|
+
this.dismiss = dismiss;
|
|
67
|
+
this.dismissSelector = dismissSelector;
|
|
63
68
|
this.profileDir = profileDir || path.join(__dirname, 'profile');
|
|
64
69
|
this.headless = headless;
|
|
65
70
|
this.device = DEVICES[device] ? device : 'desktop';
|
|
@@ -85,11 +90,9 @@ class VoilaSession {
|
|
|
85
90
|
// Zero-install path: fetch Chromium on first use instead of making the
|
|
86
91
|
// user run `npx playwright install` themselves.
|
|
87
92
|
if (!/Executable doesn't exist|missing dependencies|browser.*not found/i.test(String(e.message))) throw e;
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
catch { cliPath = path.join(path.dirname(require.resolve('playwright/package.json')), 'cli.js'); }
|
|
92
|
-
execFileSync(process.execPath, [cliPath, 'install', 'chromium'], { stdio: 'pipe', timeout: 600000 });
|
|
93
|
+
// Visible progress: a silent 150MB download reads as a hang.
|
|
94
|
+
process.stderr.write('[voila] first run: downloading Chromium (~150MB, one time)\n');
|
|
95
|
+
require('./doctor').installChromium();
|
|
93
96
|
this.context = await launch();
|
|
94
97
|
}
|
|
95
98
|
await this.context.addInitScript(OVERLAY_SOURCE);
|
|
@@ -99,6 +102,10 @@ class VoilaSession {
|
|
|
99
102
|
this.page.on('close', () => { this.page = null; });
|
|
100
103
|
await this.page.goto(url, { waitUntil: 'domcontentloaded', timeout: 45000 });
|
|
101
104
|
await this.page.waitForLoadState('networkidle', { timeout: 8000 }).catch(() => {});
|
|
105
|
+
if (this.dismiss) {
|
|
106
|
+
const { dismissConsent } = require('./consent');
|
|
107
|
+
this.dismissed = await dismissConsent(this.page, { selector: this.dismissSelector });
|
|
108
|
+
}
|
|
102
109
|
return this.page;
|
|
103
110
|
}
|
|
104
111
|
|
|
@@ -173,6 +180,7 @@ class VoilaSession {
|
|
|
173
180
|
frames: frames.sort((a, b) => a.t - b.t),
|
|
174
181
|
moves: tl.moves, zooms: tl.zooms, segments: tl.segments,
|
|
175
182
|
warnings: tl.warnings || [],
|
|
183
|
+
dismissed: this.dismissed || [],
|
|
176
184
|
framesDir,
|
|
177
185
|
};
|
|
178
186
|
fs.writeFileSync(path.join(workDir, 'meta.json'), JSON.stringify(meta));
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
// CI helper: assert the rendered demo actually carries a narration track.
|
|
2
|
+
const { execFileSync } = require('child_process');
|
|
3
|
+
const ffmpeg = require('ffmpeg-static');
|
|
4
|
+
|
|
5
|
+
const file = process.argv[2];
|
|
6
|
+
let out = '';
|
|
7
|
+
try { execFileSync(ffmpeg, ['-i', file], { stdio: 'pipe' }); }
|
|
8
|
+
catch (e) { out = String(e.stderr || ''); }
|
|
9
|
+
|
|
10
|
+
if (!/Audio: aac/.test(out)) {
|
|
11
|
+
console.error(out || '(no ffmpeg output)');
|
|
12
|
+
throw new Error(`no narration track in ${file}`);
|
|
13
|
+
}
|
|
14
|
+
console.log(`narration track OK in ${file}`);
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
// CI fixture: a stand-in "external TTS engine". Writes a short tone to {out}
|
|
2
|
+
// so the --tts-cmd plumbing can be exercised on every platform, including
|
|
3
|
+
// ones with no system voices installed.
|
|
4
|
+
const { execFileSync } = require('child_process');
|
|
5
|
+
const ffmpeg = require('ffmpeg-static');
|
|
6
|
+
const out = process.argv[2];
|
|
7
|
+
const text = process.argv.slice(3).join(' ');
|
|
8
|
+
const seconds = Math.max(1, Math.min(8, text.split(/\s+/).length / 3)).toFixed(2);
|
|
9
|
+
execFileSync(ffmpeg, ['-y', '-f', 'lavfi', '-i', `sine=frequency=340:duration=${seconds}`, '-ac', '1', out], { stdio: 'pipe' });
|
|
10
|
+
console.log(`fixture wrote ${seconds}s to ${out}`);
|
package/skills/voila/SKILL.md
CHANGED
|
@@ -14,9 +14,12 @@ Prefer the MCP tools if registered (`voila_outline`, `voila_record`,
|
|
|
14
14
|
`voila_review`); otherwise use the CLI via npx:
|
|
15
15
|
|
|
16
16
|
```bash
|
|
17
|
+
npx -y voila-recorder doctor # first run: pre-downloads chromium + voice model
|
|
17
18
|
npx -y voila-recorder outline <url>
|
|
18
19
|
npx -y voila-recorder record <url> --steps steps.yaml [--device mobile]
|
|
19
20
|
npx -y voila-recorder review demo.mp4
|
|
21
|
+
npx -y voila-recorder fork demo.mp4 --url https://other.example
|
|
22
|
+
npx -y voila-recorder rerender <dir> --voice bf_emma
|
|
20
23
|
```
|
|
21
24
|
|
|
22
25
|
Register the MCP server once with: `claude mcp add voila -- npx -y voila-recorder mcp`
|
|
@@ -61,9 +64,23 @@ Reference: https://voila.anzalabidi.dev/llms.txt · https://github.com/anzal1/vo
|
|
|
61
64
|
for them in chat.
|
|
62
65
|
- Prefer `zoom` with a `selector` over a bare `level`: voila measures the
|
|
63
66
|
element and picks the level and camera centre, so nothing is cropped.
|
|
64
|
-
-
|
|
67
|
+
- Languages: narration voice is per step, so demos can mix languages. Kokoro
|
|
68
|
+
is ENGLISH ONLY (af_heart A, af_bella A-, bf_emma British). For other
|
|
69
|
+
languages use a macOS system voice (`voice: "say:Monica"`, `voila voices
|
|
70
|
+
--all` lists ~180 across ~50 languages), or `--tts-cmd` with any engine on
|
|
71
|
+
any platform (placeholders {out} {text} {voice}), or point a step at a
|
|
72
|
+
ready-made clip with `audio: file.mp3`. On Linux/Windows non-English needs
|
|
73
|
+
--tts-cmd: tell the user rather than silently narrating in English.
|
|
74
|
+
- Voices: `voila voices` lists the English voices with quality grades. af_heart
|
|
65
75
|
(A) default, af_bella (A-), af_nicole (B-), bf_emma (B-, British). `--speed`
|
|
66
76
|
or the speed param (0.5-1.6) changes pace; 0.9 reads calmer.
|
|
67
77
|
- A failing step is retried once automatically; warnings appear in the result.
|
|
78
|
+
- Iterating on narration? Record once with `--keep-frames`, then `rerender <dir>
|
|
79
|
+
--voice x --speed n`. It skips the browser entirely and finishes in seconds.
|
|
80
|
+
- Consent banners are auto-dismissed (reject preferred over accept). Use
|
|
81
|
+
`--dismiss <selector>` for an unusual one, `--no-dismiss` to leave it alone.
|
|
82
|
+
- First run on a new machine downloads ~240MB. Run `voila doctor` first and tell
|
|
83
|
+
the user it is downloading, so it does not look frozen.
|
|
84
|
+
- `fork <video.mp4>` rebuilds any voila demo from the recipe inside the file.
|
|
68
85
|
- Narration style: short sentences, product language, no "as you can see".
|
|
69
86
|
8–15 words per beat reads best at Kokoro's pace.
|
package/tour.js
CHANGED
|
@@ -242,7 +242,7 @@ async function runSteps(page, tl, steps, opts) {
|
|
|
242
242
|
const step = steps[si];
|
|
243
243
|
if (step.caption || step.narration) {
|
|
244
244
|
await finishSegment();
|
|
245
|
-
tl.recordSegment(step.caption, step.narration, step._narrDurMs || null);
|
|
245
|
+
tl.recordSegment(step.caption, step.narration, step._narrDurMs || null, step.voice || null, step.audio || null);
|
|
246
246
|
segStart = Date.now();
|
|
247
247
|
segMinMs = (step._narrDurMs || 0) + 600;
|
|
248
248
|
}
|
package/voices.js
CHANGED
|
@@ -55,4 +55,52 @@ function format() {
|
|
|
55
55
|
return lines.join('\n');
|
|
56
56
|
}
|
|
57
57
|
|
|
58
|
-
|
|
58
|
+
// --- system voices (macOS `say`) ---------------------------------------------
|
|
59
|
+
// Kokoro is English-only in JS (its other voice files ship without a
|
|
60
|
+
// grapheme-to-phoneme stage for those languages). Every Mac already carries
|
|
61
|
+
// ~180 voices across ~50 languages, so those cover non-English narration.
|
|
62
|
+
|
|
63
|
+
let sysCache = null;
|
|
64
|
+
function systemVoices() {
|
|
65
|
+
if (sysCache) return sysCache;
|
|
66
|
+
sysCache = [];
|
|
67
|
+
if (process.platform !== 'darwin') return sysCache;
|
|
68
|
+
try {
|
|
69
|
+
const { execFileSync } = require('child_process');
|
|
70
|
+
const out = String(execFileSync('say', ['-v', '?'], { maxBuffer: 4e6 }));
|
|
71
|
+
for (const line of out.split('\n')) {
|
|
72
|
+
const m = /^(.+?)\s{2,}([a-z]{2}_[A-Z]{2})\s/.exec(line);
|
|
73
|
+
if (m) sysCache.push({ id: `say:${m[1].trim()}`, name: m[1].trim(), language: m[2].replace('_', '-'), gender: '', grade: 'system' });
|
|
74
|
+
}
|
|
75
|
+
} catch { /* no say binary */ }
|
|
76
|
+
return sysCache;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
function systemLanguages() {
|
|
80
|
+
const langs = {};
|
|
81
|
+
for (const v of systemVoices()) (langs[v.language] = langs[v.language] || []).push(v.name);
|
|
82
|
+
return langs;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
function isSystemVoice(id) {
|
|
86
|
+
if (!id) return false;
|
|
87
|
+
const name = String(id).replace(/^say:/, '').toLowerCase();
|
|
88
|
+
return systemVoices().some(v => v.name.toLowerCase() === name);
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
function formatAll() {
|
|
92
|
+
const lines = [format()];
|
|
93
|
+
const langs = systemLanguages();
|
|
94
|
+
const codes = Object.keys(langs).sort();
|
|
95
|
+
if (!codes.length) {
|
|
96
|
+
lines.push('\nSystem voices: none found (macOS only). For other languages use --tts-cmd.');
|
|
97
|
+
return lines.join('\n');
|
|
98
|
+
}
|
|
99
|
+
lines.push(`\nSystem voices (macOS, ${systemVoices().length} across ${codes.length} languages)`);
|
|
100
|
+
for (const c of codes) lines.push(` ${c.padEnd(7)} ${langs[c].slice(0, 6).join(', ')}${langs[c].length > 6 ? ` +${langs[c].length - 6}` : ''}`);
|
|
101
|
+
lines.push('\nUse a system voice for non-English narration: --voice "say:Monica"');
|
|
102
|
+
lines.push('Any other engine: --tts-cmd \'piper --model es.onnx -f {out} -- "{text}"\'');
|
|
103
|
+
return lines.join('\n');
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
module.exports = { allVoices, ranked, isValid, suggest, format, formatAll, systemVoices, systemLanguages, isSystemVoice };
|