@spark-apps/quickpeek 1.2.3 → 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.mts CHANGED
@@ -1,16 +1,11 @@
1
- import { U as UserTier, C as Config } from './scraping-C334VxX5.mjs';
2
- export { A as AIError, a as AIResponse, b as AIResult, c as CLI_BACKOFF_MS, d as CONFIG_FILE, e as CaptionStyle, f as CaptionWordStyle, g as ChatOptions, D as DEFAULT_CAPTIONS, h as DEFAULT_CONFIG, E as EXCLUDED_LINK_PATTERNS, i as ElementInfo, j as ElementType, F as FEATURES, H as HighlightMode, I as INTERACTIVE_SELECTORS, P as Plan, R as RateLimitInfo, k as RetryNotice, S as SERVER_BACKOFF_MS, l as SparkStatus, m as SparkSubscription, n as SparkTrial, o as Step, p as SuggestedAction, T as Tier, V as VERSION, q as buildSystemPrompt, r as buildUserPrompt, s as callAI, t as callAIViaRelay, u as crawlPage, v as getErrorMessage, w as getLanguageName, x as getStatus, y as getStatusCached, z as getTierByEmail, B as hasAccess, G as hasTrialRemaining, J as idSelector, K as isCanceling, L as isExcludedLink, M as isPaid, N as isVerified, O as normalizeUrl, Q as parseAIPlanResponse, W as parsePartial, X as pricingUrl, Y as rgbaToHex, Z as shouldSkipLink, _ as showPaywall, $ as toTitleCase } from './scraping-C334VxX5.mjs';
3
- export { WIDE_VOICES as LANGUAGE_VOICES, hexToAss as hexToASS, windowsDrivePathToWsl as windowsPathToWSL } from '@spark-apps/video-kit';
1
+ import { U as UserTier, C as CaptionStyle, a as CaptionPosition, b as Config, V as VoiceRate } from './voices-ufga0oOn.mjs';
2
+ export { A as AIError, c as AIResponse, d as AIResult, e as CLI_BACKOFF_MS, f as CONFIG_FILE, g as CaptionPreset, h as CaptionWordStyle, i as ChatOptions, D as DEFAULT_CAPTIONS, j as DEFAULT_CONFIG, E as EXCLUDED_LINK_PATTERNS, k as ElementInfo, l as ElementType, H as HighlightMode, I as INTERACTIVE_SELECTORS, m as INTERNATIONAL_VOICE, N as NO_CLIP_OUTRO, P as PLAN_MAX_TOKENS, n as Plan, R as RATE_FACTOR, o as RELAY_BASE, p as RateLimitInfo, q as RetryNotice, S as SERVER_BACKOFF_MS, r as SIZE_PRESETS, s as SparkStatus, t as SparkSubscription, u as SparkTrial, v as Step, w as SuggestedAction, T as Tier, x as VERSION, y as VIDEO_PROFILES, z as VideoProfileName, B as VideoSize, F as VoiceGender, G as applyProfile, J as ariaLabelSelector, K as buildSystemPrompt, L as buildUserPrompt, M as callAI, O as callAIViaRelay, Q as crawlPage, W as deStock, X as defaultCaptionsFor, Y as getErrorMessage, Z as getLanguageName, _ as getOrRefreshRelayToken, $ as getStatus, a0 as getStatusCached, a1 as getTierByEmail, a2 as gotoSettled, a3 as hasTrialRemaining, a4 as hasVoice, a5 as idSelector, a6 as internationalVoiceFor, a7 as isCanceling, a8 as isExcludedLink, a9 as isPaid, aa as isVerified, ab as normalizeUrl, ac as parseAIPlanResponse, ad as parsePartial, ae as pricingUrl, af as rgbaToHex, ag as runPooled, ah as shouldSkipLink, ai as showPaywall, aj as stripLongDashes, ak as toTitleCase, al as voiceFor, am as waitForPlaceholdersGone } from './voices-ufga0oOn.mjs';
3
+ export { MALE_VOICES, MULTILINGUAL_VOICES, WIDE_VOICES, hexToAss as hexToASS, windowsDrivePathToWsl as windowsPathToWSL } from '@spark-apps/video-kit';
4
+ import 'playwright';
4
5
 
5
6
  /**
6
- * Billing utilities — watermark decisions and upgrade URL generation
7
+ * Billing utilities — upgrade URL generation
7
8
  */
8
-
9
- /**
10
- * Check if watermark should be applied
11
- * Watermark applies to free tier videos longer than 20 seconds
12
- */
13
- declare function shouldWatermark(tier: UserTier, videoDurationSeconds: number): boolean;
14
9
  /**
15
10
  * Get upgrade URL for pricing page
16
11
  */
@@ -33,14 +28,27 @@ interface StepSync {
33
28
  targetDuration: number;
34
29
  }
35
30
  interface ComposeOptions {
31
+ /** Input 0 carries a spliced clip's own audio, to be mixed under the voice. */
32
+ hasClipAudio?: boolean;
36
33
  videoPath: string;
37
- assPath: string;
34
+ /** Burn-in captions. Omit to skip the ass filter entirely (draft). */
35
+ assPath?: string | undefined;
38
36
  outputPath: string;
39
37
  voiceoverPath?: string | undefined;
40
38
  musicPath?: string | undefined;
41
39
  musicVolume?: number | undefined;
42
40
  musicFadeIn?: number | undefined;
43
41
  musicFadeOut?: number | undefined;
42
+ /**
43
+ * Loudness the bed is normalised to before it meets the narration, in LUFS.
44
+ *
45
+ * A fixed `musicVolume` cannot produce a consistent bed, because the source
46
+ * decides everything: two CC0 tracks in the same folder measure -31 dB and
47
+ * -15 dB mean, so the same gain makes one inaudible and the other a
48
+ * competitor. Normalising first makes the bed a level rather than a
49
+ * multiplier, and `musicVolume` goes back to being the trim it reads as.
50
+ */
51
+ musicTargetLufs?: number | undefined;
44
52
  width: number;
45
53
  height: number;
46
54
  fps: number;
@@ -53,9 +61,44 @@ interface ComposeOptions {
53
61
  contrast?: number | undefined;
54
62
  trimStart?: number | undefined;
55
63
  userTier?: UserTier | undefined;
64
+ endsOnCreditsCard?: boolean | undefined;
56
65
  mainContentDuration?: number | undefined;
57
66
  stepSync?: StepSync[] | undefined;
58
67
  stepTransitions?: StepTransition[] | undefined;
68
+ /** 'off' pins the deliverable encode to libx264; anything else probes for a GPU. */
69
+ hwaccel?: 'auto' | 'off' | undefined;
70
+ /**
71
+ * The tempo pass that will run AFTER this compose, if any.
72
+ *
73
+ * That pass speeds the finished file up, and it cannot tell a narrator from
74
+ * a song: a demo scored with a track came out with the music playing 1.2x,
75
+ * which is audible on anything with a beat and is simply wrong - the voice
76
+ * is what the pacing is for. The bed is laid down pre-slowed by the same
77
+ * factor here, so the later speed-up returns it to its natural tempo.
78
+ * atempo preserves pitch in both directions, so the round trip is a tempo
79
+ * change and not a transposition. 1 (the default) changes nothing.
80
+ */
81
+ tempo?: number | undefined;
82
+ /**
83
+ * Where the music bed starts, in seconds on this compose's timeline.
84
+ *
85
+ * 0 (the default) scores the whole video. Set to the closing card's start
86
+ * and the demo plays dry, with the song arriving as the card does - which
87
+ * is what you want when the track is the product's own and is meant to be
88
+ * heard rather than ducked under a narrator for a minute first.
89
+ */
90
+ musicStartSecs?: number | undefined;
91
+ /**
92
+ * Length of the music file, in seconds, so its END can be landed on the
93
+ * video's end.
94
+ *
95
+ * A song has an ending, and a song cut off two thirds through does not: it
96
+ * stops. Given the track's length, the bed is seeked so that its last note
97
+ * falls on the last frame, which is what makes a closing card feel closed
98
+ * rather than interrupted. Omitted, the bed plays from its beginning as
99
+ * before.
100
+ */
101
+ musicDurationSecs?: number | undefined;
59
102
  }
60
103
  interface ConcatVideoOptions {
61
104
  inputPath: string;
@@ -69,7 +112,27 @@ interface VideoOverlay {
69
112
  path: string;
70
113
  startTime: number;
71
114
  duration: number;
72
- mode: 'stretch' | 'center';
115
+ /**
116
+ * How the source is fitted to the frame.
117
+ *
118
+ * stretch = fill the frame's width, crop the height if it overflows.
119
+ * center = scale to fit, letterbox the rest.
120
+ * cover = fill the frame with a blurred, enlarged copy of the source and
121
+ * sit the source itself on top of it, untouched. A 1200x630 OG
122
+ * card in a 9:16 frame is 16:9 either way; the only question is
123
+ * what fills the two thirds of the frame it cannot reach, and a
124
+ * blurred continuation of the card reads as design where a black
125
+ * band reads as a mistake.
126
+ */
127
+ mode: 'stretch' | 'center' | 'cover';
128
+ /**
129
+ * The source is a still image (png/jpg/webp), not a clip.
130
+ *
131
+ * A still has exactly one frame, so it has to be looped for the length of
132
+ * its window or the overlay holds a single frame at EOF and the window
133
+ * plays whatever is underneath it.
134
+ */
135
+ still?: boolean;
73
136
  trimStart?: number;
74
137
  trimEnd?: number;
75
138
  }
@@ -105,6 +168,34 @@ interface TimeSegment {
105
168
  end: number;
106
169
  }
107
170
 
171
+ /**
172
+ * A beat of silence before the narrator starts, in milliseconds.
173
+ *
174
+ * The voice used to be faded in over the same duration as the picture, which
175
+ * ate the first syllable - and the first syllable belongs to the hook, the one
176
+ * line that has to land. A short delay gives the viewer the same moment to
177
+ * settle without softening the words.
178
+ */
179
+ declare const VOICE_LEAD_MS = 200;
180
+ /**
181
+ * Loudness the music bed is normalised to before the mix, in LUFS.
182
+ *
183
+ * Measured, not guessed. Running this exact graph over the accordio
184
+ * narration puts the bed at -26.7 dB mean in a gap between lines while the
185
+ * speech measures -13.7 dB: thirteen dB under the voice, which is a bed. Each
186
+ * dB here moves that reading a dB, so the number is a real control - -38 gives
187
+ * -30.6, -42 gives -34.7.
188
+ *
189
+ * Over a long passage with no narration at all - the closing card - the master
190
+ * chain's single-pass loudnorm is dynamic and opens up, and the bed rises to
191
+ * about -18 dB. That is the intended shape: discreet under the voice, present
192
+ * where there is nothing else.
193
+ *
194
+ * What it replaced measured -39 dB at source and -45 dB after its own volume
195
+ * trim, and read in the finished file at -32 dB against -15 dB of speech.
196
+ * That is not a quiet bed, it is no bed: the video was speech in a vacuum.
197
+ */
198
+ declare const MUSIC_BED_LUFS = -34;
108
199
  /**
109
200
  * Compose final video with subtitles, voiceover, and music
110
201
  */
@@ -120,6 +211,45 @@ declare function buildKeepSegments(eventTimestampsSeconds: number[], totalDurati
120
211
  */
121
212
  declare function mapTimestampsAfterCompression(originalTimestampsSeconds: number[], segments: TimeSegment[]): number[];
122
213
 
214
+ /**
215
+ * Video → optimized GIF conversion (ported from VidLet's togif tool).
216
+ * Two-pass: generate a tuned palette, then map the video through it —
217
+ * dramatically better colour than ffmpeg's default 256-colour GIF path.
218
+ */
219
+ interface GifOptions {
220
+ input: string;
221
+ output: string;
222
+ fps?: number;
223
+ width?: number;
224
+ dither?: string;
225
+ statsMode?: string;
226
+ }
227
+ declare const GIF_DEFAULTS: {
228
+ readonly fps: 15;
229
+ readonly width: 480;
230
+ readonly dither: "sierra2_4a";
231
+ readonly statsMode: "full";
232
+ };
233
+ declare function buildPaletteArgs(input: string, palettePath: string, fps: number, width: number, statsMode: string): string[];
234
+ declare function buildGifArgs(input: string, palettePath: string, output: string, fps: number, width: number, dither: string): string[];
235
+ declare function convertToGif(options: GifOptions): Promise<string>;
236
+
237
+ /**
238
+ * The video stream's pixel dimensions.
239
+ *
240
+ * Needed because the splice pass scales its overlays to match the video it is
241
+ * pasting them onto, and that video is the raw recording, which is smaller
242
+ * than the configured output whenever video.zoom is in play: a short records
243
+ * at 540x960 and is upscaled to 1080x1920 only at compose. Scaling overlays to
244
+ * the output size instead put them in at 2x and cropped everything but the
245
+ * middle of each card.
246
+ */
247
+ declare function getVideoDimensions(filePath: string): Promise<{
248
+ width: number;
249
+ height: number;
250
+ }>;
251
+ /** Whether a file carries an audio stream at all. */
252
+ declare function hasAudioStream(filePath: string): Promise<boolean>;
123
253
  /**
124
254
  * Run ffprobe to get media duration in milliseconds
125
255
  */
@@ -146,6 +276,22 @@ declare function isHonestWav(filePath: string): Promise<boolean>;
146
276
  */
147
277
  declare function buildConcatArgs(listPath: string, output: string): string[];
148
278
  declare function concatMedia(listPath: string, output: string): Promise<void>;
279
+ /**
280
+ * Strip the silence a TTS engine leaves on both ends of a clip.
281
+ *
282
+ * Every edge-tts clip arrives with its own lead-in and tail, typically a few
283
+ * hundred ms each. One clip per caption, concatenated, means each boundary
284
+ * stacks tail + the pad that aligns the step + the next clip's lead-in - the
285
+ * "robotic pause" between sentences. Trimming here rather than at concat time
286
+ * keeps the clip the unit of truth: the duration measured after this call,
287
+ * the padding computed from it, and the word timings rebased in
288
+ * spokenWordTimings all describe the same audio.
289
+ *
290
+ * -45dB rather than a hard zero: edge-tts pads with near-silent noise, not
291
+ * digital black, so a stricter gate finds nothing to remove.
292
+ */
293
+ declare const TRIM_EDGE_SILENCE: string;
294
+ declare function trimEdgeSilence(filePath: string): Promise<void>;
149
295
  /**
150
296
  * Generate silence audio file of specified duration
151
297
  */
@@ -160,6 +306,262 @@ declare function speedUpVideo(opts: SpeedUpOptions): Promise<void>;
160
306
  * Returns the combined video for further processing (composition with audio)
161
307
  */
162
308
  declare function concatVideoWithOutro(opts: ConcatVideoOptions): Promise<void>;
309
+ /**
310
+ * Append an outro that is already a full frame, unchanged.
311
+ *
312
+ * concatVideoWithOutro treats its outro as a badge: it shrinks it to 60% and
313
+ * overlays it on black, which is right for the bundled square watermark and
314
+ * wrong for a rendered credits card, which is composed at the video's exact
315
+ * size and must not be shrunk inside its own frame.
316
+ *
317
+ * setsar on both inputs because concat refuses streams whose sample aspect
318
+ * ratios disagree, and a card rendered by the browser does not necessarily
319
+ * carry the same SAR as a recorded page.
320
+ */
321
+ declare function concatVideoWithFullFrameOutro(opts: ConcatVideoOptions): Promise<void>;
322
+
323
+ /**
324
+ * Compressing dead air out of a finished demo.
325
+ *
326
+ * A step whose narration ends before the step does leaves the video sitting
327
+ * on a still frame with nothing being said - padding that keeps audio and
328
+ * video in sync, plus any deliberate hold while an async result loads. Held
329
+ * for a second it reads as a beat; held for five it reads as a hang.
330
+ *
331
+ * This runs AFTER compose on purpose. By then the captions are burned in and
332
+ * the silent spans have none on screen (a karaoke line ends with its last
333
+ * word), so speeding those spans up removes the dead time without touching
334
+ * caption timing, the recorder, or the TTS timeline - the three things that
335
+ * have to agree with each other. What is sped up is silence, so it is
336
+ * inaudible, and the visuals still play through: a result that appears during
337
+ * a hold is still seen, just briskly.
338
+ */
339
+ /** A silent stretch of the mixed audio, in seconds. */
340
+ interface SilenceSpan {
341
+ start: number;
342
+ end: number;
343
+ /** Longest this span may run after compression. Defaults to the pass's cap. */
344
+ cap?: number;
345
+ }
346
+ /**
347
+ * Length of the video stream in seconds, or 0 if it cannot be read.
348
+ *
349
+ * Distinct from the container's duration, which reports whichever stream runs
350
+ * longest - usually the audio, since compose pads it to the voiceover.
351
+ */
352
+ declare function getVideoStreamDuration(inputPath: string): Promise<number>;
353
+ declare function detectSilences(inputPath: string, minDurSecs: number): Promise<SilenceSpan[]>;
354
+ /**
355
+ * Find stretches where the picture stops changing.
356
+ *
357
+ * Silence is not the only kind of dead air. A page that finishes an animation
358
+ * - a confetti burst settling, a spinner resolving - holds a still frame for
359
+ * as long as the step is paced to last, and that reads as a hang even with
360
+ * narration over it. freezedetect reports those stretches; they are treated
361
+ * like silence, and for the same reason.
362
+ */
363
+ declare function detectFrozenSpans(inputPath: string, minDurSecs: number): Promise<SilenceSpan[]>;
364
+ /**
365
+ * Spans where the picture is essentially black.
366
+ *
367
+ * A frozen span and a black span are not the same thing, and the difference
368
+ * matters. A frozen span is a still screen: the viewer is looking at something,
369
+ * so it earns the FROZEN_MAX_SECS ceiling and gets shortened, not cut. A black
370
+ * span is a page mid-transition, and it is worth nothing at any length. One run
371
+ * shipped 9.4 seconds of it in a 75 second video.
372
+ *
373
+ * Playwright records continuously and cannot be paused across a navigation, so
374
+ * these frames cannot be prevented at capture time. They are removed here
375
+ * instead, and only where the audio is silent too, so no narration is ever cut.
376
+ */
377
+ declare function detectBlackSpans(inputPath: string, minDurSecs: number): Promise<SilenceSpan[]>;
378
+ /**
379
+ * Spans where the picture is essentially white.
380
+ *
381
+ * blackdetect over an inverted picture. A page that flashes white between
382
+ * routes is as empty as one that flashes black, and a light-themed app makes
383
+ * white the common case, so detecting only one of the two fixes half the sites.
384
+ */
385
+ declare function detectWhiteSpans(inputPath: string, minDurSecs: number): Promise<SilenceSpan[]>;
386
+ /**
387
+ * Where two sets of spans overlap.
388
+ *
389
+ * A frozen picture is only safe to speed up while nothing is being said over
390
+ * it - rushing a still frame is free, rushing a sentence is not. The overlap
391
+ * of "frozen" and "silent" is exactly the footage with nothing happening in
392
+ * either channel.
393
+ */
394
+ declare function intersectSpans(a: SilenceSpan[], b: SilenceSpan[]): SilenceSpan[];
395
+ /** Fold overlapping or touching spans into one. */
396
+ declare function mergeSpans(spans: SilenceSpan[]): SilenceSpan[];
397
+ /** atempo caps at 2x per instance, so a bigger speed-up is a chain of them. */
398
+ declare function atempoChain(factor: number): string;
399
+ /**
400
+ * Build the filter_complex that plays `spans` fast and the rest untouched.
401
+ *
402
+ * Split out from the run so the graph can be unit-tested: an ffmpeg filter
403
+ * error surfaces as a wall of text at render time, which is a bad place to
404
+ * discover an off-by-one in the segment list.
405
+ */
406
+ declare function buildCompressionFilter(spans: SilenceSpan[], totalSecs: number, maxSecs: number, protectAfterSecs?: number,
407
+ /**
408
+ * Everything BEFORE this is the opening card, protected for the same reason
409
+ * as the closing one and previously not protected at all.
410
+ *
411
+ * It did not need to be while the intro card was narrated over: a spoken
412
+ * card is not a silent span. A card prepended as its own step has no
413
+ * narration, so it is silent AND frozen, and this pass gave it the frozen
414
+ * ceiling of 0.6s - a title card gone before it can be read.
415
+ */
416
+ protectBeforeSecs?: number): string | null;
417
+ /**
418
+ * Drop spans from a silent video, in its own timeline.
419
+ *
420
+ * The recording has no audio yet, so there is nothing to desync: this is the
421
+ * one place in the pipeline where a span can simply be deleted. Doing it here
422
+ * rather than after compose also avoids the coordinate problem that made an
423
+ * earlier attempt a no-op. A recording is 145s where its composed video is
424
+ * 74s, because compose trims the page load off the front and the later passes
425
+ * squeeze the rest, so a span measured during recording means nothing once
426
+ * those have run.
427
+ */
428
+ declare function cutSpansFromVideo(inputPath: string, outputPath: string, spans: Array<{
429
+ start: number;
430
+ end: number;
431
+ }>, preset: string): Promise<boolean>;
432
+ /**
433
+ * How much faster a loading screen plays than the footage around it.
434
+ *
435
+ * The empty stretches in a recording are almost always one thing: the app's
436
+ * own splash while a route loads. A sellular run spent 17 of its 66 seconds
437
+ * on that screen. Five is fast enough that it reads as a flicker rather than
438
+ * a wait, and slow enough that the logo is still legible going past.
439
+ */
440
+ declare const EMPTY_SPAN_SPEED = 5;
441
+ /**
442
+ * Build the filter that removes every empty span from view without changing
443
+ * how long anything lasts, one step at a time.
444
+ *
445
+ * The point of doing it per step is that every step boundary comes out exactly
446
+ * where it went in. Cutting empty frames is the obvious move and it does not
447
+ * work: the narration for each step is already synthesised at a fixed length,
448
+ * so shortening the picture under it desyncs everything downstream, and the
449
+ * guard that stops that from happening starves the cut until it removes
450
+ * nothing at all. So the seconds are not removed, they are re-spent.
451
+ *
452
+ * A step with real footage either side of the splash plays the splash at
453
+ * `speed` and stretches that footage to fill exactly what it freed. A step
454
+ * that is nearly all splash - a route whose loading screen outlasts the line
455
+ * narrated over it - has nothing to stretch, so the splash is replaced by a
456
+ * still of the page that arrives after it. Both leave the step the length it
457
+ * was, and neither leaves an empty screen up.
458
+ *
459
+ * Returns null when there is nothing worth doing, so the caller can skip a
460
+ * re-encode that would change nothing.
461
+ */
462
+ declare function buildRetimeFilter(spans: Array<{
463
+ start: number;
464
+ end: number;
465
+ }>, stepBounds: Array<{
466
+ from: number;
467
+ to: number;
468
+ }>, speed: number, fps: number): string | null;
469
+ /**
470
+ * Play the empty stretches of a silent recording fast, keeping every step the
471
+ * length it was.
472
+ *
473
+ * Resolves false when the filter had nothing to do or ffmpeg refused it, and
474
+ * the caller keeps the original recording.
475
+ */
476
+ declare function retimeSpansInVideo(inputPath: string, outputPath: string, spans: Array<{
477
+ start: number;
478
+ end: number;
479
+ }>, stepBounds: Array<{
480
+ from: number;
481
+ to: number;
482
+ }>, speed: number, fps: number, preset: string): Promise<boolean>;
483
+ declare function compressSilentSpans(inputPath: string, outputPath: string, totalSecs: number, maxSecs: number, preset: string, protectAfterSecs?: number,
484
+ /**
485
+ * The narration on its own, when there is one.
486
+ *
487
+ * Silence has to be measured against the voice, not against the mix: any
488
+ * backing track - a music bed, room tone - is continuous by design, so a
489
+ * mixed file has no silence in it at all and every pause looks like speech.
490
+ */
491
+ narrationPath?: string,
492
+ /** Opening card to leave alone, in seconds from the start. See the filter. */
493
+ protectBeforeSecs?: number,
494
+ /**
495
+ * Spans the recorder KNOWS were a page loading rather than a page showing.
496
+ *
497
+ * Cut on the recorder's word, not on a detector's guess. blackdetect cannot
498
+ * tell a loading screen from a dark-themed app, and the silence guard
499
+ * refuses to cut anything with a voice over it, which is exactly the case
500
+ * here: the narration describing a result is queued behind the wait for it.
501
+ * The recorder measured this span, so it needs no evidence.
502
+ */
503
+ leadSpans?: Array<{
504
+ start: number;
505
+ end: number;
506
+ }>,
507
+ /**
508
+ * Spans that are playing CONTENT, however quiet they are.
509
+ *
510
+ * A clip spliced in under no narration is silent by nature and frozen to a
511
+ * freeze detector between cuts, so every heuristic here reads it as dead
512
+ * air and compresses it to the ceiling. A nine-second demo arrived as
513
+ * three-quarters of a second. Silence during a clip is the clip, not a
514
+ * hang, so these spans are removed from the compression list outright.
515
+ */
516
+ keepSpans?: Array<{
517
+ start: number;
518
+ end: number;
519
+ }>): Promise<boolean>;
520
+ /**
521
+ * Speed up only the middle of a video, leaving its ends untouched.
522
+ *
523
+ * A demo opens on a title card and closes on a credits card. Speeding the
524
+ * whole file to fit a length limit rushes both: the card the viewer is meant
525
+ * to read, and the offer they are meant to write down. Only the demo between
526
+ * them is compressible.
527
+ *
528
+ * Returns false when the ends already fill the budget - there is nothing to
529
+ * squeeze then, and the honest answer is a shorter plan.
530
+ */
531
+ declare function speedUpMiddle(inputPath: string, outputPath: string, totalSecs: number, targetSecs: number, introSecs: number, outroSecs: number, maxFactor: number, preset: string): Promise<{
532
+ ok: boolean;
533
+ factor: number;
534
+ reachedTarget: boolean;
535
+ }>;
536
+
537
+ /**
538
+ * Fill the frame with the card's own colours, then lay the card on it.
539
+ *
540
+ * A 1200x630 OG card is the shape every site publishes and the wrong shape for
541
+ * a 9:16 opening: centred, it floats in two thirds of a frame of flat #1a1a2e,
542
+ * which is what "letterboxed with dead space" looks like. Cropping it to fill
543
+ * instead would throw away the half of the card the headline is written
544
+ * across. So the backdrop is built from the card itself - the frame is full,
545
+ * the colours are the card's own, and not a pixel of the card is lost.
546
+ *
547
+ * The backdrop is RESAMPLED DOWN to a handful of pixels before it is blown
548
+ * back up, rather than blurred at full resolution. Blur is the obvious move
549
+ * and it is wrong here, because an OG card is mostly TYPE: a 1200px-wide
550
+ * headline enlarged 3x stays legible through sigma=28, so the opening frame
551
+ * showed the same words twice - huge and soft behind the card, sharp on it.
552
+ * That reads as a rendering fault rather than a design, and it was the first
553
+ * thing anyone saw of the product.
554
+ *
555
+ * Ten pixels of height cannot carry a glyph, so nothing survives to be read
556
+ * while the card's colour field does. The gblur afterwards only takes the
557
+ * blockiness off the upscale.
558
+ *
559
+ * The downscale is its own `scale` with BOTH dimensions given. `scale=40:-2`
560
+ * is silently ignored at these sizes - the first attempt at this looked
561
+ * completely unchanged because of it, which is why the test asserts the
562
+ * literal pair rather than just "smaller than the frame".
563
+ */
564
+ declare function coverFilterChain(inputIdx: number, width: number, height: number): string;
163
565
 
164
566
  /**
165
567
  * Splice video files into the recording at specific timestamps
@@ -248,32 +650,6 @@ interface LenientParse<T> {
248
650
  */
249
651
  declare function parseAIJson<T>(content: string): LenientParse<T> | null;
250
652
 
251
- declare const CURSOR_CSS = "\n #qp-cursor {\n position: fixed;\n width: 32px;\n height: 38px;\n background: url(\"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAMgAAAEsBAMAAAB01OGNAAAAIGNIUk0AAHomAACAhAAA+gAAAIDoAAB1MAAA6mAAADqYAAAXcJy6UTwAAAAtUExURUdwTAAAAAAAAAAAAAAAAAAAAAAAAB0dHVZWVtvb2/X19f///6mpqXV1df////vm1ewAAAAGdFJOUwAltTfpcLO208EAAAABYktHRA5vvTBPAAAAB3RJTUUH6QwFFREKplBWTAAABqdJREFUeNq93c+OFFUUx/EeCHud6H7A4HpCOu6NiWsWZh5Ax8YFbAdNWBPxBcCo+yGjD+CAW1ywh0Rexu6uulX3Vp269/c7f7o3Bpjhk6+H6TrU0LdXR6sDPI4/OgCyvn0I5NsDpKw3B0hZbw6Qst4cIGWLxKdskfiUHRKeskPCU/ZIdMoeiU7pkOCUDglO6ZHYlB6JTUlIaEpCQlMGJDJlQCJTRiQwZUQCUzIkLiVD4lJyJCwlR8JSCiQqpUCiUkokKKVEglImSEzKBIlJmSIhKT3yY2hKjzx7GpnSI7+8iExJyN+RKQl5FZkyIJEpAxKZMiKBKSMSmJIhcSkZEpeSI2EpORKWUiBRKQUSlVIiQSklEpQyQWJSJkhMyhQJSZkiISkzJCJlhkSkzJGAlDkSkCIg/ikC4p8iIe4pEuKeIiLeKSLinSIjziky4pyygPimLCC+KUuIa8oS4pqyiHimLCKeKcuIY8oy4phSQfxSKohfSg1xS6khbilVxCulinil1BGnlDrilNJAfFIaiE9KC3FJaSEuKU3EI6WJeKS0EYeUNuKQAiD2FACxpyCIOQVBzCkQYk2BEGsKhhhTMMSYAiK2FBCxpaCIKQVFTCkwYkmBEUsKjhhScMSQQiD6FALRpzCIOoVB1CkUok2hEG0KhyhTOESZQiK6FBLRpbCIKoVFVCk0okmhEU0KjyhSeESRokD4FAXCp2gQOkWD0CkqhE1RIWyKDiFTdAiZokS4FCXCpWgRKkWLUClqhElRI0yKHiFS9AiRYkDwFAOCp1gQOMWCwCkmBE0xIWiKDQFTbAiYYkSwFCOCpVgRKMWKQClmBEkxI0iKHQFS7AiQ4oC0UxyQdooH0kzxQJopLkgrxQVppfggjRQfpJHihNRTnJB6ihdSTfFCqiluSC3FDaml+CGVFD+kkuKILKc4IsspnshiiieymOKKLKW4Iq8u5RRf5FpO8UUWpuKMyCnOiJzijYgp3oiY4o5IKe6IlEIhL38FHr9fzFK47/0On195PL6YpVDIX4CRP1IKhbxGSvLHbQUy/snhUrg/XW9IpE/hkOcs0qVwyD8s0qWQ349nJ9+lcAg9+S6FfFqhJ79PIZFLGtmlkMifPLJNYf/5wgWvfAYg11f5j951H39+B3/cBZA/3uY/+nccJ/5oItdPi19Lkz91RV5snuQ/fN0jJ57I9svvkTT5rzyR7RX7wW/C5L93RPbPI8Xknysmj7zWQZz8fTeke0J8aJ088vqTR1czl5t8Fel/w2Ly12/6oXghaRn8IEz+nJh8DRkuUc/yn02XYGLy0Ou0ysn3X44+/75rvNb+IE3+OxdkXM/Lr3l+8stIvjSIkz91QMaQzeYn2+QXkWL7KSafnogd/jlcHiJPHn8iXkLKNe7BW9Pk26/73T/+y3/tkp38AjLdR4vJp+XrxIhMQuRL8Jc2ZLZYF5OnL8Gt18ff6f9bTJ5dvkQku6vw6Ub4AHb5ap1ZcPS5MHn2Eiwhxe2Rs8rk0Utw8xyJ/gNMy5eAlPd5bkiTJ5ev5tkeN7+uTP6+FpncsEqTt6zd7fNWpMmTy9cMmd15+0SYPLl8tc/AuVWZPLh8TZH5LURx8unOFzb59rlEafKG5WuCSPdC0+SvhMljyxdwVtSx9DVPTb5ExPvTafIfpMmf8oh4pz1NXr92F8jCtz+kyVPLF3Km2jfdz+mXrxxZCLFPHjnnTpw8s3xlyFLIMHn18gWdPVibPLJ8jchiyPBB5eTf4ZOHzoNMky+eiInlCzrZMk1eWruR5Qs6o/NImjxxCcZOGz2rTB64BGPnpkqTJ5Yv7ARY4/LVIz9XQ4bJK5cv7FRe49oNni/cT165fIEnJdfW7vYTMXjm8w3pax6ePHh6tW35As/hltfui8ZniUglvDb55vKFno0urd3wJRg95b26dp8SSO3/rTh5dPlCT94XJ48uX/B7CKTJa5Yv+N0QLJOH39fBsnavwRBx7X75vv/kExBpftn2k38oGOAraYBnuS+6DxwvwaPRmvwaDZmt3ZkBvlwHuFBPJp8breVrjYZM1u7SaEx+jYaUk58Y58gLj5i/lO0mPzE2d4GZYPetxrWbNDoEu68wrN2ssUfAG3Bp7X7CGnsE/R7F2UZ6tI0dwt5Dpo3dJ8LfbDlWGivm3dpuKY0V875z6RLMGtw76J3pDO69ANc6Y3WPCBn+wkUaq48JY3giJg3ukU8+ysgnH2ek5SvUGCYfaaTJhxr95GONbvLRxm7y4cb2iTjeWN08gLFaaYz/AWAIeJtzdhugAAAAAElFTkSuQmCC\") no-repeat;\n background-size: contain;\n pointer-events: none;\n z-index: 2147483647;\n transform: translate(-3px, -3px);\n transition: left 800ms ease-out, top 800ms ease-out, transform 0.08s ease-out;\n filter: drop-shadow(1px 1px 2px rgba(0,0,0,0.3));\n }\n #qp-cursor.clicking {\n transform: translate(-3px, -3px) scale(0.9);\n }\n";
252
- /**
253
- * JavaScript to inject for cursor tracking
254
- */
255
- declare const CURSOR_SCRIPT = "\n (function() {\n if (document.getElementById('qp-cursor')) return;\n const cursor = document.createElement('div');\n cursor.id = 'qp-cursor';\n document.body.appendChild(cursor);\n\n document.addEventListener('mousemove', (e) => {\n cursor.style.left = e.clientX + 'px';\n cursor.style.top = e.clientY + 'px';\n });\n\n document.addEventListener('mousedown', () => cursor.classList.add('clicking'));\n document.addEventListener('mouseup', () => cursor.classList.remove('clicking'));\n\n // Start at center\n cursor.style.left = window.innerWidth / 2 + 'px';\n cursor.style.top = window.innerHeight / 2 + 'px';\n })();\n";
256
- /**
257
- * Generate ASS subtitle file header
258
- */
259
- declare function generateASSHeader(width: number, height: number, font: string, size: number, primaryColor: string, secondaryColor: string, outlineColor: string, backColor: string, bold: boolean, outlineWidth: number, shadow: number, marginBottom: number): string;
260
-
261
- /**
262
- * Subscription and user registration client
263
- * CLI-only headless registration fallback. Tier/verification checks
264
- * are now in the SparkStripe gating module (getTierByEmail, isVerified).
265
- */
266
- interface RegistrationResult {
267
- success: boolean;
268
- pendingVerification?: boolean;
269
- alreadyVerified?: boolean;
270
- error?: string;
271
- }
272
- /**
273
- * Register user for free plan via SparkStripe
274
- */
275
- declare function registerFreeUser(email: string): Promise<RegistrationResult>;
276
-
277
653
  /** Minimal step shape needed by TTS modules — compatible with full Step from plan.ts */
278
654
  interface TTSStep {
279
655
  caption: string;
@@ -304,9 +680,111 @@ interface VoiceoverMeta {
304
680
  totalCaptionLength: number;
305
681
  stepCount: number;
306
682
  voice: string;
683
+ rate?: string;
307
684
  stepDurations: Record<number, number>;
685
+ stepWords?: Record<number, WordTiming[]>;
308
686
  }
309
687
 
688
+ /**
689
+ * CSS and ASS styling for QuickPeek
690
+ */
691
+
692
+ declare const CURSOR_CSS = "\n #qp-cursor {\n position: fixed;\n width: 32px;\n height: 38px;\n background: url(\"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAMgAAAEsBAMAAAB01OGNAAAAIGNIUk0AAHomAACAhAAA+gAAAIDoAAB1MAAA6mAAADqYAAAXcJy6UTwAAAAtUExURUdwTAAAAAAAAAAAAAAAAAAAAAAAAB0dHVZWVtvb2/X19f///6mpqXV1df////vm1ewAAAAGdFJOUwAltTfpcLO208EAAAABYktHRA5vvTBPAAAAB3RJTUUH6QwFFREKplBWTAAABqdJREFUeNq93c+OFFUUx/EeCHud6H7A4HpCOu6NiWsWZh5Ax8YFbAdNWBPxBcCo+yGjD+CAW1ywh0Rexu6uulX3Vp269/c7f7o3Bpjhk6+H6TrU0LdXR6sDPI4/OgCyvn0I5NsDpKw3B0hZbw6Qst4cIGWLxKdskfiUHRKeskPCU/ZIdMoeiU7pkOCUDglO6ZHYlB6JTUlIaEpCQlMGJDJlQCJTRiQwZUQCUzIkLiVD4lJyJCwlR8JSCiQqpUCiUkokKKVEglImSEzKBIlJmSIhKT3yY2hKjzx7GpnSI7+8iExJyN+RKQl5FZkyIJEpAxKZMiKBKSMSmJIhcSkZEpeSI2EpORKWUiBRKQUSlVIiQSklEpQyQWJSJkhMyhQJSZkiISkzJCJlhkSkzJGAlDkSkCIg/ikC4p8iIe4pEuKeIiLeKSLinSIjziky4pyygPimLCC+KUuIa8oS4pqyiHimLCKeKcuIY8oy4phSQfxSKohfSg1xS6khbilVxCulinil1BGnlDrilNJAfFIaiE9KC3FJaSEuKU3EI6WJeKS0EYeUNuKQAiD2FACxpyCIOQVBzCkQYk2BEGsKhhhTMMSYAiK2FBCxpaCIKQVFTCkwYkmBEUsKjhhScMSQQiD6FALRpzCIOoVB1CkUok2hEG0KhyhTOESZQiK6FBLRpbCIKoVFVCk0okmhEU0KjyhSeESRokD4FAXCp2gQOkWD0CkqhE1RIWyKDiFTdAiZokS4FCXCpWgRKkWLUClqhElRI0yKHiFS9AiRYkDwFAOCp1gQOMWCwCkmBE0xIWiKDQFTbAiYYkSwFCOCpVgRKMWKQClmBEkxI0iKHQFS7AiQ4oC0UxyQdooH0kzxQJopLkgrxQVppfggjRQfpJHihNRTnJB6ihdSTfFCqiluSC3FDaml+CGVFD+kkuKILKc4IsspnshiiieymOKKLKW4Iq8u5RRf5FpO8UUWpuKMyCnOiJzijYgp3oiY4o5IKe6IlEIhL38FHr9fzFK47/0On195PL6YpVDIX4CRP1IKhbxGSvLHbQUy/snhUrg/XW9IpE/hkOcs0qVwyD8s0qWQ349nJ9+lcAg9+S6FfFqhJ79PIZFLGtmlkMifPLJNYf/5wgWvfAYg11f5j951H39+B3/cBZA/3uY/+nccJ/5oItdPi19Lkz91RV5snuQ/fN0jJ57I9svvkTT5rzyR7RX7wW/C5L93RPbPI8Xknysmj7zWQZz8fTeke0J8aJ088vqTR1czl5t8Fel/w2Ly12/6oXghaRn8IEz+nJh8DRkuUc/yn02XYGLy0Ou0ysn3X44+/75rvNb+IE3+OxdkXM/Lr3l+8stIvjSIkz91QMaQzeYn2+QXkWL7KSafnogd/jlcHiJPHn8iXkLKNe7BW9Pk26/73T/+y3/tkp38AjLdR4vJp+XrxIhMQuRL8Jc2ZLZYF5OnL8Gt18ff6f9bTJ5dvkQku6vw6Ub4AHb5ap1ZcPS5MHn2Eiwhxe2Rs8rk0Utw8xyJ/gNMy5eAlPd5bkiTJ5ev5tkeN7+uTP6+FpncsEqTt6zd7fNWpMmTy9cMmd15+0SYPLl8tc/AuVWZPLh8TZH5LURx8unOFzb59rlEafKG5WuCSPdC0+SvhMljyxdwVtSx9DVPTb5ExPvTafIfpMmf8oh4pz1NXr92F8jCtz+kyVPLF3Km2jfdz+mXrxxZCLFPHjnnTpw8s3xlyFLIMHn18gWdPVibPLJ8jchiyPBB5eTf4ZOHzoNMky+eiInlCzrZMk1eWruR5Qs6o/NImjxxCcZOGz2rTB64BGPnpkqTJ5Yv7ARY4/LVIz9XQ4bJK5cv7FRe49oNni/cT165fIEnJdfW7vYTMXjm8w3pax6ePHh6tW35As/hltfui8ZniUglvDb55vKFno0urd3wJRg95b26dp8SSO3/rTh5dPlCT94XJ48uX/B7CKTJa5Yv+N0QLJOH39fBsnavwRBx7X75vv/kExBpftn2k38oGOAraYBnuS+6DxwvwaPRmvwaDZmt3ZkBvlwHuFBPJp8breVrjYZM1u7SaEx+jYaUk58Y58gLj5i/lO0mPzE2d4GZYPetxrWbNDoEu68wrN2ssUfAG3Bp7X7CGnsE/R7F2UZ6tI0dwt5Dpo3dJ8LfbDlWGivm3dpuKY0V875z6RLMGtw76J3pDO69ANc6Y3WPCBn+wkUaq48JY3giJg3ukU8+ysgnH2ek5SvUGCYfaaTJhxr95GONbvLRxm7y4cb2iTjeWN08gLFaaYz/AWAIeJtzdhugAAAAAElFTkSuQmCC\") no-repeat;\n background-size: contain;\n pointer-events: none;\n z-index: 2147483647;\n transform: translate(-3px, -3px);\n transition: left 800ms ease-out, top 800ms ease-out, transform 0.08s ease-out;\n filter: drop-shadow(1px 1px 2px rgba(0,0,0,0.3));\n }\n #qp-cursor.clicking {\n transform: translate(-3px, -3px) scale(0.9);\n }\n";
693
+ /**
694
+ * JavaScript to inject for cursor tracking
695
+ */
696
+ declare const CURSOR_SCRIPT = "\n (function() {\n if (document.getElementById('qp-cursor')) return;\n const cursor = document.createElement('div');\n cursor.id = 'qp-cursor';\n document.body.appendChild(cursor);\n\n document.addEventListener('mousemove', (e) => {\n cursor.style.left = e.clientX + 'px';\n cursor.style.top = e.clientY + 'px';\n });\n\n document.addEventListener('mousedown', () => cursor.classList.add('clicking'));\n document.addEventListener('mouseup', () => cursor.classList.remove('clicking'));\n\n // Start at center\n cursor.style.left = window.innerWidth / 2 + 'px';\n cursor.style.top = window.innerHeight / 2 + 'px';\n })();\n";
697
+ /**
698
+ * Generate ASS subtitle file header
699
+ */
700
+ declare function generateASSHeader(width: number, height: number, font: string, size: number, primaryColor: string, secondaryColor: string, outlineColor: string, backColor: string, bold: boolean, outlineWidth: number, shadow: number, marginBottom: number, alignment?: number): string;
701
+ /**
702
+ * Linux font-substitution guard (from VidLet): "Arial Black" is not installed
703
+ * on Linux, and fontconfig silently falls back to Noto Sans Regular with a
704
+ * synthesised weight. DejaVu Sans ships a real Bold face on every distro this
705
+ * runs on.
706
+ */
707
+ declare function resolveCaptionFont(font: string): string;
708
+ /** ASS numpad alignment: 1-3 bottom, 4-6 middle, 7-9 top. */
709
+ declare function positionToAlignment(position: CaptionPosition, leftAligned?: boolean): number;
710
+ /** One step's caption words, timed in absolute ms on the final video timeline. */
711
+ interface CaptionCue {
712
+ words: WordTiming[];
713
+ }
714
+ interface CaptionRenderContext {
715
+ cues: CaptionCue[];
716
+ width: number;
717
+ height: number;
718
+ captions: CaptionStyle;
719
+ }
720
+ /**
721
+ * The coordinate space captions are drawn in, which is NOT the output size.
722
+ *
723
+ * ASS declares its own PlayResX/PlayResY and libass scales that space onto
724
+ * whatever frame it burns into. Handing it the real output size made every
725
+ * caption number mean something different at every resolution: at 360x640 a
726
+ * 112px font goes from 5.8% of the frame height to 17.5%, and a 300px bottom
727
+ * margin - the "lower third" the preset intends - lands 47% of the way up,
728
+ * which is dead centre over the product. So the space is pinned to the size
729
+ * the numbers were authored for and libass does the scaling.
730
+ *
731
+ * A 1080x1920 render is unaffected: the scale factor is exactly 1.
732
+ */
733
+ declare function captionCanvas(width: number, height: number): {
734
+ width: number;
735
+ height: number;
736
+ };
737
+ /**
738
+ * The caption style a render actually uses: the canvas decides the geometry,
739
+ * captions.css supplies the face and the colours, `video.captions` overrides
740
+ * anything.
741
+ *
742
+ * The canvas wins over captions.css because the numbers in that file were
743
+ * never a preference. QuickPeek writes it on first run with 56px text 80px
744
+ * off the bottom - fine for a 16:9 tutorial, a desktop subtitle lost under
745
+ * the player chrome on the 9:16 short that is now the default shape - and no
746
+ * user ever chose those numbers. `--profile short` already overrode them for
747
+ * exactly this reason; the only change here is that a plain run gets the same
748
+ * treatment instead of depending on which command was typed.
749
+ *
750
+ * What the frame size genuinely cannot know stays with the CSS: the typeface
751
+ * and the colours, which are brand decisions. Anything else belongs to
752
+ * `video.captions` in quickpeek.config.json, which wins over both.
753
+ */
754
+ declare function resolveCaptions(css: CaptionStyle, video: Config['video']): CaptionStyle;
755
+ /** Render a complete ASS document for the configured caption preset. */
756
+ declare function renderCaptionsAss(context: CaptionRenderContext): string;
757
+
758
+ /**
759
+ * Subscription and user registration client
760
+ * CLI-only headless registration fallback. Tier/verification checks
761
+ * are now in the SparkStripe gating module (getTierByEmail, isVerified).
762
+ */
763
+ interface RegistrationResult {
764
+ success: boolean;
765
+ pendingVerification?: boolean;
766
+ alreadyVerified?: boolean;
767
+ error?: string;
768
+ }
769
+ /**
770
+ * Register user for free plan via SparkStripe
771
+ */
772
+ declare function registerFreeUser(email: string): Promise<RegistrationResult>;
773
+
774
+ /**
775
+ * Silence before the first word.
776
+ *
777
+ * Narration that starts on frame one talks over a viewer who is still taking
778
+ * in the opening card, and the first syllable lands before anyone is
779
+ * listening. Part of the audio rather than a compose-time delay, so the
780
+ * captions, the step durations and the dead-air pass all measure the same
781
+ * timeline.
782
+ *
783
+ * Here rather than beside the code that lays it down, because the
784
+ * realignment pass has to know about it too and importing across those two
785
+ * modules the other way would close a cycle.
786
+ */
787
+ declare const NARRATION_LEAD_IN_MS = 300;
310
788
  /**
311
789
  * Load existing voiceover metadata if available
312
790
  */
@@ -323,7 +801,7 @@ interface Step {
323
801
  /**
324
802
  * Check if existing voiceover can be reused
325
803
  */
326
- declare function canReuseVoiceover(steps: Step[], voice: string, existingMeta: VoiceoverMeta | null): {
804
+ declare function canReuseVoiceover(steps: Step[], voice: string, rate: string, existingMeta: VoiceoverMeta | null): {
327
805
  canReuse: boolean;
328
806
  reason?: string;
329
807
  };
@@ -347,17 +825,94 @@ declare function loadExistingVoiceover(steps: Step[], outputDir: string, meta: V
347
825
  }>;
348
826
 
349
827
  declare function resolveVoice(config: Config): string | null;
828
+ /** The narration pace for this config, as the enum the meta stores. */
829
+ declare function resolveRate(config: Config): VoiceRate;
830
+ /**
831
+ * Characters an SSML payload cannot carry.
832
+ *
833
+ * Both engines build SSML and drop the caption into it as-is, so a bare "&"
834
+ * makes the request invalid XML by the time the service sees it. It does not
835
+ * fail loudly: the Node engine returns an empty stream ("No audio data
836
+ * received"), the step is left with no clip, and because `stepWords` only
837
+ * records steps that produced words, the step vanishes from the metadata too.
838
+ * The video still renders - one step just has no voice.
839
+ *
840
+ * Found on Anysite's closing card, "Live Web Data for GTM & Marketing Teams":
841
+ * the whole closing line was silent while every other step in the same run
842
+ * synthesised fine. That is the worst shape a bug can take here, because the
843
+ * missing narration is at the end where nobody re-watches.
844
+ *
845
+ * Spoken, not escaped. "&" is read "and" by any English speaker, the burned
846
+ * caption still shows the symbol, and nothing downstream has to trust an
847
+ * engine to unescape an entity it may not have escaped itself.
848
+ */
849
+ declare function speakSymbols(text: string): string;
850
+ /**
851
+ * A caption as the voice should receive it.
852
+ *
853
+ * Order is load-bearing and each pass documents why above it: symbols before
854
+ * anything (a bare "&" invalidates the SSML), domains before toSpokenForm
855
+ * (matching "vidlet dot app" cannot tell an address from prose), acronyms
856
+ * after (so a TLD spelled here is not re-spelled), numbers last.
857
+ *
858
+ * Exported so the shaping can be asserted directly. It used to be one inline
859
+ * expression, which meant the only way to test a pronunciation was to
860
+ * re-create the chain in the test and hope it stayed in step.
861
+ */
862
+ declare function spokenForm(caption: string): string;
350
863
  declare function generateVoiceover(steps: TTSStep[], config: Config, workDir: string, outputDir: string, logError: (context: string, error: unknown) => Promise<void>, task: (text: string) => {
351
864
  succeed: (text?: string) => void;
352
865
  }, info: (text: string) => void): Promise<AudioResult>;
353
866
 
354
867
  /**
355
- * Synthesize speech using Python edge-tts (more reliable than Node.js msedge-tts)
356
- * Falls back to Node.js msedge-tts if Python is unavailable
868
+ * A word as the synthesizer actually spoke it, in ms from the start of the
869
+ * clip it was asked to produce — before any trimming this pipeline does to it.
870
+ */
871
+ interface SpokenWord {
872
+ word: string;
873
+ start: number;
874
+ end: number;
875
+ }
876
+ /** Starts a new run, so the next line chooses the engine again. */
877
+ declare function resetVoiceEngine(): void;
878
+ declare function synthesizeSpeech(rawText: string, outputPath: string, voice: string, rate?: string): Promise<SpokenWord[] | null>;
879
+ /**
880
+ * Pull the word boundaries out of the raw metadata stream.
881
+ *
882
+ * Each websocket frame is a complete JSON object, but a Readable is free to
883
+ * coalesce them into one chunk, so the concatenated text is split on brace
884
+ * depth rather than parsed whole. A frame that does not parse is skipped: a
885
+ * missing word costs a slightly-off highlight, throwing costs the clip.
357
886
  */
358
- declare function synthesizeSpeech(text: string, outputPath: string, voice: string): Promise<void>;
887
+ declare function parseWordBoundaries(raw: string): SpokenWord[] | null;
359
888
 
360
889
  declare function estimateWordTimings(caption: string, duration: number): WordTiming[];
890
+ /**
891
+ * Marry the authored caption text to the timings the voice reported.
892
+ *
893
+ * The burned-in captions must show the caption as WRITTEN, and the spoken word
894
+ * list does not always match it one for one — the text handed to the voice is
895
+ * the spoken form ("vidlet dot app" for a domain), and a number reads as
896
+ * several words. So the spoken words are used only as a clock. When the counts
897
+ * line up, each caption word takes its spoken counterpart's timing. When they
898
+ * don't, the caption words are spread evenly across the actual speech
899
+ * envelope — still better than a blind estimate, because the envelope excludes
900
+ * the silence the voice leaves at either end of a clip.
901
+ */
902
+ declare function alignWordsToCaption(caption: string, spoken: SpokenWord[]): WordTiming[];
903
+ /**
904
+ * Real word timings for a clip, from the boundaries the voice reported while
905
+ * speaking it. Null when the engine gave none, and callers fall back to
906
+ * estimateWordTimings. Timings are ms relative to the start of the clip.
907
+ *
908
+ * The engine times words against the audio it produced, but the clip on disk
909
+ * has since had its leading silence trimmed, so every offset is early by
910
+ * however much came off the front. The trim gate is set just under the noise
911
+ * edge-tts pads with, which puts the cut where the first word begins — so the
912
+ * first word's own offset IS what was removed, and rebasing on it re-anchors
913
+ * the whole clip without measuring the file again.
914
+ */
915
+ declare function spokenWordTimings(caption: string, spoken: SpokenWord[] | null): WordTiming[] | null;
361
916
  /**
362
917
  * Align voiceover to actual recording durations
363
918
  * If any step's recording took longer than its voiceover, pad with silence
@@ -377,6 +932,15 @@ declare function alignVoiceoverToRecording(stepAudio: Map<number, StepAudio>, ac
377
932
  declare function buildWatermarkFilters(opts: {
378
933
  tier: UserTier;
379
934
  videoDuration: number;
935
+ /**
936
+ * The video ends on a credits card that already names QuickPeek.
937
+ *
938
+ * The end text is drawn along the bottom of the last three seconds, which
939
+ * is exactly where that card carries the offer - two overlays stacked on
940
+ * the same strip, saying the same thing twice. The corner mark stays: it is
941
+ * the part that marks the whole video, not just its ending.
942
+ */
943
+ endsOnCreditsCard?: boolean;
380
944
  }): string;
381
945
 
382
- export { type ApplyTransitionsOptions, type AudioResult, type BlendTransition, CURSOR_CSS, CURSOR_PNG, CURSOR_SCRIPT, type ComposeOptions, type ConcatVideoOptions, Config, type LenientParse, type SpeedUpOptions, type SpliceVideosOptions, type StepAudio, type StepSync, type StepTransition, type TTSStep, type TimeSegment, UserTier, type VideoOverlay, type VoiceoverMeta, type WordTiming, alignVoiceoverToRecording, applyTransitions, buildConcatArgs, buildKeepSegments, buildStepTransitions, buildWatermarkFilters, canReuseVoiceover, composeVideo, concatMedia, concatVideoWithOutro, estimateWordTimings, extractJsonCandidates, generateASSHeader, generateSilence, generateVoiceover, getAudioCodec, getMediaDuration, getUpgradeUrl, isHonestWav, loadExistingVoiceover, loadVoiceoverMeta, mapTimestampsAfterCompression, parseAIJson, registerFreeUser, repairUnicodeEscapes, resolveVoice, salvageTruncatedJson, shouldWatermark, speedUpVideo, spliceVideos, synthesizeSpeech };
946
+ export { type ApplyTransitionsOptions, type AudioResult, type BlendTransition, CURSOR_CSS, CURSOR_PNG, CURSOR_SCRIPT, type CaptionCue, CaptionPosition, type CaptionRenderContext, CaptionStyle, type ComposeOptions, type ConcatVideoOptions, Config, EMPTY_SPAN_SPEED, GIF_DEFAULTS, type GifOptions, type LenientParse, MUSIC_BED_LUFS, NARRATION_LEAD_IN_MS, type SilenceSpan, type SpeedUpOptions, type SpliceVideosOptions, type SpokenWord, type StepAudio, type StepSync, type StepTransition, TRIM_EDGE_SILENCE, type TTSStep, type TimeSegment, UserTier, VOICE_LEAD_MS, type VideoOverlay, VoiceRate, type VoiceoverMeta, type WordTiming, alignVoiceoverToRecording, alignWordsToCaption, applyTransitions, atempoChain, buildCompressionFilter, buildConcatArgs, buildGifArgs, buildKeepSegments, buildPaletteArgs, buildRetimeFilter, buildStepTransitions, buildWatermarkFilters, canReuseVoiceover, captionCanvas, composeVideo, compressSilentSpans, concatMedia, concatVideoWithFullFrameOutro, concatVideoWithOutro, convertToGif, coverFilterChain, cutSpansFromVideo, detectBlackSpans, detectFrozenSpans, detectSilences, detectWhiteSpans, estimateWordTimings, extractJsonCandidates, generateASSHeader, generateSilence, generateVoiceover, getAudioCodec, getMediaDuration, getUpgradeUrl, getVideoDimensions, getVideoStreamDuration, hasAudioStream, intersectSpans, isHonestWav, loadExistingVoiceover, loadVoiceoverMeta, mapTimestampsAfterCompression, mergeSpans, parseAIJson, parseWordBoundaries, positionToAlignment, registerFreeUser, renderCaptionsAss, repairUnicodeEscapes, resetVoiceEngine, resolveCaptionFont, resolveCaptions, resolveRate, resolveVoice, retimeSpansInVideo, salvageTruncatedJson, speakSymbols, speedUpMiddle, speedUpVideo, spliceVideos, spokenForm, spokenWordTimings, synthesizeSpeech, trimEdgeSilence };