@spark-apps/quickpeek 1.2.3 → 1.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.mts CHANGED
@@ -1,6 +1,7 @@
1
- import { U as UserTier, C as Config } from './scraping-C334VxX5.mjs';
2
- export { A as AIError, a as AIResponse, b as AIResult, c as CLI_BACKOFF_MS, d as CONFIG_FILE, e as CaptionStyle, f as CaptionWordStyle, g as ChatOptions, D as DEFAULT_CAPTIONS, h as DEFAULT_CONFIG, E as EXCLUDED_LINK_PATTERNS, i as ElementInfo, j as ElementType, F as FEATURES, H as HighlightMode, I as INTERACTIVE_SELECTORS, P as Plan, R as RateLimitInfo, k as RetryNotice, S as SERVER_BACKOFF_MS, l as SparkStatus, m as SparkSubscription, n as SparkTrial, o as Step, p as SuggestedAction, T as Tier, V as VERSION, q as buildSystemPrompt, r as buildUserPrompt, s as callAI, t as callAIViaRelay, u as crawlPage, v as getErrorMessage, w as getLanguageName, x as getStatus, y as getStatusCached, z as getTierByEmail, B as hasAccess, G as hasTrialRemaining, J as idSelector, K as isCanceling, L as isExcludedLink, M as isPaid, N as isVerified, O as normalizeUrl, Q as parseAIPlanResponse, W as parsePartial, X as pricingUrl, Y as rgbaToHex, Z as shouldSkipLink, _ as showPaywall, $ as toTitleCase } from './scraping-C334VxX5.mjs';
3
- export { WIDE_VOICES as LANGUAGE_VOICES, hexToAss as hexToASS, windowsDrivePathToWsl as windowsPathToWSL } from '@spark-apps/video-kit';
1
+ import { U as UserTier, C as CaptionStyle, a as CaptionPosition, b as Config, V as VoiceRate } from './voices-CPpnWn39.mjs';
2
+ export { A as AIError, c as AIResponse, d as AIResult, e as CLI_BACKOFF_MS, f as CONFIG_FILE, g as CaptionPreset, h as CaptionWordStyle, i as ChatOptions, D as DEFAULT_CAPTIONS, j as DEFAULT_CONFIG, E as EXCLUDED_LINK_PATTERNS, k as ElementInfo, l as ElementType, H as HighlightMode, I as INTERACTIVE_SELECTORS, m as INTERNATIONAL_VOICE, N as NO_CLIP_OUTRO, P as PLAN_MAX_TOKENS, n as Plan, R as RATE_FACTOR, o as RateLimitInfo, p as RetryNotice, S as SERVER_BACKOFF_MS, q as SIZE_PRESETS, r as SparkStatus, s as SparkSubscription, t as SparkTrial, u as Step, v as SuggestedAction, T as Tier, w as VERSION, x as VIDEO_PROFILES, y as VideoProfileName, z as VideoSize, B as VoiceGender, F as applyProfile, G as ariaLabelSelector, J as buildSystemPrompt, K as buildUserPrompt, L as callAI, M as callAIViaRelay, O as crawlPage, Q as deStock, W as defaultCaptionsFor, X as getErrorMessage, Y as getLanguageName, Z as getStatus, _ as getStatusCached, $ as getTierByEmail, a0 as gotoSettled, a1 as hasTrialRemaining, a2 as hasVoice, a3 as idSelector, a4 as internationalVoiceFor, a5 as isCanceling, a6 as isExcludedLink, a7 as isPaid, a8 as isVerified, a9 as normalizeUrl, aa as parseAIPlanResponse, ab as parsePartial, ac as pricingUrl, ad as rgbaToHex, ae as runPooled, af as shouldSkipLink, ag as showPaywall, ah as stripLongDashes, ai as toTitleCase, aj as voiceFor, ak as waitForPlaceholdersGone } from './voices-CPpnWn39.mjs';
3
+ export { MALE_VOICES, MULTILINGUAL_VOICES, WIDE_VOICES, hexToAss as hexToASS, windowsDrivePathToWsl as windowsPathToWSL } from '@spark-apps/video-kit';
4
+ import 'playwright';
4
5
 
5
6
  /**
6
7
  * Billing utilities — watermark decisions and upgrade URL generation
@@ -33,14 +34,27 @@ interface StepSync {
33
34
  targetDuration: number;
34
35
  }
35
36
  interface ComposeOptions {
37
+ /** Input 0 carries a spliced clip's own audio, to be mixed under the voice. */
38
+ hasClipAudio?: boolean;
36
39
  videoPath: string;
37
- assPath: string;
40
+ /** Burn-in captions. Omit to skip the ass filter entirely (draft). */
41
+ assPath?: string | undefined;
38
42
  outputPath: string;
39
43
  voiceoverPath?: string | undefined;
40
44
  musicPath?: string | undefined;
41
45
  musicVolume?: number | undefined;
42
46
  musicFadeIn?: number | undefined;
43
47
  musicFadeOut?: number | undefined;
48
+ /**
49
+ * Loudness the bed is normalised to before it meets the narration, in LUFS.
50
+ *
51
+ * A fixed `musicVolume` cannot produce a consistent bed, because the source
52
+ * decides everything: two CC0 tracks in the same folder measure -31 dB and
53
+ * -15 dB mean, so the same gain makes one inaudible and the other a
54
+ * competitor. Normalising first makes the bed a level rather than a
55
+ * multiplier, and `musicVolume` goes back to being the trim it reads as.
56
+ */
57
+ musicTargetLufs?: number | undefined;
44
58
  width: number;
45
59
  height: number;
46
60
  fps: number;
@@ -53,9 +67,12 @@ interface ComposeOptions {
53
67
  contrast?: number | undefined;
54
68
  trimStart?: number | undefined;
55
69
  userTier?: UserTier | undefined;
70
+ endsOnCreditsCard?: boolean | undefined;
56
71
  mainContentDuration?: number | undefined;
57
72
  stepSync?: StepSync[] | undefined;
58
73
  stepTransitions?: StepTransition[] | undefined;
74
+ /** 'off' pins the deliverable encode to libx264; anything else probes for a GPU. */
75
+ hwaccel?: 'auto' | 'off' | undefined;
59
76
  }
60
77
  interface ConcatVideoOptions {
61
78
  inputPath: string;
@@ -69,7 +86,27 @@ interface VideoOverlay {
69
86
  path: string;
70
87
  startTime: number;
71
88
  duration: number;
72
- mode: 'stretch' | 'center';
89
+ /**
90
+ * How the source is fitted to the frame.
91
+ *
92
+ * stretch = fill the frame's width, crop the height if it overflows.
93
+ * center = scale to fit, letterbox the rest.
94
+ * cover = fill the frame with a blurred, enlarged copy of the source and
95
+ * sit the source itself on top of it, untouched. A 1200x630 OG
96
+ * card in a 9:16 frame is 16:9 either way; the only question is
97
+ * what fills the two thirds of the frame it cannot reach, and a
98
+ * blurred continuation of the card reads as design where a black
99
+ * band reads as a mistake.
100
+ */
101
+ mode: 'stretch' | 'center' | 'cover';
102
+ /**
103
+ * The source is a still image (png/jpg/webp), not a clip.
104
+ *
105
+ * A still has exactly one frame, so it has to be looped for the length of
106
+ * its window or the overlay holds a single frame at EOF and the window
107
+ * plays whatever is underneath it.
108
+ */
109
+ still?: boolean;
73
110
  trimStart?: number;
74
111
  trimEnd?: number;
75
112
  }
@@ -105,6 +142,34 @@ interface TimeSegment {
105
142
  end: number;
106
143
  }
107
144
 
145
+ /**
146
+ * A beat of silence before the narrator starts, in milliseconds.
147
+ *
148
+ * The voice used to be faded in over the same duration as the picture, which
149
+ * ate the first syllable - and the first syllable belongs to the hook, the one
150
+ * line that has to land. A short delay gives the viewer the same moment to
151
+ * settle without softening the words.
152
+ */
153
+ declare const VOICE_LEAD_MS = 200;
154
+ /**
155
+ * Loudness the music bed is normalised to before the mix, in LUFS.
156
+ *
157
+ * Measured, not guessed. Running this exact graph over the accordio
158
+ * narration puts the bed at -26.7 dB mean in a gap between lines while the
159
+ * speech measures -13.7 dB: thirteen dB under the voice, which is a bed. Each
160
+ * dB here moves that reading a dB, so the number is a real control - -38 gives
161
+ * -30.6, -42 gives -34.7.
162
+ *
163
+ * Over a long passage with no narration at all - the closing card - the master
164
+ * chain's single-pass loudnorm is dynamic and opens up, and the bed rises to
165
+ * about -18 dB. That is the intended shape: discreet under the voice, present
166
+ * where there is nothing else.
167
+ *
168
+ * What it replaced measured -39 dB at source and -45 dB after its own volume
169
+ * trim, and read in the finished file at -32 dB against -15 dB of speech.
170
+ * That is not a quiet bed, it is no bed: the video was speech in a vacuum.
171
+ */
172
+ declare const MUSIC_BED_LUFS = -34;
108
173
  /**
109
174
  * Compose final video with subtitles, voiceover, and music
110
175
  */
@@ -120,6 +185,45 @@ declare function buildKeepSegments(eventTimestampsSeconds: number[], totalDurati
120
185
  */
121
186
  declare function mapTimestampsAfterCompression(originalTimestampsSeconds: number[], segments: TimeSegment[]): number[];
122
187
 
188
+ /**
189
+ * Video → optimized GIF conversion (ported from VidLet's togif tool).
190
+ * Two-pass: generate a tuned palette, then map the video through it —
191
+ * dramatically better colour than ffmpeg's default 256-colour GIF path.
192
+ */
193
+ interface GifOptions {
194
+ input: string;
195
+ output: string;
196
+ fps?: number;
197
+ width?: number;
198
+ dither?: string;
199
+ statsMode?: string;
200
+ }
201
+ declare const GIF_DEFAULTS: {
202
+ readonly fps: 15;
203
+ readonly width: 480;
204
+ readonly dither: "sierra2_4a";
205
+ readonly statsMode: "full";
206
+ };
207
+ declare function buildPaletteArgs(input: string, palettePath: string, fps: number, width: number, statsMode: string): string[];
208
+ declare function buildGifArgs(input: string, palettePath: string, output: string, fps: number, width: number, dither: string): string[];
209
+ declare function convertToGif(options: GifOptions): Promise<string>;
210
+
211
+ /**
212
+ * The video stream's pixel dimensions.
213
+ *
214
+ * Needed because the splice pass scales its overlays to match the video it is
215
+ * pasting them onto, and that video is the raw recording, which is smaller
216
+ * than the configured output whenever video.zoom is in play: a short records
217
+ * at 540x960 and is upscaled to 1080x1920 only at compose. Scaling overlays to
218
+ * the output size instead put them in at 2x and cropped everything but the
219
+ * middle of each card.
220
+ */
221
+ declare function getVideoDimensions(filePath: string): Promise<{
222
+ width: number;
223
+ height: number;
224
+ }>;
225
+ /** Whether a file carries an audio stream at all. */
226
+ declare function hasAudioStream(filePath: string): Promise<boolean>;
123
227
  /**
124
228
  * Run ffprobe to get media duration in milliseconds
125
229
  */
@@ -146,6 +250,22 @@ declare function isHonestWav(filePath: string): Promise<boolean>;
146
250
  */
147
251
  declare function buildConcatArgs(listPath: string, output: string): string[];
148
252
  declare function concatMedia(listPath: string, output: string): Promise<void>;
253
+ /**
254
+ * Strip the silence a TTS engine leaves on both ends of a clip.
255
+ *
256
+ * Every edge-tts clip arrives with its own lead-in and tail, typically a few
257
+ * hundred ms each. One clip per caption, concatenated, means each boundary
258
+ * stacks tail + the pad that aligns the step + the next clip's lead-in - the
259
+ * "robotic pause" between sentences. Trimming here rather than at concat time
260
+ * keeps the clip the unit of truth: the duration measured after this call,
261
+ * the padding computed from it, and the word timings rebased in
262
+ * spokenWordTimings all describe the same audio.
263
+ *
264
+ * -45dB rather than a hard zero: edge-tts pads with near-silent noise, not
265
+ * digital black, so a stricter gate finds nothing to remove.
266
+ */
267
+ declare const TRIM_EDGE_SILENCE: string;
268
+ declare function trimEdgeSilence(filePath: string): Promise<void>;
149
269
  /**
150
270
  * Generate silence audio file of specified duration
151
271
  */
@@ -160,6 +280,262 @@ declare function speedUpVideo(opts: SpeedUpOptions): Promise<void>;
160
280
  * Returns the combined video for further processing (composition with audio)
161
281
  */
162
282
  declare function concatVideoWithOutro(opts: ConcatVideoOptions): Promise<void>;
283
+ /**
284
+ * Append an outro that is already a full frame, unchanged.
285
+ *
286
+ * concatVideoWithOutro treats its outro as a badge: it shrinks it to 60% and
287
+ * overlays it on black, which is right for the bundled square watermark and
288
+ * wrong for a rendered credits card, which is composed at the video's exact
289
+ * size and must not be shrunk inside its own frame.
290
+ *
291
+ * setsar on both inputs because concat refuses streams whose sample aspect
292
+ * ratios disagree, and a card rendered by the browser does not necessarily
293
+ * carry the same SAR as a recorded page.
294
+ */
295
+ declare function concatVideoWithFullFrameOutro(opts: ConcatVideoOptions): Promise<void>;
296
+
297
+ /**
298
+ * Compressing dead air out of a finished demo.
299
+ *
300
+ * A step whose narration ends before the step does leaves the video sitting
301
+ * on a still frame with nothing being said - padding that keeps audio and
302
+ * video in sync, plus any deliberate hold while an async result loads. Held
303
+ * for a second it reads as a beat; held for five it reads as a hang.
304
+ *
305
+ * This runs AFTER compose on purpose. By then the captions are burned in and
306
+ * the silent spans have none on screen (a karaoke line ends with its last
307
+ * word), so speeding those spans up removes the dead time without touching
308
+ * caption timing, the recorder, or the TTS timeline - the three things that
309
+ * have to agree with each other. What is sped up is silence, so it is
310
+ * inaudible, and the visuals still play through: a result that appears during
311
+ * a hold is still seen, just briskly.
312
+ */
313
+ /** A silent stretch of the mixed audio, in seconds. */
314
+ interface SilenceSpan {
315
+ start: number;
316
+ end: number;
317
+ /** Longest this span may run after compression. Defaults to the pass's cap. */
318
+ cap?: number;
319
+ }
320
+ /**
321
+ * Length of the video stream in seconds, or 0 if it cannot be read.
322
+ *
323
+ * Distinct from the container's duration, which reports whichever stream runs
324
+ * longest - usually the audio, since compose pads it to the voiceover.
325
+ */
326
+ declare function getVideoStreamDuration(inputPath: string): Promise<number>;
327
+ declare function detectSilences(inputPath: string, minDurSecs: number): Promise<SilenceSpan[]>;
328
+ /**
329
+ * Find stretches where the picture stops changing.
330
+ *
331
+ * Silence is not the only kind of dead air. A page that finishes an animation
332
+ * - a confetti burst settling, a spinner resolving - holds a still frame for
333
+ * as long as the step is paced to last, and that reads as a hang even with
334
+ * narration over it. freezedetect reports those stretches; they are treated
335
+ * like silence, and for the same reason.
336
+ */
337
+ declare function detectFrozenSpans(inputPath: string, minDurSecs: number): Promise<SilenceSpan[]>;
338
+ /**
339
+ * Spans where the picture is essentially black.
340
+ *
341
+ * A frozen span and a black span are not the same thing, and the difference
342
+ * matters. A frozen span is a still screen: the viewer is looking at something,
343
+ * so it earns the FROZEN_MAX_SECS ceiling and gets shortened, not cut. A black
344
+ * span is a page mid-transition, and it is worth nothing at any length. One run
345
+ * shipped 9.4 seconds of it in a 75 second video.
346
+ *
347
+ * Playwright records continuously and cannot be paused across a navigation, so
348
+ * these frames cannot be prevented at capture time. They are removed here
349
+ * instead, and only where the audio is silent too, so no narration is ever cut.
350
+ */
351
+ declare function detectBlackSpans(inputPath: string, minDurSecs: number): Promise<SilenceSpan[]>;
352
+ /**
353
+ * Spans where the picture is essentially white.
354
+ *
355
+ * blackdetect over an inverted picture. A page that flashes white between
356
+ * routes is as empty as one that flashes black, and a light-themed app makes
357
+ * white the common case, so detecting only one of the two fixes half the sites.
358
+ */
359
+ declare function detectWhiteSpans(inputPath: string, minDurSecs: number): Promise<SilenceSpan[]>;
360
+ /**
361
+ * Where two sets of spans overlap.
362
+ *
363
+ * A frozen picture is only safe to speed up while nothing is being said over
364
+ * it - rushing a still frame is free, rushing a sentence is not. The overlap
365
+ * of "frozen" and "silent" is exactly the footage with nothing happening in
366
+ * either channel.
367
+ */
368
+ declare function intersectSpans(a: SilenceSpan[], b: SilenceSpan[]): SilenceSpan[];
369
+ /** Fold overlapping or touching spans into one. */
370
+ declare function mergeSpans(spans: SilenceSpan[]): SilenceSpan[];
371
+ /** atempo caps at 2x per instance, so a bigger speed-up is a chain of them. */
372
+ declare function atempoChain(factor: number): string;
373
+ /**
374
+ * Build the filter_complex that plays `spans` fast and the rest untouched.
375
+ *
376
+ * Split out from the run so the graph can be unit-tested: an ffmpeg filter
377
+ * error surfaces as a wall of text at render time, which is a bad place to
378
+ * discover an off-by-one in the segment list.
379
+ */
380
+ declare function buildCompressionFilter(spans: SilenceSpan[], totalSecs: number, maxSecs: number, protectAfterSecs?: number,
381
+ /**
382
+ * Everything BEFORE this is the opening card, protected for the same reason
383
+ * as the closing one and previously not protected at all.
384
+ *
385
+ * It did not need to be while the intro card was narrated over: a spoken
386
+ * card is not a silent span. A card prepended as its own step has no
387
+ * narration, so it is silent AND frozen, and this pass gave it the frozen
388
+ * ceiling of 0.6s - a title card gone before it can be read.
389
+ */
390
+ protectBeforeSecs?: number): string | null;
391
+ /**
392
+ * Drop spans from a silent video, in its own timeline.
393
+ *
394
+ * The recording has no audio yet, so there is nothing to desync: this is the
395
+ * one place in the pipeline where a span can simply be deleted. Doing it here
396
+ * rather than after compose also avoids the coordinate problem that made an
397
+ * earlier attempt a no-op. A recording is 145s where its composed video is
398
+ * 74s, because compose trims the page load off the front and the later passes
399
+ * squeeze the rest, so a span measured during recording means nothing once
400
+ * those have run.
401
+ */
402
+ declare function cutSpansFromVideo(inputPath: string, outputPath: string, spans: Array<{
403
+ start: number;
404
+ end: number;
405
+ }>, preset: string): Promise<boolean>;
406
+ /**
407
+ * How much faster a loading screen plays than the footage around it.
408
+ *
409
+ * The empty stretches in a recording are almost always one thing: the app's
410
+ * own splash while a route loads. A sellular run spent 17 of its 66 seconds
411
+ * on that screen. Five is fast enough that it reads as a flicker rather than
412
+ * a wait, and slow enough that the logo is still legible going past.
413
+ */
414
+ declare const EMPTY_SPAN_SPEED = 5;
415
+ /**
416
+ * Build the filter that removes every empty span from view without changing
417
+ * how long anything lasts, one step at a time.
418
+ *
419
+ * The point of doing it per step is that every step boundary comes out exactly
420
+ * where it went in. Cutting empty frames is the obvious move and it does not
421
+ * work: the narration for each step is already synthesised at a fixed length,
422
+ * so shortening the picture under it desyncs everything downstream, and the
423
+ * guard that stops that from happening starves the cut until it removes
424
+ * nothing at all. So the seconds are not removed, they are re-spent.
425
+ *
426
+ * A step with real footage either side of the splash plays the splash at
427
+ * `speed` and stretches that footage to fill exactly what it freed. A step
428
+ * that is nearly all splash - a route whose loading screen outlasts the line
429
+ * narrated over it - has nothing to stretch, so the splash is replaced by a
430
+ * still of the page that arrives after it. Both leave the step the length it
431
+ * was, and neither leaves an empty screen up.
432
+ *
433
+ * Returns null when there is nothing worth doing, so the caller can skip a
434
+ * re-encode that would change nothing.
435
+ */
436
+ declare function buildRetimeFilter(spans: Array<{
437
+ start: number;
438
+ end: number;
439
+ }>, stepBounds: Array<{
440
+ from: number;
441
+ to: number;
442
+ }>, speed: number, fps: number): string | null;
443
+ /**
444
+ * Play the empty stretches of a silent recording fast, keeping every step the
445
+ * length it was.
446
+ *
447
+ * Resolves false when the filter had nothing to do or ffmpeg refused it, and
448
+ * the caller keeps the original recording.
449
+ */
450
+ declare function retimeSpansInVideo(inputPath: string, outputPath: string, spans: Array<{
451
+ start: number;
452
+ end: number;
453
+ }>, stepBounds: Array<{
454
+ from: number;
455
+ to: number;
456
+ }>, speed: number, fps: number, preset: string): Promise<boolean>;
457
+ declare function compressSilentSpans(inputPath: string, outputPath: string, totalSecs: number, maxSecs: number, preset: string, protectAfterSecs?: number,
458
+ /**
459
+ * The narration on its own, when there is one.
460
+ *
461
+ * Silence has to be measured against the voice, not against the mix: any
462
+ * backing track - a music bed, room tone - is continuous by design, so a
463
+ * mixed file has no silence in it at all and every pause looks like speech.
464
+ */
465
+ narrationPath?: string,
466
+ /** Opening card to leave alone, in seconds from the start. See the filter. */
467
+ protectBeforeSecs?: number,
468
+ /**
469
+ * Spans the recorder KNOWS were a page loading rather than a page showing.
470
+ *
471
+ * Cut on the recorder's word, not on a detector's guess. blackdetect cannot
472
+ * tell a loading screen from a dark-themed app, and the silence guard
473
+ * refuses to cut anything with a voice over it, which is exactly the case
474
+ * here: the narration describing a result is queued behind the wait for it.
475
+ * The recorder measured this span, so it needs no evidence.
476
+ */
477
+ leadSpans?: Array<{
478
+ start: number;
479
+ end: number;
480
+ }>,
481
+ /**
482
+ * Spans that are playing CONTENT, however quiet they are.
483
+ *
484
+ * A clip spliced in under no narration is silent by nature and frozen to a
485
+ * freeze detector between cuts, so every heuristic here reads it as dead
486
+ * air and compresses it to the ceiling. A nine-second demo arrived as
487
+ * three-quarters of a second. Silence during a clip is the clip, not a
488
+ * hang, so these spans are removed from the compression list outright.
489
+ */
490
+ keepSpans?: Array<{
491
+ start: number;
492
+ end: number;
493
+ }>): Promise<boolean>;
494
+ /**
495
+ * Speed up only the middle of a video, leaving its ends untouched.
496
+ *
497
+ * A demo opens on a title card and closes on a credits card. Speeding the
498
+ * whole file to fit a length limit rushes both: the card the viewer is meant
499
+ * to read, and the offer they are meant to write down. Only the demo between
500
+ * them is compressible.
501
+ *
502
+ * Returns false when the ends already fill the budget - there is nothing to
503
+ * squeeze then, and the honest answer is a shorter plan.
504
+ */
505
+ declare function speedUpMiddle(inputPath: string, outputPath: string, totalSecs: number, targetSecs: number, introSecs: number, outroSecs: number, maxFactor: number, preset: string): Promise<{
506
+ ok: boolean;
507
+ factor: number;
508
+ reachedTarget: boolean;
509
+ }>;
510
+
511
+ /**
512
+ * Fill the frame with the card's own colours, then lay the card on it.
513
+ *
514
+ * A 1200x630 OG card is the shape every site publishes and the wrong shape for
515
+ * a 9:16 opening: centred, it floats in two thirds of a frame of flat #1a1a2e,
516
+ * which is what "letterboxed with dead space" looks like. Cropping it to fill
517
+ * instead would throw away the half of the card the headline is written
518
+ * across. So the backdrop is built from the card itself - the frame is full,
519
+ * the colours are the card's own, and not a pixel of the card is lost.
520
+ *
521
+ * The backdrop is RESAMPLED DOWN to a handful of pixels before it is blown
522
+ * back up, rather than blurred at full resolution. Blur is the obvious move
523
+ * and it is wrong here, because an OG card is mostly TYPE: a 1200px-wide
524
+ * headline enlarged 3x stays legible through sigma=28, so the opening frame
525
+ * showed the same words twice - huge and soft behind the card, sharp on it.
526
+ * That reads as a rendering fault rather than a design, and it was the first
527
+ * thing anyone saw of the product.
528
+ *
529
+ * Ten pixels of height cannot carry a glyph, so nothing survives to be read
530
+ * while the card's colour field does. The gblur afterwards only takes the
531
+ * blockiness off the upscale.
532
+ *
533
+ * The downscale is its own `scale` with BOTH dimensions given. `scale=40:-2`
534
+ * is silently ignored at these sizes - the first attempt at this looked
535
+ * completely unchanged because of it, which is why the test asserts the
536
+ * literal pair rather than just "smaller than the frame".
537
+ */
538
+ declare function coverFilterChain(inputIdx: number, width: number, height: number): string;
163
539
 
164
540
  /**
165
541
  * Splice video files into the recording at specific timestamps
@@ -248,32 +624,6 @@ interface LenientParse<T> {
248
624
  */
249
625
  declare function parseAIJson<T>(content: string): LenientParse<T> | null;
250
626
 
251
- declare const CURSOR_CSS = "\n #qp-cursor {\n position: fixed;\n width: 32px;\n height: 38px;\n background: url(\"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAMgAAAEsBAMAAAB01OGNAAAAIGNIUk0AAHomAACAhAAA+gAAAIDoAAB1MAAA6mAAADqYAAAXcJy6UTwAAAAtUExURUdwTAAAAAAAAAAAAAAAAAAAAAAAAB0dHVZWVtvb2/X19f///6mpqXV1df////vm1ewAAAAGdFJOUwAltTfpcLO208EAAAABYktHRA5vvTBPAAAAB3RJTUUH6QwFFREKplBWTAAABqdJREFUeNq93c+OFFUUx/EeCHud6H7A4HpCOu6NiWsWZh5Ax8YFbAdNWBPxBcCo+yGjD+CAW1ywh0Rexu6uulX3Vp269/c7f7o3Bpjhk6+H6TrU0LdXR6sDPI4/OgCyvn0I5NsDpKw3B0hZbw6Qst4cIGWLxKdskfiUHRKeskPCU/ZIdMoeiU7pkOCUDglO6ZHYlB6JTUlIaEpCQlMGJDJlQCJTRiQwZUQCUzIkLiVD4lJyJCwlR8JSCiQqpUCiUkokKKVEglImSEzKBIlJmSIhKT3yY2hKjzx7GpnSI7+8iExJyN+RKQl5FZkyIJEpAxKZMiKBKSMSmJIhcSkZEpeSI2EpORKWUiBRKQUSlVIiQSklEpQyQWJSJkhMyhQJSZkiISkzJCJlhkSkzJGAlDkSkCIg/ikC4p8iIe4pEuKeIiLeKSLinSIjziky4pyygPimLCC+KUuIa8oS4pqyiHimLCKeKcuIY8oy4phSQfxSKohfSg1xS6khbilVxCulinil1BGnlDrilNJAfFIaiE9KC3FJaSEuKU3EI6WJeKS0EYeUNuKQAiD2FACxpyCIOQVBzCkQYk2BEGsKhhhTMMSYAiK2FBCxpaCIKQVFTCkwYkmBEUsKjhhScMSQQiD6FALRpzCIOoVB1CkUok2hEG0KhyhTOESZQiK6FBLRpbCIKoVFVCk0okmhEU0KjyhSeESRokD4FAXCp2gQOkWD0CkqhE1RIWyKDiFTdAiZokS4FCXCpWgRKkWLUClqhElRI0yKHiFS9AiRYkDwFAOCp1gQOMWCwCkmBE0xIWiKDQFTbAiYYkSwFCOCpVgRKMWKQClmBEkxI0iKHQFS7AiQ4oC0UxyQdooH0kzxQJopLkgrxQVppfggjRQfpJHihNRTnJB6ihdSTfFCqiluSC3FDaml+CGVFD+kkuKILKc4IsspnshiiieymOKKLKW4Iq8u5RRf5FpO8UUWpuKMyCnOiJzijYgp3oiY4o5IKe6IlEIhL38FHr9fzFK47/0On195PL6YpVDIX4CRP1IKhbxGSvLHbQUy/snhUrg/XW9IpE/hkOcs0qVwyD8s0qWQ349nJ9+lcAg9+S6FfFqhJ79PIZFLGtmlkMifPLJNYf/5wgWvfAYg11f5j951H39+B3/cBZA/3uY/+nccJ/5oItdPi19Lkz91RV5snuQ/fN0jJ57I9svvkTT5rzyR7RX7wW/C5L93RPbPI8Xknysmj7zWQZz8fTeke0J8aJ088vqTR1czl5t8Fel/w2Ly12/6oXghaRn8IEz+nJh8DRkuUc/yn02XYGLy0Ou0ysn3X44+/75rvNb+IE3+OxdkXM/Lr3l+8stIvjSIkz91QMaQzeYn2+QXkWL7KSafnogd/jlcHiJPHn8iXkLKNe7BW9Pk26/73T/+y3/tkp38AjLdR4vJp+XrxIhMQuRL8Jc2ZLZYF5OnL8Gt18ff6f9bTJ5dvkQku6vw6Ub4AHb5ap1ZcPS5MHn2Eiwhxe2Rs8rk0Utw8xyJ/gNMy5eAlPd5bkiTJ5ev5tkeN7+uTP6+FpncsEqTt6zd7fNWpMmTy9cMmd15+0SYPLl8tc/AuVWZPLh8TZH5LURx8unOFzb59rlEafKG5WuCSPdC0+SvhMljyxdwVtSx9DVPTb5ExPvTafIfpMmf8oh4pz1NXr92F8jCtz+kyVPLF3Km2jfdz+mXrxxZCLFPHjnnTpw8s3xlyFLIMHn18gWdPVibPLJ8jchiyPBB5eTf4ZOHzoNMky+eiInlCzrZMk1eWruR5Qs6o/NImjxxCcZOGz2rTB64BGPnpkqTJ5Yv7ARY4/LVIz9XQ4bJK5cv7FRe49oNni/cT165fIEnJdfW7vYTMXjm8w3pax6ePHh6tW35As/hltfui8ZniUglvDb55vKFno0urd3wJRg95b26dp8SSO3/rTh5dPlCT94XJ48uX/B7CKTJa5Yv+N0QLJOH39fBsnavwRBx7X75vv/kExBpftn2k38oGOAraYBnuS+6DxwvwaPRmvwaDZmt3ZkBvlwHuFBPJp8breVrjYZM1u7SaEx+jYaUk58Y58gLj5i/lO0mPzE2d4GZYPetxrWbNDoEu68wrN2ssUfAG3Bp7X7CGnsE/R7F2UZ6tI0dwt5Dpo3dJ8LfbDlWGivm3dpuKY0V875z6RLMGtw76J3pDO69ANc6Y3WPCBn+wkUaq48JY3giJg3ukU8+ysgnH2ek5SvUGCYfaaTJhxr95GONbvLRxm7y4cb2iTjeWN08gLFaaYz/AWAIeJtzdhugAAAAAElFTkSuQmCC\") no-repeat;\n background-size: contain;\n pointer-events: none;\n z-index: 2147483647;\n transform: translate(-3px, -3px);\n transition: left 800ms ease-out, top 800ms ease-out, transform 0.08s ease-out;\n filter: drop-shadow(1px 1px 2px rgba(0,0,0,0.3));\n }\n #qp-cursor.clicking {\n transform: translate(-3px, -3px) scale(0.9);\n }\n";
252
- /**
253
- * JavaScript to inject for cursor tracking
254
- */
255
- declare const CURSOR_SCRIPT = "\n (function() {\n if (document.getElementById('qp-cursor')) return;\n const cursor = document.createElement('div');\n cursor.id = 'qp-cursor';\n document.body.appendChild(cursor);\n\n document.addEventListener('mousemove', (e) => {\n cursor.style.left = e.clientX + 'px';\n cursor.style.top = e.clientY + 'px';\n });\n\n document.addEventListener('mousedown', () => cursor.classList.add('clicking'));\n document.addEventListener('mouseup', () => cursor.classList.remove('clicking'));\n\n // Start at center\n cursor.style.left = window.innerWidth / 2 + 'px';\n cursor.style.top = window.innerHeight / 2 + 'px';\n })();\n";
256
- /**
257
- * Generate ASS subtitle file header
258
- */
259
- declare function generateASSHeader(width: number, height: number, font: string, size: number, primaryColor: string, secondaryColor: string, outlineColor: string, backColor: string, bold: boolean, outlineWidth: number, shadow: number, marginBottom: number): string;
260
-
261
- /**
262
- * Subscription and user registration client
263
- * CLI-only headless registration fallback. Tier/verification checks
264
- * are now in the SparkStripe gating module (getTierByEmail, isVerified).
265
- */
266
- interface RegistrationResult {
267
- success: boolean;
268
- pendingVerification?: boolean;
269
- alreadyVerified?: boolean;
270
- error?: string;
271
- }
272
- /**
273
- * Register user for free plan via SparkStripe
274
- */
275
- declare function registerFreeUser(email: string): Promise<RegistrationResult>;
276
-
277
627
  /** Minimal step shape needed by TTS modules — compatible with full Step from plan.ts */
278
628
  interface TTSStep {
279
629
  caption: string;
@@ -304,9 +654,111 @@ interface VoiceoverMeta {
304
654
  totalCaptionLength: number;
305
655
  stepCount: number;
306
656
  voice: string;
657
+ rate?: string;
307
658
  stepDurations: Record<number, number>;
659
+ stepWords?: Record<number, WordTiming[]>;
660
+ }
661
+
662
+ /**
663
+ * CSS and ASS styling for QuickPeek
664
+ */
665
+
666
+ declare const CURSOR_CSS = "\n #qp-cursor {\n position: fixed;\n width: 32px;\n height: 38px;\n background: url(\"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAMgAAAEsBAMAAAB01OGNAAAAIGNIUk0AAHomAACAhAAA+gAAAIDoAAB1MAAA6mAAADqYAAAXcJy6UTwAAAAtUExURUdwTAAAAAAAAAAAAAAAAAAAAAAAAB0dHVZWVtvb2/X19f///6mpqXV1df////vm1ewAAAAGdFJOUwAltTfpcLO208EAAAABYktHRA5vvTBPAAAAB3RJTUUH6QwFFREKplBWTAAABqdJREFUeNq93c+OFFUUx/EeCHud6H7A4HpCOu6NiWsWZh5Ax8YFbAdNWBPxBcCo+yGjD+CAW1ywh0Rexu6uulX3Vp269/c7f7o3Bpjhk6+H6TrU0LdXR6sDPI4/OgCyvn0I5NsDpKw3B0hZbw6Qst4cIGWLxKdskfiUHRKeskPCU/ZIdMoeiU7pkOCUDglO6ZHYlB6JTUlIaEpCQlMGJDJlQCJTRiQwZUQCUzIkLiVD4lJyJCwlR8JSCiQqpUCiUkokKKVEglImSEzKBIlJmSIhKT3yY2hKjzx7GpnSI7+8iExJyN+RKQl5FZkyIJEpAxKZMiKBKSMSmJIhcSkZEpeSI2EpORKWUiBRKQUSlVIiQSklEpQyQWJSJkhMyhQJSZkiISkzJCJlhkSkzJGAlDkSkCIg/ikC4p8iIe4pEuKeIiLeKSLinSIjziky4pyygPimLCC+KUuIa8oS4pqyiHimLCKeKcuIY8oy4phSQfxSKohfSg1xS6khbilVxCulinil1BGnlDrilNJAfFIaiE9KC3FJaSEuKU3EI6WJeKS0EYeUNuKQAiD2FACxpyCIOQVBzCkQYk2BEGsKhhhTMMSYAiK2FBCxpaCIKQVFTCkwYkmBEUsKjhhScMSQQiD6FALRpzCIOoVB1CkUok2hEG0KhyhTOESZQiK6FBLRpbCIKoVFVCk0okmhEU0KjyhSeESRokD4FAXCp2gQOkWD0CkqhE1RIWyKDiFTdAiZokS4FCXCpWgRKkWLUClqhElRI0yKHiFS9AiRYkDwFAOCp1gQOMWCwCkmBE0xIWiKDQFTbAiYYkSwFCOCpVgRKMWKQClmBEkxI0iKHQFS7AiQ4oC0UxyQdooH0kzxQJopLkgrxQVppfggjRQfpJHihNRTnJB6ihdSTfFCqiluSC3FDaml+CGVFD+kkuKILKc4IsspnshiiieymOKKLKW4Iq8u5RRf5FpO8UUWpuKMyCnOiJzijYgp3oiY4o5IKe6IlEIhL38FHr9fzFK47/0On195PL6YpVDIX4CRP1IKhbxGSvLHbQUy/snhUrg/XW9IpE/hkOcs0qVwyD8s0qWQ349nJ9+lcAg9+S6FfFqhJ79PIZFLGtmlkMifPLJNYf/5wgWvfAYg11f5j951H39+B3/cBZA/3uY/+nccJ/5oItdPi19Lkz91RV5snuQ/fN0jJ57I9svvkTT5rzyR7RX7wW/C5L93RPbPI8Xknysmj7zWQZz8fTeke0J8aJ088vqTR1czl5t8Fel/w2Ly12/6oXghaRn8IEz+nJh8DRkuUc/yn02XYGLy0Ou0ysn3X44+/75rvNb+IE3+OxdkXM/Lr3l+8stIvjSIkz91QMaQzeYn2+QXkWL7KSafnogd/jlcHiJPHn8iXkLKNe7BW9Pk26/73T/+y3/tkp38AjLdR4vJp+XrxIhMQuRL8Jc2ZLZYF5OnL8Gt18ff6f9bTJ5dvkQku6vw6Ub4AHb5ap1ZcPS5MHn2Eiwhxe2Rs8rk0Utw8xyJ/gNMy5eAlPd5bkiTJ5ev5tkeN7+uTP6+FpncsEqTt6zd7fNWpMmTy9cMmd15+0SYPLl8tc/AuVWZPLh8TZH5LURx8unOFzb59rlEafKG5WuCSPdC0+SvhMljyxdwVtSx9DVPTb5ExPvTafIfpMmf8oh4pz1NXr92F8jCtz+kyVPLF3Km2jfdz+mXrxxZCLFPHjnnTpw8s3xlyFLIMHn18gWdPVibPLJ8jchiyPBB5eTf4ZOHzoNMky+eiInlCzrZMk1eWruR5Qs6o/NImjxxCcZOGz2rTB64BGPnpkqTJ5Yv7ARY4/LVIz9XQ4bJK5cv7FRe49oNni/cT165fIEnJdfW7vYTMXjm8w3pax6ePHh6tW35As/hltfui8ZniUglvDb55vKFno0urd3wJRg95b26dp8SSO3/rTh5dPlCT94XJ48uX/B7CKTJa5Yv+N0QLJOH39fBsnavwRBx7X75vv/kExBpftn2k38oGOAraYBnuS+6DxwvwaPRmvwaDZmt3ZkBvlwHuFBPJp8breVrjYZM1u7SaEx+jYaUk58Y58gLj5i/lO0mPzE2d4GZYPetxrWbNDoEu68wrN2ssUfAG3Bp7X7CGnsE/R7F2UZ6tI0dwt5Dpo3dJ8LfbDlWGivm3dpuKY0V875z6RLMGtw76J3pDO69ANc6Y3WPCBn+wkUaq48JY3giJg3ukU8+ysgnH2ek5SvUGCYfaaTJhxr95GONbvLRxm7y4cb2iTjeWN08gLFaaYz/AWAIeJtzdhugAAAAAElFTkSuQmCC\") no-repeat;\n background-size: contain;\n pointer-events: none;\n z-index: 2147483647;\n transform: translate(-3px, -3px);\n transition: left 800ms ease-out, top 800ms ease-out, transform 0.08s ease-out;\n filter: drop-shadow(1px 1px 2px rgba(0,0,0,0.3));\n }\n #qp-cursor.clicking {\n transform: translate(-3px, -3px) scale(0.9);\n }\n";
667
+ /**
668
+ * JavaScript to inject for cursor tracking
669
+ */
670
+ declare const CURSOR_SCRIPT = "\n (function() {\n if (document.getElementById('qp-cursor')) return;\n const cursor = document.createElement('div');\n cursor.id = 'qp-cursor';\n document.body.appendChild(cursor);\n\n document.addEventListener('mousemove', (e) => {\n cursor.style.left = e.clientX + 'px';\n cursor.style.top = e.clientY + 'px';\n });\n\n document.addEventListener('mousedown', () => cursor.classList.add('clicking'));\n document.addEventListener('mouseup', () => cursor.classList.remove('clicking'));\n\n // Start at center\n cursor.style.left = window.innerWidth / 2 + 'px';\n cursor.style.top = window.innerHeight / 2 + 'px';\n })();\n";
671
+ /**
672
+ * Generate ASS subtitle file header
673
+ */
674
+ declare function generateASSHeader(width: number, height: number, font: string, size: number, primaryColor: string, secondaryColor: string, outlineColor: string, backColor: string, bold: boolean, outlineWidth: number, shadow: number, marginBottom: number, alignment?: number): string;
675
+ /**
676
+ * Linux font-substitution guard (from VidLet): "Arial Black" is not installed
677
+ * on Linux, and fontconfig silently falls back to Noto Sans Regular with a
678
+ * synthesised weight. DejaVu Sans ships a real Bold face on every distro this
679
+ * runs on.
680
+ */
681
+ declare function resolveCaptionFont(font: string): string;
682
+ /** ASS numpad alignment: 1-3 bottom, 4-6 middle, 7-9 top. */
683
+ declare function positionToAlignment(position: CaptionPosition, leftAligned?: boolean): number;
684
+ /** One step's caption words, timed in absolute ms on the final video timeline. */
685
+ interface CaptionCue {
686
+ words: WordTiming[];
687
+ }
688
+ interface CaptionRenderContext {
689
+ cues: CaptionCue[];
690
+ width: number;
691
+ height: number;
692
+ captions: CaptionStyle;
308
693
  }
694
+ /**
695
+ * The coordinate space captions are drawn in, which is NOT the output size.
696
+ *
697
+ * ASS declares its own PlayResX/PlayResY and libass scales that space onto
698
+ * whatever frame it burns into. Handing it the real output size made every
699
+ * caption number mean something different at every resolution: at 360x640 a
700
+ * 112px font goes from 5.8% of the frame height to 17.5%, and a 300px bottom
701
+ * margin - the "lower third" the preset intends - lands 47% of the way up,
702
+ * which is dead centre over the product. So the space is pinned to the size
703
+ * the numbers were authored for and libass does the scaling.
704
+ *
705
+ * A 1080x1920 render is unaffected: the scale factor is exactly 1.
706
+ */
707
+ declare function captionCanvas(width: number, height: number): {
708
+ width: number;
709
+ height: number;
710
+ };
711
+ /**
712
+ * The caption style a render actually uses: the canvas decides the geometry,
713
+ * captions.css supplies the face and the colours, `video.captions` overrides
714
+ * anything.
715
+ *
716
+ * The canvas wins over captions.css because the numbers in that file were
717
+ * never a preference. QuickPeek writes it on first run with 56px text 80px
718
+ * off the bottom - fine for a 16:9 tutorial, a desktop subtitle lost under
719
+ * the player chrome on the 9:16 short that is now the default shape - and no
720
+ * user ever chose those numbers. `--profile short` already overrode them for
721
+ * exactly this reason; the only change here is that a plain run gets the same
722
+ * treatment instead of depending on which command was typed.
723
+ *
724
+ * What the frame size genuinely cannot know stays with the CSS: the typeface
725
+ * and the colours, which are brand decisions. Anything else belongs to
726
+ * `video.captions` in quickpeek.config.json, which wins over both.
727
+ */
728
+ declare function resolveCaptions(css: CaptionStyle, video: Config['video']): CaptionStyle;
729
+ /** Render a complete ASS document for the configured caption preset. */
730
+ declare function renderCaptionsAss(context: CaptionRenderContext): string;
309
731
 
732
+ /**
733
+ * Subscription and user registration client
734
+ * CLI-only headless registration fallback. Tier/verification checks
735
+ * are now in the SparkStripe gating module (getTierByEmail, isVerified).
736
+ */
737
+ interface RegistrationResult {
738
+ success: boolean;
739
+ pendingVerification?: boolean;
740
+ alreadyVerified?: boolean;
741
+ error?: string;
742
+ }
743
+ /**
744
+ * Register user for free plan via SparkStripe
745
+ */
746
+ declare function registerFreeUser(email: string): Promise<RegistrationResult>;
747
+
748
+ /**
749
+ * Silence before the first word.
750
+ *
751
+ * Narration that starts on frame one talks over a viewer who is still taking
752
+ * in the opening card, and the first syllable lands before anyone is
753
+ * listening. Part of the audio rather than a compose-time delay, so the
754
+ * captions, the step durations and the dead-air pass all measure the same
755
+ * timeline.
756
+ *
757
+ * Here rather than beside the code that lays it down, because the
758
+ * realignment pass has to know about it too and importing across those two
759
+ * modules the other way would close a cycle.
760
+ */
761
+ declare const NARRATION_LEAD_IN_MS = 300;
310
762
  /**
311
763
  * Load existing voiceover metadata if available
312
764
  */
@@ -323,7 +775,7 @@ interface Step {
323
775
  /**
324
776
  * Check if existing voiceover can be reused
325
777
  */
326
- declare function canReuseVoiceover(steps: Step[], voice: string, existingMeta: VoiceoverMeta | null): {
778
+ declare function canReuseVoiceover(steps: Step[], voice: string, rate: string, existingMeta: VoiceoverMeta | null): {
327
779
  canReuse: boolean;
328
780
  reason?: string;
329
781
  };
@@ -347,17 +799,81 @@ declare function loadExistingVoiceover(steps: Step[], outputDir: string, meta: V
347
799
  }>;
348
800
 
349
801
  declare function resolveVoice(config: Config): string | null;
802
+ /** The narration pace for this config, as the enum the meta stores. */
803
+ declare function resolveRate(config: Config): VoiceRate;
804
+ /**
805
+ * Characters an SSML payload cannot carry.
806
+ *
807
+ * Both engines build SSML and drop the caption into it as-is, so a bare "&"
808
+ * makes the request invalid XML by the time the service sees it. It does not
809
+ * fail loudly: the Node engine returns an empty stream ("No audio data
810
+ * received"), the step is left with no clip, and because `stepWords` only
811
+ * records steps that produced words, the step vanishes from the metadata too.
812
+ * The video still renders - one step just has no voice.
813
+ *
814
+ * Found on Anysite's closing card, "Live Web Data for GTM & Marketing Teams":
815
+ * the whole closing line was silent while every other step in the same run
816
+ * synthesised fine. That is the worst shape a bug can take here, because the
817
+ * missing narration is at the end where nobody re-watches.
818
+ *
819
+ * Spoken, not escaped. "&" is read "and" by any English speaker, the burned
820
+ * caption still shows the symbol, and nothing downstream has to trust an
821
+ * engine to unescape an entity it may not have escaped itself.
822
+ */
823
+ declare function speakSymbols(text: string): string;
350
824
  declare function generateVoiceover(steps: TTSStep[], config: Config, workDir: string, outputDir: string, logError: (context: string, error: unknown) => Promise<void>, task: (text: string) => {
351
825
  succeed: (text?: string) => void;
352
826
  }, info: (text: string) => void): Promise<AudioResult>;
353
827
 
354
828
  /**
355
- * Synthesize speech using Python edge-tts (more reliable than Node.js msedge-tts)
356
- * Falls back to Node.js msedge-tts if Python is unavailable
829
+ * A word as the synthesizer actually spoke it, in ms from the start of the
830
+ * clip it was asked to produce — before any trimming this pipeline does to it.
357
831
  */
358
- declare function synthesizeSpeech(text: string, outputPath: string, voice: string): Promise<void>;
832
+ interface SpokenWord {
833
+ word: string;
834
+ start: number;
835
+ end: number;
836
+ }
837
+ /** Starts a new run, so the next line chooses the engine again. */
838
+ declare function resetVoiceEngine(): void;
839
+ declare function synthesizeSpeech(rawText: string, outputPath: string, voice: string, rate?: string): Promise<SpokenWord[] | null>;
840
+ /**
841
+ * Pull the word boundaries out of the raw metadata stream.
842
+ *
843
+ * Each websocket frame is a complete JSON object, but a Readable is free to
844
+ * coalesce them into one chunk, so the concatenated text is split on brace
845
+ * depth rather than parsed whole. A frame that does not parse is skipped: a
846
+ * missing word costs a slightly-off highlight, throwing costs the clip.
847
+ */
848
+ declare function parseWordBoundaries(raw: string): SpokenWord[] | null;
359
849
 
360
850
  declare function estimateWordTimings(caption: string, duration: number): WordTiming[];
851
+ /**
852
+ * Marry the authored caption text to the timings the voice reported.
853
+ *
854
+ * The burned-in captions must show the caption as WRITTEN, and the spoken word
855
+ * list does not always match it one for one — the text handed to the voice is
856
+ * the spoken form ("vidlet dot app" for a domain), and a number reads as
857
+ * several words. So the spoken words are used only as a clock. When the counts
858
+ * line up, each caption word takes its spoken counterpart's timing. When they
859
+ * don't, the caption words are spread evenly across the actual speech
860
+ * envelope — still better than a blind estimate, because the envelope excludes
861
+ * the silence the voice leaves at either end of a clip.
862
+ */
863
+ declare function alignWordsToCaption(caption: string, spoken: SpokenWord[]): WordTiming[];
864
+ /**
865
+ * Real word timings for a clip, from the boundaries the voice reported while
866
+ * speaking it. Null when the engine gave none, and callers fall back to
867
+ * estimateWordTimings. Timings are ms relative to the start of the clip.
868
+ *
869
+ * The engine times words against the audio it produced, but the clip on disk
870
+ * has since had its leading silence trimmed, so every offset is early by
871
+ * however much came off the front. The trim gate is set just under the noise
872
+ * edge-tts pads with, which puts the cut where the first word begins — so the
873
+ * first word's own offset IS what was removed, and rebasing on it re-anchors
874
+ * the whole clip without measuring the file again.
875
+ */
876
+ declare function spokenWordTimings(caption: string, spoken: SpokenWord[] | null): WordTiming[] | null;
361
877
  /**
362
878
  * Align voiceover to actual recording durations
363
879
  * If any step's recording took longer than its voiceover, pad with silence
@@ -377,6 +893,15 @@ declare function alignVoiceoverToRecording(stepAudio: Map<number, StepAudio>, ac
377
893
  declare function buildWatermarkFilters(opts: {
378
894
  tier: UserTier;
379
895
  videoDuration: number;
896
+ /**
897
+ * The video ends on a credits card that already names QuickPeek.
898
+ *
899
+ * The end text is drawn along the bottom of the last three seconds, which
900
+ * is exactly where that card carries the offer - two overlays stacked on
901
+ * the same strip, saying the same thing twice. The corner mark stays: it is
902
+ * the part that marks the whole video, not just its ending.
903
+ */
904
+ endsOnCreditsCard?: boolean;
380
905
  }): string;
381
906
 
382
- export { type ApplyTransitionsOptions, type AudioResult, type BlendTransition, CURSOR_CSS, CURSOR_PNG, CURSOR_SCRIPT, type ComposeOptions, type ConcatVideoOptions, Config, type LenientParse, type SpeedUpOptions, type SpliceVideosOptions, type StepAudio, type StepSync, type StepTransition, type TTSStep, type TimeSegment, UserTier, type VideoOverlay, type VoiceoverMeta, type WordTiming, alignVoiceoverToRecording, applyTransitions, buildConcatArgs, buildKeepSegments, buildStepTransitions, buildWatermarkFilters, canReuseVoiceover, composeVideo, concatMedia, concatVideoWithOutro, estimateWordTimings, extractJsonCandidates, generateASSHeader, generateSilence, generateVoiceover, getAudioCodec, getMediaDuration, getUpgradeUrl, isHonestWav, loadExistingVoiceover, loadVoiceoverMeta, mapTimestampsAfterCompression, parseAIJson, registerFreeUser, repairUnicodeEscapes, resolveVoice, salvageTruncatedJson, shouldWatermark, speedUpVideo, spliceVideos, synthesizeSpeech };
907
+ export { type ApplyTransitionsOptions, type AudioResult, type BlendTransition, CURSOR_CSS, CURSOR_PNG, CURSOR_SCRIPT, type CaptionCue, CaptionPosition, type CaptionRenderContext, CaptionStyle, type ComposeOptions, type ConcatVideoOptions, Config, EMPTY_SPAN_SPEED, GIF_DEFAULTS, type GifOptions, type LenientParse, MUSIC_BED_LUFS, NARRATION_LEAD_IN_MS, type SilenceSpan, type SpeedUpOptions, type SpliceVideosOptions, type SpokenWord, type StepAudio, type StepSync, type StepTransition, TRIM_EDGE_SILENCE, type TTSStep, type TimeSegment, UserTier, VOICE_LEAD_MS, type VideoOverlay, VoiceRate, type VoiceoverMeta, type WordTiming, alignVoiceoverToRecording, alignWordsToCaption, applyTransitions, atempoChain, buildCompressionFilter, buildConcatArgs, buildGifArgs, buildKeepSegments, buildPaletteArgs, buildRetimeFilter, buildStepTransitions, buildWatermarkFilters, canReuseVoiceover, captionCanvas, composeVideo, compressSilentSpans, concatMedia, concatVideoWithFullFrameOutro, concatVideoWithOutro, convertToGif, coverFilterChain, cutSpansFromVideo, detectBlackSpans, detectFrozenSpans, detectSilences, detectWhiteSpans, estimateWordTimings, extractJsonCandidates, generateASSHeader, generateSilence, generateVoiceover, getAudioCodec, getMediaDuration, getUpgradeUrl, getVideoDimensions, getVideoStreamDuration, hasAudioStream, intersectSpans, isHonestWav, loadExistingVoiceover, loadVoiceoverMeta, mapTimestampsAfterCompression, mergeSpans, parseAIJson, parseWordBoundaries, positionToAlignment, registerFreeUser, renderCaptionsAss, repairUnicodeEscapes, resetVoiceEngine, resolveCaptionFont, resolveCaptions, resolveRate, resolveVoice, retimeSpansInVideo, salvageTruncatedJson, shouldWatermark, speakSymbols, speedUpMiddle, speedUpVideo, spliceVideos, spokenWordTimings, synthesizeSpeech, trimEdgeSilence };