@spark-apps/quickpeek 1.2.2 → 1.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/dist/assets/music/calm-drifting-piano.mp3 +0 -0
- package/dist/assets/music/lofi-roof-tops.mp3 +0 -0
- package/dist/assets/music/manifest.json +32 -0
- package/dist/assets/music/upbeat-spring-on-the-horizon.mp3 +0 -0
- package/dist/highlight.css +8 -7
- package/dist/index.d.mts +561 -36
- package/dist/index.mjs +2288 -399
- package/dist/mcp-tools.mjs +6965 -1866
- package/dist/qp.js +6281 -1566
- package/dist/voices-CPpnWn39.d.mts +613 -0
- package/dist/web.d.mts +5 -3
- package/dist/web.mjs +398 -164
- package/mcp.mjs +94 -21
- package/package.json +5 -2
- package/dist/scraping-BgepklP3.d.mts +0 -335
package/dist/index.d.mts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
|
-
import { U as UserTier, C as Config } from './
|
|
2
|
-
export { A as AIError,
|
|
3
|
-
export {
|
|
1
|
+
import { U as UserTier, C as CaptionStyle, a as CaptionPosition, b as Config, V as VoiceRate } from './voices-CPpnWn39.mjs';
|
|
2
|
+
export { A as AIError, c as AIResponse, d as AIResult, e as CLI_BACKOFF_MS, f as CONFIG_FILE, g as CaptionPreset, h as CaptionWordStyle, i as ChatOptions, D as DEFAULT_CAPTIONS, j as DEFAULT_CONFIG, E as EXCLUDED_LINK_PATTERNS, k as ElementInfo, l as ElementType, H as HighlightMode, I as INTERACTIVE_SELECTORS, m as INTERNATIONAL_VOICE, N as NO_CLIP_OUTRO, P as PLAN_MAX_TOKENS, n as Plan, R as RATE_FACTOR, o as RateLimitInfo, p as RetryNotice, S as SERVER_BACKOFF_MS, q as SIZE_PRESETS, r as SparkStatus, s as SparkSubscription, t as SparkTrial, u as Step, v as SuggestedAction, T as Tier, w as VERSION, x as VIDEO_PROFILES, y as VideoProfileName, z as VideoSize, B as VoiceGender, F as applyProfile, G as ariaLabelSelector, J as buildSystemPrompt, K as buildUserPrompt, L as callAI, M as callAIViaRelay, O as crawlPage, Q as deStock, W as defaultCaptionsFor, X as getErrorMessage, Y as getLanguageName, Z as getStatus, _ as getStatusCached, $ as getTierByEmail, a0 as gotoSettled, a1 as hasTrialRemaining, a2 as hasVoice, a3 as idSelector, a4 as internationalVoiceFor, a5 as isCanceling, a6 as isExcludedLink, a7 as isPaid, a8 as isVerified, a9 as normalizeUrl, aa as parseAIPlanResponse, ab as parsePartial, ac as pricingUrl, ad as rgbaToHex, ae as runPooled, af as shouldSkipLink, ag as showPaywall, ah as stripLongDashes, ai as toTitleCase, aj as voiceFor, ak as waitForPlaceholdersGone } from './voices-CPpnWn39.mjs';
|
|
3
|
+
export { MALE_VOICES, MULTILINGUAL_VOICES, WIDE_VOICES, hexToAss as hexToASS, windowsDrivePathToWsl as windowsPathToWSL } from '@spark-apps/video-kit';
|
|
4
|
+
import 'playwright';
|
|
4
5
|
|
|
5
6
|
/**
|
|
6
7
|
* Billing utilities — watermark decisions and upgrade URL generation
|
|
@@ -33,14 +34,27 @@ interface StepSync {
|
|
|
33
34
|
targetDuration: number;
|
|
34
35
|
}
|
|
35
36
|
interface ComposeOptions {
|
|
37
|
+
/** Input 0 carries a spliced clip's own audio, to be mixed under the voice. */
|
|
38
|
+
hasClipAudio?: boolean;
|
|
36
39
|
videoPath: string;
|
|
37
|
-
|
|
40
|
+
/** Burn-in captions. Omit to skip the ass filter entirely (draft). */
|
|
41
|
+
assPath?: string | undefined;
|
|
38
42
|
outputPath: string;
|
|
39
43
|
voiceoverPath?: string | undefined;
|
|
40
44
|
musicPath?: string | undefined;
|
|
41
45
|
musicVolume?: number | undefined;
|
|
42
46
|
musicFadeIn?: number | undefined;
|
|
43
47
|
musicFadeOut?: number | undefined;
|
|
48
|
+
/**
|
|
49
|
+
* Loudness the bed is normalised to before it meets the narration, in LUFS.
|
|
50
|
+
*
|
|
51
|
+
* A fixed `musicVolume` cannot produce a consistent bed, because the source
|
|
52
|
+
* decides everything: two CC0 tracks in the same folder measure -31 dB and
|
|
53
|
+
* -15 dB mean, so the same gain makes one inaudible and the other a
|
|
54
|
+
* competitor. Normalising first makes the bed a level rather than a
|
|
55
|
+
* multiplier, and `musicVolume` goes back to being the trim it reads as.
|
|
56
|
+
*/
|
|
57
|
+
musicTargetLufs?: number | undefined;
|
|
44
58
|
width: number;
|
|
45
59
|
height: number;
|
|
46
60
|
fps: number;
|
|
@@ -53,9 +67,12 @@ interface ComposeOptions {
|
|
|
53
67
|
contrast?: number | undefined;
|
|
54
68
|
trimStart?: number | undefined;
|
|
55
69
|
userTier?: UserTier | undefined;
|
|
70
|
+
endsOnCreditsCard?: boolean | undefined;
|
|
56
71
|
mainContentDuration?: number | undefined;
|
|
57
72
|
stepSync?: StepSync[] | undefined;
|
|
58
73
|
stepTransitions?: StepTransition[] | undefined;
|
|
74
|
+
/** 'off' pins the deliverable encode to libx264; anything else probes for a GPU. */
|
|
75
|
+
hwaccel?: 'auto' | 'off' | undefined;
|
|
59
76
|
}
|
|
60
77
|
interface ConcatVideoOptions {
|
|
61
78
|
inputPath: string;
|
|
@@ -69,7 +86,27 @@ interface VideoOverlay {
|
|
|
69
86
|
path: string;
|
|
70
87
|
startTime: number;
|
|
71
88
|
duration: number;
|
|
72
|
-
|
|
89
|
+
/**
|
|
90
|
+
* How the source is fitted to the frame.
|
|
91
|
+
*
|
|
92
|
+
* stretch = fill the frame's width, crop the height if it overflows.
|
|
93
|
+
* center = scale to fit, letterbox the rest.
|
|
94
|
+
* cover = fill the frame with a blurred, enlarged copy of the source and
|
|
95
|
+
* sit the source itself on top of it, untouched. A 1200x630 OG
|
|
96
|
+
* card in a 9:16 frame is 16:9 either way; the only question is
|
|
97
|
+
* what fills the two thirds of the frame it cannot reach, and a
|
|
98
|
+
* blurred continuation of the card reads as design where a black
|
|
99
|
+
* band reads as a mistake.
|
|
100
|
+
*/
|
|
101
|
+
mode: 'stretch' | 'center' | 'cover';
|
|
102
|
+
/**
|
|
103
|
+
* The source is a still image (png/jpg/webp), not a clip.
|
|
104
|
+
*
|
|
105
|
+
* A still has exactly one frame, so it has to be looped for the length of
|
|
106
|
+
* its window or the overlay holds a single frame at EOF and the window
|
|
107
|
+
* plays whatever is underneath it.
|
|
108
|
+
*/
|
|
109
|
+
still?: boolean;
|
|
73
110
|
trimStart?: number;
|
|
74
111
|
trimEnd?: number;
|
|
75
112
|
}
|
|
@@ -105,6 +142,34 @@ interface TimeSegment {
|
|
|
105
142
|
end: number;
|
|
106
143
|
}
|
|
107
144
|
|
|
145
|
+
/**
|
|
146
|
+
* A beat of silence before the narrator starts, in milliseconds.
|
|
147
|
+
*
|
|
148
|
+
* The voice used to be faded in over the same duration as the picture, which
|
|
149
|
+
* ate the first syllable - and the first syllable belongs to the hook, the one
|
|
150
|
+
* line that has to land. A short delay gives the viewer the same moment to
|
|
151
|
+
* settle without softening the words.
|
|
152
|
+
*/
|
|
153
|
+
declare const VOICE_LEAD_MS = 200;
|
|
154
|
+
/**
|
|
155
|
+
* Loudness the music bed is normalised to before the mix, in LUFS.
|
|
156
|
+
*
|
|
157
|
+
* Measured, not guessed. Running this exact graph over the accordio
|
|
158
|
+
* narration puts the bed at -26.7 dB mean in a gap between lines while the
|
|
159
|
+
* speech measures -13.7 dB: thirteen dB under the voice, which is a bed. Each
|
|
160
|
+
* dB here moves that reading a dB, so the number is a real control - -38 gives
|
|
161
|
+
* -30.6, -42 gives -34.7.
|
|
162
|
+
*
|
|
163
|
+
* Over a long passage with no narration at all - the closing card - the master
|
|
164
|
+
* chain's single-pass loudnorm is dynamic and opens up, and the bed rises to
|
|
165
|
+
* about -18 dB. That is the intended shape: discreet under the voice, present
|
|
166
|
+
* where there is nothing else.
|
|
167
|
+
*
|
|
168
|
+
* What it replaced measured -39 dB at source and -45 dB after its own volume
|
|
169
|
+
* trim, and read in the finished file at -32 dB against -15 dB of speech.
|
|
170
|
+
* That is not a quiet bed, it is no bed: the video was speech in a vacuum.
|
|
171
|
+
*/
|
|
172
|
+
declare const MUSIC_BED_LUFS = -34;
|
|
108
173
|
/**
|
|
109
174
|
* Compose final video with subtitles, voiceover, and music
|
|
110
175
|
*/
|
|
@@ -120,6 +185,45 @@ declare function buildKeepSegments(eventTimestampsSeconds: number[], totalDurati
|
|
|
120
185
|
*/
|
|
121
186
|
declare function mapTimestampsAfterCompression(originalTimestampsSeconds: number[], segments: TimeSegment[]): number[];
|
|
122
187
|
|
|
188
|
+
/**
|
|
189
|
+
* Video → optimized GIF conversion (ported from VidLet's togif tool).
|
|
190
|
+
* Two-pass: generate a tuned palette, then map the video through it —
|
|
191
|
+
* dramatically better colour than ffmpeg's default 256-colour GIF path.
|
|
192
|
+
*/
|
|
193
|
+
interface GifOptions {
|
|
194
|
+
input: string;
|
|
195
|
+
output: string;
|
|
196
|
+
fps?: number;
|
|
197
|
+
width?: number;
|
|
198
|
+
dither?: string;
|
|
199
|
+
statsMode?: string;
|
|
200
|
+
}
|
|
201
|
+
declare const GIF_DEFAULTS: {
|
|
202
|
+
readonly fps: 15;
|
|
203
|
+
readonly width: 480;
|
|
204
|
+
readonly dither: "sierra2_4a";
|
|
205
|
+
readonly statsMode: "full";
|
|
206
|
+
};
|
|
207
|
+
declare function buildPaletteArgs(input: string, palettePath: string, fps: number, width: number, statsMode: string): string[];
|
|
208
|
+
declare function buildGifArgs(input: string, palettePath: string, output: string, fps: number, width: number, dither: string): string[];
|
|
209
|
+
declare function convertToGif(options: GifOptions): Promise<string>;
|
|
210
|
+
|
|
211
|
+
/**
|
|
212
|
+
* The video stream's pixel dimensions.
|
|
213
|
+
*
|
|
214
|
+
* Needed because the splice pass scales its overlays to match the video it is
|
|
215
|
+
* pasting them onto, and that video is the raw recording, which is smaller
|
|
216
|
+
* than the configured output whenever video.zoom is in play: a short records
|
|
217
|
+
* at 540x960 and is upscaled to 1080x1920 only at compose. Scaling overlays to
|
|
218
|
+
* the output size instead put them in at 2x and cropped everything but the
|
|
219
|
+
* middle of each card.
|
|
220
|
+
*/
|
|
221
|
+
declare function getVideoDimensions(filePath: string): Promise<{
|
|
222
|
+
width: number;
|
|
223
|
+
height: number;
|
|
224
|
+
}>;
|
|
225
|
+
/** Whether a file carries an audio stream at all. */
|
|
226
|
+
declare function hasAudioStream(filePath: string): Promise<boolean>;
|
|
123
227
|
/**
|
|
124
228
|
* Run ffprobe to get media duration in milliseconds
|
|
125
229
|
*/
|
|
@@ -146,6 +250,22 @@ declare function isHonestWav(filePath: string): Promise<boolean>;
|
|
|
146
250
|
*/
|
|
147
251
|
declare function buildConcatArgs(listPath: string, output: string): string[];
|
|
148
252
|
declare function concatMedia(listPath: string, output: string): Promise<void>;
|
|
253
|
+
/**
|
|
254
|
+
* Strip the silence a TTS engine leaves on both ends of a clip.
|
|
255
|
+
*
|
|
256
|
+
* Every edge-tts clip arrives with its own lead-in and tail, typically a few
|
|
257
|
+
* hundred ms each. One clip per caption, concatenated, means each boundary
|
|
258
|
+
* stacks tail + the pad that aligns the step + the next clip's lead-in - the
|
|
259
|
+
* "robotic pause" between sentences. Trimming here rather than at concat time
|
|
260
|
+
* keeps the clip the unit of truth: the duration measured after this call,
|
|
261
|
+
* the padding computed from it, and the word timings rebased in
|
|
262
|
+
* spokenWordTimings all describe the same audio.
|
|
263
|
+
*
|
|
264
|
+
* -45dB rather than a hard zero: edge-tts pads with near-silent noise, not
|
|
265
|
+
* digital black, so a stricter gate finds nothing to remove.
|
|
266
|
+
*/
|
|
267
|
+
declare const TRIM_EDGE_SILENCE: string;
|
|
268
|
+
declare function trimEdgeSilence(filePath: string): Promise<void>;
|
|
149
269
|
/**
|
|
150
270
|
* Generate silence audio file of specified duration
|
|
151
271
|
*/
|
|
@@ -160,6 +280,262 @@ declare function speedUpVideo(opts: SpeedUpOptions): Promise<void>;
|
|
|
160
280
|
* Returns the combined video for further processing (composition with audio)
|
|
161
281
|
*/
|
|
162
282
|
declare function concatVideoWithOutro(opts: ConcatVideoOptions): Promise<void>;
|
|
283
|
+
/**
|
|
284
|
+
* Append an outro that is already a full frame, unchanged.
|
|
285
|
+
*
|
|
286
|
+
* concatVideoWithOutro treats its outro as a badge: it shrinks it to 60% and
|
|
287
|
+
* overlays it on black, which is right for the bundled square watermark and
|
|
288
|
+
* wrong for a rendered credits card, which is composed at the video's exact
|
|
289
|
+
* size and must not be shrunk inside its own frame.
|
|
290
|
+
*
|
|
291
|
+
* setsar on both inputs because concat refuses streams whose sample aspect
|
|
292
|
+
* ratios disagree, and a card rendered by the browser does not necessarily
|
|
293
|
+
* carry the same SAR as a recorded page.
|
|
294
|
+
*/
|
|
295
|
+
declare function concatVideoWithFullFrameOutro(opts: ConcatVideoOptions): Promise<void>;
|
|
296
|
+
|
|
297
|
+
/**
|
|
298
|
+
* Compressing dead air out of a finished demo.
|
|
299
|
+
*
|
|
300
|
+
* A step whose narration ends before the step does leaves the video sitting
|
|
301
|
+
* on a still frame with nothing being said - padding that keeps audio and
|
|
302
|
+
* video in sync, plus any deliberate hold while an async result loads. Held
|
|
303
|
+
* for a second it reads as a beat; held for five it reads as a hang.
|
|
304
|
+
*
|
|
305
|
+
* This runs AFTER compose on purpose. By then the captions are burned in and
|
|
306
|
+
* the silent spans have none on screen (a karaoke line ends with its last
|
|
307
|
+
* word), so speeding those spans up removes the dead time without touching
|
|
308
|
+
* caption timing, the recorder, or the TTS timeline - the three things that
|
|
309
|
+
* have to agree with each other. What is sped up is silence, so it is
|
|
310
|
+
* inaudible, and the visuals still play through: a result that appears during
|
|
311
|
+
* a hold is still seen, just briskly.
|
|
312
|
+
*/
|
|
313
|
+
/** A silent stretch of the mixed audio, in seconds. */
|
|
314
|
+
interface SilenceSpan {
|
|
315
|
+
start: number;
|
|
316
|
+
end: number;
|
|
317
|
+
/** Longest this span may run after compression. Defaults to the pass's cap. */
|
|
318
|
+
cap?: number;
|
|
319
|
+
}
|
|
320
|
+
/**
|
|
321
|
+
* Length of the video stream in seconds, or 0 if it cannot be read.
|
|
322
|
+
*
|
|
323
|
+
* Distinct from the container's duration, which reports whichever stream runs
|
|
324
|
+
* longest - usually the audio, since compose pads it to the voiceover.
|
|
325
|
+
*/
|
|
326
|
+
declare function getVideoStreamDuration(inputPath: string): Promise<number>;
|
|
327
|
+
declare function detectSilences(inputPath: string, minDurSecs: number): Promise<SilenceSpan[]>;
|
|
328
|
+
/**
|
|
329
|
+
* Find stretches where the picture stops changing.
|
|
330
|
+
*
|
|
331
|
+
* Silence is not the only kind of dead air. A page that finishes an animation
|
|
332
|
+
* - a confetti burst settling, a spinner resolving - holds a still frame for
|
|
333
|
+
* as long as the step is paced to last, and that reads as a hang even with
|
|
334
|
+
* narration over it. freezedetect reports those stretches; they are treated
|
|
335
|
+
* like silence, and for the same reason.
|
|
336
|
+
*/
|
|
337
|
+
declare function detectFrozenSpans(inputPath: string, minDurSecs: number): Promise<SilenceSpan[]>;
|
|
338
|
+
/**
|
|
339
|
+
* Spans where the picture is essentially black.
|
|
340
|
+
*
|
|
341
|
+
* A frozen span and a black span are not the same thing, and the difference
|
|
342
|
+
* matters. A frozen span is a still screen: the viewer is looking at something,
|
|
343
|
+
* so it earns the FROZEN_MAX_SECS ceiling and gets shortened, not cut. A black
|
|
344
|
+
* span is a page mid-transition, and it is worth nothing at any length. One run
|
|
345
|
+
* shipped 9.4 seconds of it in a 75 second video.
|
|
346
|
+
*
|
|
347
|
+
* Playwright records continuously and cannot be paused across a navigation, so
|
|
348
|
+
* these frames cannot be prevented at capture time. They are removed here
|
|
349
|
+
* instead, and only where the audio is silent too, so no narration is ever cut.
|
|
350
|
+
*/
|
|
351
|
+
declare function detectBlackSpans(inputPath: string, minDurSecs: number): Promise<SilenceSpan[]>;
|
|
352
|
+
/**
|
|
353
|
+
* Spans where the picture is essentially white.
|
|
354
|
+
*
|
|
355
|
+
* blackdetect over an inverted picture. A page that flashes white between
|
|
356
|
+
* routes is as empty as one that flashes black, and a light-themed app makes
|
|
357
|
+
* white the common case, so detecting only one of the two fixes half the sites.
|
|
358
|
+
*/
|
|
359
|
+
declare function detectWhiteSpans(inputPath: string, minDurSecs: number): Promise<SilenceSpan[]>;
|
|
360
|
+
/**
|
|
361
|
+
* Where two sets of spans overlap.
|
|
362
|
+
*
|
|
363
|
+
* A frozen picture is only safe to speed up while nothing is being said over
|
|
364
|
+
* it - rushing a still frame is free, rushing a sentence is not. The overlap
|
|
365
|
+
* of "frozen" and "silent" is exactly the footage with nothing happening in
|
|
366
|
+
* either channel.
|
|
367
|
+
*/
|
|
368
|
+
declare function intersectSpans(a: SilenceSpan[], b: SilenceSpan[]): SilenceSpan[];
|
|
369
|
+
/** Fold overlapping or touching spans into one. */
|
|
370
|
+
declare function mergeSpans(spans: SilenceSpan[]): SilenceSpan[];
|
|
371
|
+
/** atempo caps at 2x per instance, so a bigger speed-up is a chain of them. */
|
|
372
|
+
declare function atempoChain(factor: number): string;
|
|
373
|
+
/**
|
|
374
|
+
* Build the filter_complex that plays `spans` fast and the rest untouched.
|
|
375
|
+
*
|
|
376
|
+
* Split out from the run so the graph can be unit-tested: an ffmpeg filter
|
|
377
|
+
* error surfaces as a wall of text at render time, which is a bad place to
|
|
378
|
+
* discover an off-by-one in the segment list.
|
|
379
|
+
*/
|
|
380
|
+
declare function buildCompressionFilter(spans: SilenceSpan[], totalSecs: number, maxSecs: number, protectAfterSecs?: number,
|
|
381
|
+
/**
|
|
382
|
+
* Everything BEFORE this is the opening card, protected for the same reason
|
|
383
|
+
* as the closing one and previously not protected at all.
|
|
384
|
+
*
|
|
385
|
+
* It did not need to be while the intro card was narrated over: a spoken
|
|
386
|
+
* card is not a silent span. A card prepended as its own step has no
|
|
387
|
+
* narration, so it is silent AND frozen, and this pass gave it the frozen
|
|
388
|
+
* ceiling of 0.6s - a title card gone before it can be read.
|
|
389
|
+
*/
|
|
390
|
+
protectBeforeSecs?: number): string | null;
|
|
391
|
+
/**
|
|
392
|
+
* Drop spans from a silent video, in its own timeline.
|
|
393
|
+
*
|
|
394
|
+
* The recording has no audio yet, so there is nothing to desync: this is the
|
|
395
|
+
* one place in the pipeline where a span can simply be deleted. Doing it here
|
|
396
|
+
* rather than after compose also avoids the coordinate problem that made an
|
|
397
|
+
* earlier attempt a no-op. A recording is 145s where its composed video is
|
|
398
|
+
* 74s, because compose trims the page load off the front and the later passes
|
|
399
|
+
* squeeze the rest, so a span measured during recording means nothing once
|
|
400
|
+
* those have run.
|
|
401
|
+
*/
|
|
402
|
+
declare function cutSpansFromVideo(inputPath: string, outputPath: string, spans: Array<{
|
|
403
|
+
start: number;
|
|
404
|
+
end: number;
|
|
405
|
+
}>, preset: string): Promise<boolean>;
|
|
406
|
+
/**
|
|
407
|
+
* How much faster a loading screen plays than the footage around it.
|
|
408
|
+
*
|
|
409
|
+
* The empty stretches in a recording are almost always one thing: the app's
|
|
410
|
+
* own splash while a route loads. A sellular run spent 17 of its 66 seconds
|
|
411
|
+
* on that screen. Five is fast enough that it reads as a flicker rather than
|
|
412
|
+
* a wait, and slow enough that the logo is still legible going past.
|
|
413
|
+
*/
|
|
414
|
+
declare const EMPTY_SPAN_SPEED = 5;
|
|
415
|
+
/**
|
|
416
|
+
* Build the filter that removes every empty span from view without changing
|
|
417
|
+
* how long anything lasts, one step at a time.
|
|
418
|
+
*
|
|
419
|
+
* The point of doing it per step is that every step boundary comes out exactly
|
|
420
|
+
* where it went in. Cutting empty frames is the obvious move and it does not
|
|
421
|
+
* work: the narration for each step is already synthesised at a fixed length,
|
|
422
|
+
* so shortening the picture under it desyncs everything downstream, and the
|
|
423
|
+
* guard that stops that from happening starves the cut until it removes
|
|
424
|
+
* nothing at all. So the seconds are not removed, they are re-spent.
|
|
425
|
+
*
|
|
426
|
+
* A step with real footage either side of the splash plays the splash at
|
|
427
|
+
* `speed` and stretches that footage to fill exactly what it freed. A step
|
|
428
|
+
* that is nearly all splash - a route whose loading screen outlasts the line
|
|
429
|
+
* narrated over it - has nothing to stretch, so the splash is replaced by a
|
|
430
|
+
* still of the page that arrives after it. Both leave the step the length it
|
|
431
|
+
* was, and neither leaves an empty screen up.
|
|
432
|
+
*
|
|
433
|
+
* Returns null when there is nothing worth doing, so the caller can skip a
|
|
434
|
+
* re-encode that would change nothing.
|
|
435
|
+
*/
|
|
436
|
+
declare function buildRetimeFilter(spans: Array<{
|
|
437
|
+
start: number;
|
|
438
|
+
end: number;
|
|
439
|
+
}>, stepBounds: Array<{
|
|
440
|
+
from: number;
|
|
441
|
+
to: number;
|
|
442
|
+
}>, speed: number, fps: number): string | null;
|
|
443
|
+
/**
|
|
444
|
+
* Play the empty stretches of a silent recording fast, keeping every step the
|
|
445
|
+
* length it was.
|
|
446
|
+
*
|
|
447
|
+
* Resolves false when the filter had nothing to do or ffmpeg refused it, and
|
|
448
|
+
* the caller keeps the original recording.
|
|
449
|
+
*/
|
|
450
|
+
declare function retimeSpansInVideo(inputPath: string, outputPath: string, spans: Array<{
|
|
451
|
+
start: number;
|
|
452
|
+
end: number;
|
|
453
|
+
}>, stepBounds: Array<{
|
|
454
|
+
from: number;
|
|
455
|
+
to: number;
|
|
456
|
+
}>, speed: number, fps: number, preset: string): Promise<boolean>;
|
|
457
|
+
declare function compressSilentSpans(inputPath: string, outputPath: string, totalSecs: number, maxSecs: number, preset: string, protectAfterSecs?: number,
|
|
458
|
+
/**
|
|
459
|
+
* The narration on its own, when there is one.
|
|
460
|
+
*
|
|
461
|
+
* Silence has to be measured against the voice, not against the mix: any
|
|
462
|
+
* backing track - a music bed, room tone - is continuous by design, so a
|
|
463
|
+
* mixed file has no silence in it at all and every pause looks like speech.
|
|
464
|
+
*/
|
|
465
|
+
narrationPath?: string,
|
|
466
|
+
/** Opening card to leave alone, in seconds from the start. See the filter. */
|
|
467
|
+
protectBeforeSecs?: number,
|
|
468
|
+
/**
|
|
469
|
+
* Spans the recorder KNOWS were a page loading rather than a page showing.
|
|
470
|
+
*
|
|
471
|
+
* Cut on the recorder's word, not on a detector's guess. blackdetect cannot
|
|
472
|
+
* tell a loading screen from a dark-themed app, and the silence guard
|
|
473
|
+
* refuses to cut anything with a voice over it, which is exactly the case
|
|
474
|
+
* here: the narration describing a result is queued behind the wait for it.
|
|
475
|
+
* The recorder measured this span, so it needs no evidence.
|
|
476
|
+
*/
|
|
477
|
+
leadSpans?: Array<{
|
|
478
|
+
start: number;
|
|
479
|
+
end: number;
|
|
480
|
+
}>,
|
|
481
|
+
/**
|
|
482
|
+
* Spans that are playing CONTENT, however quiet they are.
|
|
483
|
+
*
|
|
484
|
+
* A clip spliced in under no narration is silent by nature and frozen to a
|
|
485
|
+
* freeze detector between cuts, so every heuristic here reads it as dead
|
|
486
|
+
* air and compresses it to the ceiling. A nine-second demo arrived as
|
|
487
|
+
* three-quarters of a second. Silence during a clip is the clip, not a
|
|
488
|
+
* hang, so these spans are removed from the compression list outright.
|
|
489
|
+
*/
|
|
490
|
+
keepSpans?: Array<{
|
|
491
|
+
start: number;
|
|
492
|
+
end: number;
|
|
493
|
+
}>): Promise<boolean>;
|
|
494
|
+
/**
|
|
495
|
+
* Speed up only the middle of a video, leaving its ends untouched.
|
|
496
|
+
*
|
|
497
|
+
* A demo opens on a title card and closes on a credits card. Speeding the
|
|
498
|
+
* whole file to fit a length limit rushes both: the card the viewer is meant
|
|
499
|
+
* to read, and the offer they are meant to write down. Only the demo between
|
|
500
|
+
* them is compressible.
|
|
501
|
+
*
|
|
502
|
+
* Returns false when the ends already fill the budget - there is nothing to
|
|
503
|
+
* squeeze then, and the honest answer is a shorter plan.
|
|
504
|
+
*/
|
|
505
|
+
declare function speedUpMiddle(inputPath: string, outputPath: string, totalSecs: number, targetSecs: number, introSecs: number, outroSecs: number, maxFactor: number, preset: string): Promise<{
|
|
506
|
+
ok: boolean;
|
|
507
|
+
factor: number;
|
|
508
|
+
reachedTarget: boolean;
|
|
509
|
+
}>;
|
|
510
|
+
|
|
511
|
+
/**
|
|
512
|
+
* Fill the frame with the card's own colours, then lay the card on it.
|
|
513
|
+
*
|
|
514
|
+
* A 1200x630 OG card is the shape every site publishes and the wrong shape for
|
|
515
|
+
* a 9:16 opening: centred, it floats in two thirds of a frame of flat #1a1a2e,
|
|
516
|
+
* which is what "letterboxed with dead space" looks like. Cropping it to fill
|
|
517
|
+
* instead would throw away the half of the card the headline is written
|
|
518
|
+
* across. So the backdrop is built from the card itself - the frame is full,
|
|
519
|
+
* the colours are the card's own, and not a pixel of the card is lost.
|
|
520
|
+
*
|
|
521
|
+
* The backdrop is RESAMPLED DOWN to a handful of pixels before it is blown
|
|
522
|
+
* back up, rather than blurred at full resolution. Blur is the obvious move
|
|
523
|
+
* and it is wrong here, because an OG card is mostly TYPE: a 1200px-wide
|
|
524
|
+
* headline enlarged 3x stays legible through sigma=28, so the opening frame
|
|
525
|
+
* showed the same words twice - huge and soft behind the card, sharp on it.
|
|
526
|
+
* That reads as a rendering fault rather than a design, and it was the first
|
|
527
|
+
* thing anyone saw of the product.
|
|
528
|
+
*
|
|
529
|
+
* Ten pixels of height cannot carry a glyph, so nothing survives to be read
|
|
530
|
+
* while the card's colour field does. The gblur afterwards only takes the
|
|
531
|
+
* blockiness off the upscale.
|
|
532
|
+
*
|
|
533
|
+
* The downscale is its own `scale` with BOTH dimensions given. `scale=40:-2`
|
|
534
|
+
* is silently ignored at these sizes - the first attempt at this looked
|
|
535
|
+
* completely unchanged because of it, which is why the test asserts the
|
|
536
|
+
* literal pair rather than just "smaller than the frame".
|
|
537
|
+
*/
|
|
538
|
+
declare function coverFilterChain(inputIdx: number, width: number, height: number): string;
|
|
163
539
|
|
|
164
540
|
/**
|
|
165
541
|
* Splice video files into the recording at specific timestamps
|
|
@@ -248,32 +624,6 @@ interface LenientParse<T> {
|
|
|
248
624
|
*/
|
|
249
625
|
declare function parseAIJson<T>(content: string): LenientParse<T> | null;
|
|
250
626
|
|
|
251
|
-
declare const CURSOR_CSS = "\n #qp-cursor {\n position: fixed;\n width: 32px;\n height: 38px;\n background: url(\"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAMgAAAEsBAMAAAB01OGNAAAAIGNIUk0AAHomAACAhAAA+gAAAIDoAAB1MAAA6mAAADqYAAAXcJy6UTwAAAAtUExURUdwTAAAAAAAAAAAAAAAAAAAAAAAAB0dHVZWVtvb2/X19f///6mpqXV1df////vm1ewAAAAGdFJOUwAltTfpcLO208EAAAABYktHRA5vvTBPAAAAB3RJTUUH6QwFFREKplBWTAAABqdJREFUeNq93c+OFFUUx/EeCHud6H7A4HpCOu6NiWsWZh5Ax8YFbAdNWBPxBcCo+yGjD+CAW1ywh0Rexu6uulX3Vp269/c7f7o3Bpjhk6+H6TrU0LdXR6sDPI4/OgCyvn0I5NsDpKw3B0hZbw6Qst4cIGWLxKdskfiUHRKeskPCU/ZIdMoeiU7pkOCUDglO6ZHYlB6JTUlIaEpCQlMGJDJlQCJTRiQwZUQCUzIkLiVD4lJyJCwlR8JSCiQqpUCiUkokKKVEglImSEzKBIlJmSIhKT3yY2hKjzx7GpnSI7+8iExJyN+RKQl5FZkyIJEpAxKZMiKBKSMSmJIhcSkZEpeSI2EpORKWUiBRKQUSlVIiQSklEpQyQWJSJkhMyhQJSZkiISkzJCJlhkSkzJGAlDkSkCIg/ikC4p8iIe4pEuKeIiLeKSLinSIjziky4pyygPimLCC+KUuIa8oS4pqyiHimLCKeKcuIY8oy4phSQfxSKohfSg1xS6khbilVxCulinil1BGnlDrilNJAfFIaiE9KC3FJaSEuKU3EI6WJeKS0EYeUNuKQAiD2FACxpyCIOQVBzCkQYk2BEGsKhhhTMMSYAiK2FBCxpaCIKQVFTCkwYkmBEUsKjhhScMSQQiD6FALRpzCIOoVB1CkUok2hEG0KhyhTOESZQiK6FBLRpbCIKoVFVCk0okmhEU0KjyhSeESRokD4FAXCp2gQOkWD0CkqhE1RIWyKDiFTdAiZokS4FCXCpWgRKkWLUClqhElRI0yKHiFS9AiRYkDwFAOCp1gQOMWCwCkmBE0xIWiKDQFTbAiYYkSwFCOCpVgRKMWKQClmBEkxI0iKHQFS7AiQ4oC0UxyQdooH0kzxQJopLkgrxQVppfggjRQfpJHihNRTnJB6ihdSTfFCqiluSC3FDaml+CGVFD+kkuKILKc4IsspnshiiieymOKKLKW4Iq8u5RRf5FpO8UUWpuKMyCnOiJzijYgp3oiY4o5IKe6IlEIhL38FHr9fzFK47/0On195PL6YpVDIX4CRP1IKhbxGSvLHbQUy/snhUrg/XW9IpE/hkOcs0qVwyD8s0qWQ349nJ9+lcAg9+S6FfFqhJ79PIZFLGtmlkMifPLJNYf/5wgWvfAYg11f5j951H39+B3/cBZA/3uY/+nccJ/5oItdPi19Lkz91RV5snuQ/fN0jJ57I9svvkTT5rzyR7RX7wW/C5L93RPbPI8Xknysmj7zWQZz8fTeke0J8aJ088vqTR1czl5t8Fel/w2Ly12/6oXghaRn8IEz+nJh8DRkuUc/yn02XYGLy0Ou0ysn3X44+/75rvNb+IE3+OxdkXM/Lr3l+8stIvjSIkz91QMaQzeYn2+QXkWL7KSafnogd/jlcHiJPHn8iXkLKNe7BW9Pk26/73T/+y3/tkp38AjLdR4vJp+XrxIhMQuRL8Jc2ZLZYF5OnL8Gt18ff6f9bTJ5dvkQku6vw6Ub4AHb5ap1ZcPS5MHn2Eiwhxe2Rs8rk0Utw8xyJ/gNMy5eAlPd5bkiTJ5ev5tkeN7+uTP6+FpncsEqTt6zd7fNWpMmTy9cMmd15+0SYPLl8tc/AuVWZPLh8TZH5LURx8unOFzb59rlEafKG5WuCSPdC0+SvhMljyxdwVtSx9DVPTb5ExPvTafIfpMmf8oh4pz1NXr92F8jCtz+kyVPLF3Km2jfdz+mXrxxZCLFPHjnnTpw8s3xlyFLIMHn18gWdPVibPLJ8jchiyPBB5eTf4ZOHzoNMky+eiInlCzrZMk1eWruR5Qs6o/NImjxxCcZOGz2rTB64BGPnpkqTJ5Yv7ARY4/LVIz9XQ4bJK5cv7FRe49oNni/cT165fIEnJdfW7vYTMXjm8w3pax6ePHh6tW35As/hltfui8ZniUglvDb55vKFno0urd3wJRg95b26dp8SSO3/rTh5dPlCT94XJ48uX/B7CKTJa5Yv+N0QLJOH39fBsnavwRBx7X75vv/kExBpftn2k38oGOAraYBnuS+6DxwvwaPRmvwaDZmt3ZkBvlwHuFBPJp8breVrjYZM1u7SaEx+jYaUk58Y58gLj5i/lO0mPzE2d4GZYPetxrWbNDoEu68wrN2ssUfAG3Bp7X7CGnsE/R7F2UZ6tI0dwt5Dpo3dJ8LfbDlWGivm3dpuKY0V875z6RLMGtw76J3pDO69ANc6Y3WPCBn+wkUaq48JY3giJg3ukU8+ysgnH2ek5SvUGCYfaaTJhxr95GONbvLRxm7y4cb2iTjeWN08gLFaaYz/AWAIeJtzdhugAAAAAElFTkSuQmCC\") no-repeat;\n background-size: contain;\n pointer-events: none;\n z-index: 2147483647;\n transform: translate(-3px, -3px);\n transition: left 800ms ease-out, top 800ms ease-out, transform 0.08s ease-out;\n filter: drop-shadow(1px 1px 2px rgba(0,0,0,0.3));\n }\n #qp-cursor.clicking {\n transform: translate(-3px, -3px) scale(0.9);\n }\n";
|
|
252
|
-
/**
|
|
253
|
-
* JavaScript to inject for cursor tracking
|
|
254
|
-
*/
|
|
255
|
-
declare const CURSOR_SCRIPT = "\n (function() {\n if (document.getElementById('qp-cursor')) return;\n const cursor = document.createElement('div');\n cursor.id = 'qp-cursor';\n document.body.appendChild(cursor);\n\n document.addEventListener('mousemove', (e) => {\n cursor.style.left = e.clientX + 'px';\n cursor.style.top = e.clientY + 'px';\n });\n\n document.addEventListener('mousedown', () => cursor.classList.add('clicking'));\n document.addEventListener('mouseup', () => cursor.classList.remove('clicking'));\n\n // Start at center\n cursor.style.left = window.innerWidth / 2 + 'px';\n cursor.style.top = window.innerHeight / 2 + 'px';\n })();\n";
|
|
256
|
-
/**
|
|
257
|
-
* Generate ASS subtitle file header
|
|
258
|
-
*/
|
|
259
|
-
declare function generateASSHeader(width: number, height: number, font: string, size: number, primaryColor: string, secondaryColor: string, outlineColor: string, backColor: string, bold: boolean, outlineWidth: number, shadow: number, marginBottom: number): string;
|
|
260
|
-
|
|
261
|
-
/**
|
|
262
|
-
* Subscription and user registration client
|
|
263
|
-
* CLI-only headless registration fallback. Tier/verification checks
|
|
264
|
-
* are now in the SparkStripe gating module (getTierByEmail, isVerified).
|
|
265
|
-
*/
|
|
266
|
-
interface RegistrationResult {
|
|
267
|
-
success: boolean;
|
|
268
|
-
pendingVerification?: boolean;
|
|
269
|
-
alreadyVerified?: boolean;
|
|
270
|
-
error?: string;
|
|
271
|
-
}
|
|
272
|
-
/**
|
|
273
|
-
* Register user for free plan via SparkStripe
|
|
274
|
-
*/
|
|
275
|
-
declare function registerFreeUser(email: string): Promise<RegistrationResult>;
|
|
276
|
-
|
|
277
627
|
/** Minimal step shape needed by TTS modules — compatible with full Step from plan.ts */
|
|
278
628
|
interface TTSStep {
|
|
279
629
|
caption: string;
|
|
@@ -304,9 +654,111 @@ interface VoiceoverMeta {
|
|
|
304
654
|
totalCaptionLength: number;
|
|
305
655
|
stepCount: number;
|
|
306
656
|
voice: string;
|
|
657
|
+
rate?: string;
|
|
307
658
|
stepDurations: Record<number, number>;
|
|
659
|
+
stepWords?: Record<number, WordTiming[]>;
|
|
660
|
+
}
|
|
661
|
+
|
|
662
|
+
/**
|
|
663
|
+
* CSS and ASS styling for QuickPeek
|
|
664
|
+
*/
|
|
665
|
+
|
|
666
|
+
declare const CURSOR_CSS = "\n #qp-cursor {\n position: fixed;\n width: 32px;\n height: 38px;\n background: url(\"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAMgAAAEsBAMAAAB01OGNAAAAIGNIUk0AAHomAACAhAAA+gAAAIDoAAB1MAAA6mAAADqYAAAXcJy6UTwAAAAtUExURUdwTAAAAAAAAAAAAAAAAAAAAAAAAB0dHVZWVtvb2/X19f///6mpqXV1df////vm1ewAAAAGdFJOUwAltTfpcLO208EAAAABYktHRA5vvTBPAAAAB3RJTUUH6QwFFREKplBWTAAABqdJREFUeNq93c+OFFUUx/EeCHud6H7A4HpCOu6NiWsWZh5Ax8YFbAdNWBPxBcCo+yGjD+CAW1ywh0Rexu6uulX3Vp269/c7f7o3Bpjhk6+H6TrU0LdXR6sDPI4/OgCyvn0I5NsDpKw3B0hZbw6Qst4cIGWLxKdskfiUHRKeskPCU/ZIdMoeiU7pkOCUDglO6ZHYlB6JTUlIaEpCQlMGJDJlQCJTRiQwZUQCUzIkLiVD4lJyJCwlR8JSCiQqpUCiUkokKKVEglImSEzKBIlJmSIhKT3yY2hKjzx7GpnSI7+8iExJyN+RKQl5FZkyIJEpAxKZMiKBKSMSmJIhcSkZEpeSI2EpORKWUiBRKQUSlVIiQSklEpQyQWJSJkhMyhQJSZkiISkzJCJlhkSkzJGAlDkSkCIg/ikC4p8iIe4pEuKeIiLeKSLinSIjziky4pyygPimLCC+KUuIa8oS4pqyiHimLCKeKcuIY8oy4phSQfxSKohfSg1xS6khbilVxCulinil1BGnlDrilNJAfFIaiE9KC3FJaSEuKU3EI6WJeKS0EYeUNuKQAiD2FACxpyCIOQVBzCkQYk2BEGsKhhhTMMSYAiK2FBCxpaCIKQVFTCkwYkmBEUsKjhhScMSQQiD6FALRpzCIOoVB1CkUok2hEG0KhyhTOESZQiK6FBLRpbCIKoVFVCk0okmhEU0KjyhSeESRokD4FAXCp2gQOkWD0CkqhE1RIWyKDiFTdAiZokS4FCXCpWgRKkWLUClqhElRI0yKHiFS9AiRYkDwFAOCp1gQOMWCwCkmBE0xIWiKDQFTbAiYYkSwFCOCpVgRKMWKQClmBEkxI0iKHQFS7AiQ4oC0UxyQdooH0kzxQJopLkgrxQVppfggjRQfpJHihNRTnJB6ihdSTfFCqiluSC3FDaml+CGVFD+kkuKILKc4IsspnshiiieymOKKLKW4Iq8u5RRf5FpO8UUWpuKMyCnOiJzijYgp3oiY4o5IKe6IlEIhL38FHr9fzFK47/0On195PL6YpVDIX4CRP1IKhbxGSvLHbQUy/snhUrg/XW9IpE/hkOcs0qVwyD8s0qWQ349nJ9+lcAg9+S6FfFqhJ79PIZFLGtmlkMifPLJNYf/5wgWvfAYg11f5j951H39+B3/cBZA/3uY/+nccJ/5oItdPi19Lkz91RV5snuQ/fN0jJ57I9svvkTT5rzyR7RX7wW/C5L93RPbPI8Xknysmj7zWQZz8fTeke0J8aJ088vqTR1czl5t8Fel/w2Ly12/6oXghaRn8IEz+nJh8DRkuUc/yn02XYGLy0Ou0ysn3X44+/75rvNb+IE3+OxdkXM/Lr3l+8stIvjSIkz91QMaQzeYn2+QXkWL7KSafnogd/jlcHiJPHn8iXkLKNe7BW9Pk26/73T/+y3/tkp38AjLdR4vJp+XrxIhMQuRL8Jc2ZLZYF5OnL8Gt18ff6f9bTJ5dvkQku6vw6Ub4AHb5ap1ZcPS5MHn2Eiwhxe2Rs8rk0Utw8xyJ/gNMy5eAlPd5bkiTJ5ev5tkeN7+uTP6+FpncsEqTt6zd7fNWpMmTy9cMmd15+0SYPLl8tc/AuVWZPLh8TZH5LURx8unOFzb59rlEafKG5WuCSPdC0+SvhMljyxdwVtSx9DVPTb5ExPvTafIfpMmf8oh4pz1NXr92F8jCtz+kyVPLF3Km2jfdz+mXrxxZCLFPHjnnTpw8s3xlyFLIMHn18gWdPVibPLJ8jchiyPBB5eTf4ZOHzoNMky+eiInlCzrZMk1eWruR5Qs6o/NImjxxCcZOGz2rTB64BGPnpkqTJ5Yv7ARY4/LVIz9XQ4bJK5cv7FRe49oNni/cT165fIEnJdfW7vYTMXjm8w3pax6ePHh6tW35As/hltfui8ZniUglvDb55vKFno0urd3wJRg95b26dp8SSO3/rTh5dPlCT94XJ48uX/B7CKTJa5Yv+N0QLJOH39fBsnavwRBx7X75vv/kExBpftn2k38oGOAraYBnuS+6DxwvwaPRmvwaDZmt3ZkBvlwHuFBPJp8breVrjYZM1u7SaEx+jYaUk58Y58gLj5i/lO0mPzE2d4GZYPetxrWbNDoEu68wrN2ssUfAG3Bp7X7CGnsE/R7F2UZ6tI0dwt5Dpo3dJ8LfbDlWGivm3dpuKY0V875z6RLMGtw76J3pDO69ANc6Y3WPCBn+wkUaq48JY3giJg3ukU8+ysgnH2ek5SvUGCYfaaTJhxr95GONbvLRxm7y4cb2iTjeWN08gLFaaYz/AWAIeJtzdhugAAAAAElFTkSuQmCC\") no-repeat;\n background-size: contain;\n pointer-events: none;\n z-index: 2147483647;\n transform: translate(-3px, -3px);\n transition: left 800ms ease-out, top 800ms ease-out, transform 0.08s ease-out;\n filter: drop-shadow(1px 1px 2px rgba(0,0,0,0.3));\n }\n #qp-cursor.clicking {\n transform: translate(-3px, -3px) scale(0.9);\n }\n";
|
|
667
|
+
/**
|
|
668
|
+
* JavaScript to inject for cursor tracking
|
|
669
|
+
*/
|
|
670
|
+
declare const CURSOR_SCRIPT = "\n (function() {\n if (document.getElementById('qp-cursor')) return;\n const cursor = document.createElement('div');\n cursor.id = 'qp-cursor';\n document.body.appendChild(cursor);\n\n document.addEventListener('mousemove', (e) => {\n cursor.style.left = e.clientX + 'px';\n cursor.style.top = e.clientY + 'px';\n });\n\n document.addEventListener('mousedown', () => cursor.classList.add('clicking'));\n document.addEventListener('mouseup', () => cursor.classList.remove('clicking'));\n\n // Start at center\n cursor.style.left = window.innerWidth / 2 + 'px';\n cursor.style.top = window.innerHeight / 2 + 'px';\n })();\n";
|
|
671
|
+
/**
|
|
672
|
+
* Generate ASS subtitle file header
|
|
673
|
+
*/
|
|
674
|
+
declare function generateASSHeader(width: number, height: number, font: string, size: number, primaryColor: string, secondaryColor: string, outlineColor: string, backColor: string, bold: boolean, outlineWidth: number, shadow: number, marginBottom: number, alignment?: number): string;
|
|
675
|
+
/**
|
|
676
|
+
* Linux font-substitution guard (from VidLet): "Arial Black" is not installed
|
|
677
|
+
* on Linux, and fontconfig silently falls back to Noto Sans Regular with a
|
|
678
|
+
* synthesised weight. DejaVu Sans ships a real Bold face on every distro this
|
|
679
|
+
* runs on.
|
|
680
|
+
*/
|
|
681
|
+
declare function resolveCaptionFont(font: string): string;
|
|
682
|
+
/** ASS numpad alignment: 1-3 bottom, 4-6 middle, 7-9 top. */
|
|
683
|
+
declare function positionToAlignment(position: CaptionPosition, leftAligned?: boolean): number;
|
|
684
|
+
/** One step's caption words, timed in absolute ms on the final video timeline. */
|
|
685
|
+
interface CaptionCue {
|
|
686
|
+
words: WordTiming[];
|
|
687
|
+
}
|
|
688
|
+
interface CaptionRenderContext {
|
|
689
|
+
cues: CaptionCue[];
|
|
690
|
+
width: number;
|
|
691
|
+
height: number;
|
|
692
|
+
captions: CaptionStyle;
|
|
308
693
|
}
|
|
694
|
+
/**
|
|
695
|
+
* The coordinate space captions are drawn in, which is NOT the output size.
|
|
696
|
+
*
|
|
697
|
+
* ASS declares its own PlayResX/PlayResY and libass scales that space onto
|
|
698
|
+
* whatever frame it burns into. Handing it the real output size made every
|
|
699
|
+
* caption number mean something different at every resolution: at 360x640 a
|
|
700
|
+
* 112px font goes from 5.8% of the frame height to 17.5%, and a 300px bottom
|
|
701
|
+
* margin - the "lower third" the preset intends - lands 47% of the way up,
|
|
702
|
+
* which is dead centre over the product. So the space is pinned to the size
|
|
703
|
+
* the numbers were authored for and libass does the scaling.
|
|
704
|
+
*
|
|
705
|
+
* A 1080x1920 render is unaffected: the scale factor is exactly 1.
|
|
706
|
+
*/
|
|
707
|
+
declare function captionCanvas(width: number, height: number): {
|
|
708
|
+
width: number;
|
|
709
|
+
height: number;
|
|
710
|
+
};
|
|
711
|
+
/**
|
|
712
|
+
* The caption style a render actually uses: the canvas decides the geometry,
|
|
713
|
+
* captions.css supplies the face and the colours, `video.captions` overrides
|
|
714
|
+
* anything.
|
|
715
|
+
*
|
|
716
|
+
* The canvas wins over captions.css because the numbers in that file were
|
|
717
|
+
* never a preference. QuickPeek writes it on first run with 56px text 80px
|
|
718
|
+
* off the bottom - fine for a 16:9 tutorial, a desktop subtitle lost under
|
|
719
|
+
* the player chrome on the 9:16 short that is now the default shape - and no
|
|
720
|
+
* user ever chose those numbers. `--profile short` already overrode them for
|
|
721
|
+
* exactly this reason; the only change here is that a plain run gets the same
|
|
722
|
+
* treatment instead of depending on which command was typed.
|
|
723
|
+
*
|
|
724
|
+
* What the frame size genuinely cannot know stays with the CSS: the typeface
|
|
725
|
+
* and the colours, which are brand decisions. Anything else belongs to
|
|
726
|
+
* `video.captions` in quickpeek.config.json, which wins over both.
|
|
727
|
+
*/
|
|
728
|
+
declare function resolveCaptions(css: CaptionStyle, video: Config['video']): CaptionStyle;
|
|
729
|
+
/** Render a complete ASS document for the configured caption preset. */
|
|
730
|
+
declare function renderCaptionsAss(context: CaptionRenderContext): string;
|
|
309
731
|
|
|
732
|
+
/**
|
|
733
|
+
* Subscription and user registration client
|
|
734
|
+
* CLI-only headless registration fallback. Tier/verification checks
|
|
735
|
+
* are now in the SparkStripe gating module (getTierByEmail, isVerified).
|
|
736
|
+
*/
|
|
737
|
+
interface RegistrationResult {
|
|
738
|
+
success: boolean;
|
|
739
|
+
pendingVerification?: boolean;
|
|
740
|
+
alreadyVerified?: boolean;
|
|
741
|
+
error?: string;
|
|
742
|
+
}
|
|
743
|
+
/**
|
|
744
|
+
* Register user for free plan via SparkStripe
|
|
745
|
+
*/
|
|
746
|
+
declare function registerFreeUser(email: string): Promise<RegistrationResult>;
|
|
747
|
+
|
|
748
|
+
/**
|
|
749
|
+
* Silence before the first word.
|
|
750
|
+
*
|
|
751
|
+
* Narration that starts on frame one talks over a viewer who is still taking
|
|
752
|
+
* in the opening card, and the first syllable lands before anyone is
|
|
753
|
+
* listening. Part of the audio rather than a compose-time delay, so the
|
|
754
|
+
* captions, the step durations and the dead-air pass all measure the same
|
|
755
|
+
* timeline.
|
|
756
|
+
*
|
|
757
|
+
* Here rather than beside the code that lays it down, because the
|
|
758
|
+
* realignment pass has to know about it too and importing across those two
|
|
759
|
+
* modules the other way would close a cycle.
|
|
760
|
+
*/
|
|
761
|
+
declare const NARRATION_LEAD_IN_MS = 300;
|
|
310
762
|
/**
|
|
311
763
|
* Load existing voiceover metadata if available
|
|
312
764
|
*/
|
|
@@ -323,7 +775,7 @@ interface Step {
|
|
|
323
775
|
/**
|
|
324
776
|
* Check if existing voiceover can be reused
|
|
325
777
|
*/
|
|
326
|
-
declare function canReuseVoiceover(steps: Step[], voice: string, existingMeta: VoiceoverMeta | null): {
|
|
778
|
+
declare function canReuseVoiceover(steps: Step[], voice: string, rate: string, existingMeta: VoiceoverMeta | null): {
|
|
327
779
|
canReuse: boolean;
|
|
328
780
|
reason?: string;
|
|
329
781
|
};
|
|
@@ -347,17 +799,81 @@ declare function loadExistingVoiceover(steps: Step[], outputDir: string, meta: V
|
|
|
347
799
|
}>;
|
|
348
800
|
|
|
349
801
|
declare function resolveVoice(config: Config): string | null;
|
|
802
|
+
/** The narration pace for this config, as the enum the meta stores. */
|
|
803
|
+
declare function resolveRate(config: Config): VoiceRate;
|
|
804
|
+
/**
|
|
805
|
+
* Characters an SSML payload cannot carry.
|
|
806
|
+
*
|
|
807
|
+
* Both engines build SSML and drop the caption into it as-is, so a bare "&"
|
|
808
|
+
* makes the request invalid XML by the time the service sees it. It does not
|
|
809
|
+
* fail loudly: the Node engine returns an empty stream ("No audio data
|
|
810
|
+
* received"), the step is left with no clip, and because `stepWords` only
|
|
811
|
+
* records steps that produced words, the step vanishes from the metadata too.
|
|
812
|
+
* The video still renders - one step just has no voice.
|
|
813
|
+
*
|
|
814
|
+
* Found on Anysite's closing card, "Live Web Data for GTM & Marketing Teams":
|
|
815
|
+
* the whole closing line was silent while every other step in the same run
|
|
816
|
+
* synthesised fine. That is the worst shape a bug can take here, because the
|
|
817
|
+
* missing narration is at the end where nobody re-watches.
|
|
818
|
+
*
|
|
819
|
+
* Spoken, not escaped. "&" is read "and" by any English speaker, the burned
|
|
820
|
+
* caption still shows the symbol, and nothing downstream has to trust an
|
|
821
|
+
* engine to unescape an entity it may not have escaped itself.
|
|
822
|
+
*/
|
|
823
|
+
declare function speakSymbols(text: string): string;
|
|
350
824
|
declare function generateVoiceover(steps: TTSStep[], config: Config, workDir: string, outputDir: string, logError: (context: string, error: unknown) => Promise<void>, task: (text: string) => {
|
|
351
825
|
succeed: (text?: string) => void;
|
|
352
826
|
}, info: (text: string) => void): Promise<AudioResult>;
|
|
353
827
|
|
|
354
828
|
/**
|
|
355
|
-
*
|
|
356
|
-
*
|
|
829
|
+
* A word as the synthesizer actually spoke it, in ms from the start of the
|
|
830
|
+
* clip it was asked to produce — before any trimming this pipeline does to it.
|
|
357
831
|
*/
|
|
358
|
-
|
|
832
|
+
interface SpokenWord {
|
|
833
|
+
word: string;
|
|
834
|
+
start: number;
|
|
835
|
+
end: number;
|
|
836
|
+
}
|
|
837
|
+
/** Starts a new run, so the next line chooses the engine again. */
|
|
838
|
+
declare function resetVoiceEngine(): void;
|
|
839
|
+
declare function synthesizeSpeech(rawText: string, outputPath: string, voice: string, rate?: string): Promise<SpokenWord[] | null>;
|
|
840
|
+
/**
|
|
841
|
+
* Pull the word boundaries out of the raw metadata stream.
|
|
842
|
+
*
|
|
843
|
+
* Each websocket frame is a complete JSON object, but a Readable is free to
|
|
844
|
+
* coalesce them into one chunk, so the concatenated text is split on brace
|
|
845
|
+
* depth rather than parsed whole. A frame that does not parse is skipped: a
|
|
846
|
+
* missing word costs a slightly-off highlight, throwing costs the clip.
|
|
847
|
+
*/
|
|
848
|
+
declare function parseWordBoundaries(raw: string): SpokenWord[] | null;
|
|
359
849
|
|
|
360
850
|
declare function estimateWordTimings(caption: string, duration: number): WordTiming[];
|
|
851
|
+
/**
|
|
852
|
+
* Marry the authored caption text to the timings the voice reported.
|
|
853
|
+
*
|
|
854
|
+
* The burned-in captions must show the caption as WRITTEN, and the spoken word
|
|
855
|
+
* list does not always match it one for one — the text handed to the voice is
|
|
856
|
+
* the spoken form ("vidlet dot app" for a domain), and a number reads as
|
|
857
|
+
* several words. So the spoken words are used only as a clock. When the counts
|
|
858
|
+
* line up, each caption word takes its spoken counterpart's timing. When they
|
|
859
|
+
* don't, the caption words are spread evenly across the actual speech
|
|
860
|
+
* envelope — still better than a blind estimate, because the envelope excludes
|
|
861
|
+
* the silence the voice leaves at either end of a clip.
|
|
862
|
+
*/
|
|
863
|
+
declare function alignWordsToCaption(caption: string, spoken: SpokenWord[]): WordTiming[];
|
|
864
|
+
/**
|
|
865
|
+
* Real word timings for a clip, from the boundaries the voice reported while
|
|
866
|
+
* speaking it. Null when the engine gave none, and callers fall back to
|
|
867
|
+
* estimateWordTimings. Timings are ms relative to the start of the clip.
|
|
868
|
+
*
|
|
869
|
+
* The engine times words against the audio it produced, but the clip on disk
|
|
870
|
+
* has since had its leading silence trimmed, so every offset is early by
|
|
871
|
+
* however much came off the front. The trim gate is set just under the noise
|
|
872
|
+
* edge-tts pads with, which puts the cut where the first word begins — so the
|
|
873
|
+
* first word's own offset IS what was removed, and rebasing on it re-anchors
|
|
874
|
+
* the whole clip without measuring the file again.
|
|
875
|
+
*/
|
|
876
|
+
declare function spokenWordTimings(caption: string, spoken: SpokenWord[] | null): WordTiming[] | null;
|
|
361
877
|
/**
|
|
362
878
|
* Align voiceover to actual recording durations
|
|
363
879
|
* If any step's recording took longer than its voiceover, pad with silence
|
|
@@ -377,6 +893,15 @@ declare function alignVoiceoverToRecording(stepAudio: Map<number, StepAudio>, ac
|
|
|
377
893
|
declare function buildWatermarkFilters(opts: {
|
|
378
894
|
tier: UserTier;
|
|
379
895
|
videoDuration: number;
|
|
896
|
+
/**
|
|
897
|
+
* The video ends on a credits card that already names QuickPeek.
|
|
898
|
+
*
|
|
899
|
+
* The end text is drawn along the bottom of the last three seconds, which
|
|
900
|
+
* is exactly where that card carries the offer - two overlays stacked on
|
|
901
|
+
* the same strip, saying the same thing twice. The corner mark stays: it is
|
|
902
|
+
* the part that marks the whole video, not just its ending.
|
|
903
|
+
*/
|
|
904
|
+
endsOnCreditsCard?: boolean;
|
|
380
905
|
}): string;
|
|
381
906
|
|
|
382
|
-
export { type ApplyTransitionsOptions, type AudioResult, type BlendTransition, CURSOR_CSS, CURSOR_PNG, CURSOR_SCRIPT, type ComposeOptions, type ConcatVideoOptions, Config, type LenientParse, type SpeedUpOptions, type SpliceVideosOptions, type StepAudio, type StepSync, type StepTransition, type TTSStep, type TimeSegment, UserTier, type VideoOverlay, type VoiceoverMeta, type WordTiming, alignVoiceoverToRecording, applyTransitions, buildConcatArgs, buildKeepSegments, buildStepTransitions, buildWatermarkFilters, canReuseVoiceover, composeVideo, concatMedia, concatVideoWithOutro, estimateWordTimings, extractJsonCandidates, generateASSHeader, generateSilence, generateVoiceover, getAudioCodec, getMediaDuration, getUpgradeUrl, isHonestWav, loadExistingVoiceover, loadVoiceoverMeta, mapTimestampsAfterCompression, parseAIJson, registerFreeUser, repairUnicodeEscapes, resolveVoice, salvageTruncatedJson, shouldWatermark, speedUpVideo, spliceVideos, synthesizeSpeech };
|
|
907
|
+
export { type ApplyTransitionsOptions, type AudioResult, type BlendTransition, CURSOR_CSS, CURSOR_PNG, CURSOR_SCRIPT, type CaptionCue, CaptionPosition, type CaptionRenderContext, CaptionStyle, type ComposeOptions, type ConcatVideoOptions, Config, EMPTY_SPAN_SPEED, GIF_DEFAULTS, type GifOptions, type LenientParse, MUSIC_BED_LUFS, NARRATION_LEAD_IN_MS, type SilenceSpan, type SpeedUpOptions, type SpliceVideosOptions, type SpokenWord, type StepAudio, type StepSync, type StepTransition, TRIM_EDGE_SILENCE, type TTSStep, type TimeSegment, UserTier, VOICE_LEAD_MS, type VideoOverlay, VoiceRate, type VoiceoverMeta, type WordTiming, alignVoiceoverToRecording, alignWordsToCaption, applyTransitions, atempoChain, buildCompressionFilter, buildConcatArgs, buildGifArgs, buildKeepSegments, buildPaletteArgs, buildRetimeFilter, buildStepTransitions, buildWatermarkFilters, canReuseVoiceover, captionCanvas, composeVideo, compressSilentSpans, concatMedia, concatVideoWithFullFrameOutro, concatVideoWithOutro, convertToGif, coverFilterChain, cutSpansFromVideo, detectBlackSpans, detectFrozenSpans, detectSilences, detectWhiteSpans, estimateWordTimings, extractJsonCandidates, generateASSHeader, generateSilence, generateVoiceover, getAudioCodec, getMediaDuration, getUpgradeUrl, getVideoDimensions, getVideoStreamDuration, hasAudioStream, intersectSpans, isHonestWav, loadExistingVoiceover, loadVoiceoverMeta, mapTimestampsAfterCompression, mergeSpans, parseAIJson, parseWordBoundaries, positionToAlignment, registerFreeUser, renderCaptionsAss, repairUnicodeEscapes, resetVoiceEngine, resolveCaptionFont, resolveCaptions, resolveRate, resolveVoice, retimeSpansInVideo, salvageTruncatedJson, shouldWatermark, speakSymbols, speedUpMiddle, speedUpVideo, spliceVideos, spokenWordTimings, synthesizeSpeech, trimEdgeSilence };
|