@ossclip/core 0.1.23 → 0.1.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/fonts/NotoNastaliqUrdu-Bold.ttf +0 -0
- package/assets/fonts/OFL.txt +93 -0
- package/assets/fonts/README.md +15 -0
- package/package.json +2 -1
- package/src/blooper.ts +91 -8
- package/src/browser.ts +8 -0
- package/src/captions.ts +45 -2
- package/src/concat.ts +5 -5
- package/src/config.ts +110 -0
- package/src/content-rect-detect.ts +16 -4
- package/src/content-rect.ts +211 -0
- package/src/cover.ts +21 -5
- package/src/cutlist.ts +38 -6
- package/src/dictionary.ts +56 -0
- package/src/export-premiere-project.ts +26 -9
- package/src/fonts.ts +17 -0
- package/src/index.ts +3 -0
- package/src/ingest.ts +88 -2
- package/src/normalize.ts +273 -127
- package/src/producer/index.ts +1 -0
- package/src/producer/repair.ts +36 -13
- package/src/producer/youtube.ts +434 -0
- package/src/retake.ts +104 -2
- package/src/scene-schema.ts +10 -0
- package/src/thumbnail.ts +412 -0
- package/src/transcribe.ts +22 -0
- package/src/zoom.ts +63 -12
package/src/thumbnail.ts
ADDED
|
@@ -0,0 +1,412 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import { z } from "zod/v4";
|
|
3
|
+
import type { LlmProvider } from "./producer/provider";
|
|
4
|
+
import { cappedText } from "./producer/beats";
|
|
5
|
+
import { coverHeadline } from "./cover";
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* The `--youtube` AI thumbnail (Y3, 2026-08-16): a Gemini-generated 16:9
|
|
9
|
+
* image built from the creator's portrait photo plus an LLM-written concept,
|
|
10
|
+
* written beside the video as `<out>.thumbnail.png`. Strictly additive — the
|
|
11
|
+
* frame-grab cover pipeline is untouched, and every failure here degrades to
|
|
12
|
+
* "the cover stands".
|
|
13
|
+
*
|
|
14
|
+
* Everything in this file except `generateThumbnailImage` is pure: the
|
|
15
|
+
* skip/generate decision, both prompt builders and the response-byte
|
|
16
|
+
* extraction are all testable without a network, an API key, or the SDK.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* User-specified slug (2026-08-16); config `thumbnailModel` overrides. The
|
|
21
|
+
* slug is taken on faith — an API rejection surfaces VERBATIM and is never
|
|
22
|
+
* retried, the isNonRetryableAgyFailure posture (FINDINGS §132): a bad model
|
|
23
|
+
* name is deterministic, so a retry loop only burns quota restating it.
|
|
24
|
+
*/
|
|
25
|
+
export const THUMBNAIL_MODEL_DEFAULT = "gemini-3.1-flash-lite-image";
|
|
26
|
+
|
|
27
|
+
export type ThumbnailDecision =
|
|
28
|
+
| "generate"
|
|
29
|
+
| "skip-no-youtube"
|
|
30
|
+
| "skip-no-portrait"
|
|
31
|
+
| "skip-no-key"
|
|
32
|
+
| "skip-portrait-missing";
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Whether a run generates an AI thumbnail, as one pure function so the whole
|
|
36
|
+
* matrix is a table test. Ordered by how early the user could have known:
|
|
37
|
+
* the feature is off (no --youtube), never configured (no portrait path),
|
|
38
|
+
* unauthenticated (no GEMINI_API_KEY — env-only, secrets never live in
|
|
39
|
+
* config.json, env.ts rule), and only then the runtime surprise (a portrait
|
|
40
|
+
* path that points at nothing). The graceful-fallback contract is the user
|
|
41
|
+
* decision of 2026-08-16: portrait/key missing → the frame-grab cover stands.
|
|
42
|
+
*/
|
|
43
|
+
export function thumbnailDecision(
|
|
44
|
+
youtube: boolean,
|
|
45
|
+
portraitPath: string | undefined,
|
|
46
|
+
hasKey: boolean,
|
|
47
|
+
portraitExists: boolean,
|
|
48
|
+
): ThumbnailDecision {
|
|
49
|
+
if (!youtube) return "skip-no-youtube";
|
|
50
|
+
if (!portraitPath) return "skip-no-portrait";
|
|
51
|
+
if (!hasKey) return "skip-no-key";
|
|
52
|
+
if (!portraitExists) return "skip-portrait-missing";
|
|
53
|
+
return "generate";
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Portrait formats the Gemini API accepts as `inlineData`, keyed by lowercase
|
|
58
|
+
* extension. A map, not a sniff: the file was pointed at by the user
|
|
59
|
+
* (`--portrait` / config `portrait`), and an extension outside this table is
|
|
60
|
+
* a loud skip at the call site rather than a guessed mime type the API
|
|
61
|
+
* rejects with a worse message.
|
|
62
|
+
*/
|
|
63
|
+
export const PORTRAIT_MIME_TYPES: Readonly<Record<string, string>> = {
|
|
64
|
+
png: "image/png",
|
|
65
|
+
jpg: "image/jpeg",
|
|
66
|
+
jpeg: "image/jpeg",
|
|
67
|
+
webp: "image/webp",
|
|
68
|
+
};
|
|
69
|
+
|
|
70
|
+
/** `photo.PNG` → `image/png`; anything outside the table → undefined. */
|
|
71
|
+
export function portraitMimeType(path: string): string | undefined {
|
|
72
|
+
const ext = /\.([^./\\]+)$/.exec(path)?.[1]?.toLowerCase();
|
|
73
|
+
return ext ? PORTRAIT_MIME_TYPES[ext] : undefined;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* The LLM-written concept the image prompt is built from. `cappedText`, not
|
|
78
|
+
* `.max()` — §112 as applied in beats.ts: LLM output is untrusted input, and
|
|
79
|
+
* a concept one word over budget must cost a word, never the thumbnail.
|
|
80
|
+
* `overlayText` is ADDITIONALLY passed through `coverHeadline` by the caller
|
|
81
|
+
* to cap WORDS — the schema caps characters, but overlay text at thumbnail
|
|
82
|
+
* size has the same 4-9 word ceiling a cover banner does (§35).
|
|
83
|
+
*/
|
|
84
|
+
export const ThumbnailConceptSchema = z.object({
|
|
85
|
+
/** One vivid scene — what the image shows. */
|
|
86
|
+
scene: cappedText(300),
|
|
87
|
+
/** The 3-6 word text rendered ON the image, verbatim. */
|
|
88
|
+
overlayText: cappedText(60),
|
|
89
|
+
/** Palette, lighting, mood — craft direction for the image model. */
|
|
90
|
+
styleNotes: cappedText(300),
|
|
91
|
+
});
|
|
92
|
+
export type ThumbnailConcept = z.infer<typeof ThumbnailConceptSchema>;
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* The §35 WORD cap on overlay text, as the ONE helper every concept-accepting
|
|
96
|
+
* path calls (2026-08-17, editor thumbnail panel): thumbnailStep, the
|
|
97
|
+
* pre-render approval, the interactive edit parser and the editor's
|
|
98
|
+
* regenerate endpoint must all produce byte-identical text — the overlay
|
|
99
|
+
* feeds `thumbnailImageCacheName`, so two spellings of this cap would mint
|
|
100
|
+
* two cache keys for one concept. `|| raw` keeps the schema-capped text when
|
|
101
|
+
* coverHeadline rejects the whole line (e.g. all-stopword): an overlay the
|
|
102
|
+
* user typed must never silently become an EMPTY overlay.
|
|
103
|
+
*/
|
|
104
|
+
export function approvedOverlayText(raw: string): string {
|
|
105
|
+
return coverHeadline(raw) || raw;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* The workdir image cache's filename: model + the concept actually prompted +
|
|
110
|
+
* the portrait's CONTENT (sha1 of its bytes, not its path or mtime — a
|
|
111
|
+
* swapped portrait at the same path must regenerate). Extracted from
|
|
112
|
+
* thumbnailStep (2026-08-17) because the editor's regenerate endpoint writes
|
|
113
|
+
* the same cache, and a second spelling of the key would let the two callers
|
|
114
|
+
* cache past each other. The concept is serialized with the field order
|
|
115
|
+
* PINNED here rather than trusting the caller's object: both callers build
|
|
116
|
+
* concepts through `ThumbnailConceptSchema`, whose parse order this matches,
|
|
117
|
+
* so existing caches stay valid — and a future caller with a hand-built
|
|
118
|
+
* object cannot silently re-key everything by ordering its literal
|
|
119
|
+
* differently.
|
|
120
|
+
*/
|
|
121
|
+
export function thumbnailImageCacheName(
|
|
122
|
+
model: string,
|
|
123
|
+
concept: ThumbnailConcept,
|
|
124
|
+
portraitSha1: string,
|
|
125
|
+
): string {
|
|
126
|
+
const key = createHash("sha1")
|
|
127
|
+
.update(
|
|
128
|
+
JSON.stringify([
|
|
129
|
+
model,
|
|
130
|
+
{
|
|
131
|
+
scene: concept.scene,
|
|
132
|
+
overlayText: concept.overlayText,
|
|
133
|
+
styleNotes: concept.styleNotes,
|
|
134
|
+
},
|
|
135
|
+
portraitSha1,
|
|
136
|
+
]),
|
|
137
|
+
)
|
|
138
|
+
.digest("hex")
|
|
139
|
+
.slice(0, 8);
|
|
140
|
+
return `thumbnail-${key}.png`;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* The workdir file the pre-render approval step writes (thumbnail UX,
|
|
145
|
+
* 2026-08-16): once it exists, `thumbnailStep` uses it VERBATIM and never
|
|
146
|
+
* asks a model for a concept again — the user approved (and possibly edited)
|
|
147
|
+
* this exact text, and a fresh concept call would discard their edit.
|
|
148
|
+
*/
|
|
149
|
+
export const THUMBNAIL_APPROVED_BASENAME = "thumbnail-concept-approved.json";
|
|
150
|
+
|
|
151
|
+
/**
|
|
152
|
+
* What the approved file holds: the approved concept, or an explicit
|
|
153
|
+
* `{skip: true}` — the user answered "skip thumbnail" at the approval prompt,
|
|
154
|
+
* and that decision must survive into the (non-interactive) replay exactly
|
|
155
|
+
* like an approval does, as a LOUD skip rather than a silent regeneration.
|
|
156
|
+
* Skip variant FIRST in the union: a concept object can never carry `skip`
|
|
157
|
+
* and a skip object can never carry the concept's required fields, so order
|
|
158
|
+
* only matters for error messages — but skip-first means a hand-added `skip`
|
|
159
|
+
* key wins over a leftover concept body instead of being ignored.
|
|
160
|
+
*/
|
|
161
|
+
export const ThumbnailConceptApprovedSchema = z.union([
|
|
162
|
+
z.object({ skip: z.literal(true) }),
|
|
163
|
+
ThumbnailConceptSchema,
|
|
164
|
+
]);
|
|
165
|
+
export type ThumbnailConceptApproved = z.infer<typeof ThumbnailConceptApprovedSchema>;
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* How much transcript the concept prompt carries. Half the youtube.ts cap: a
|
|
169
|
+
* thumbnail concept is about the video's ONE claim, not its coverage, and
|
|
170
|
+
* the hook/intent lines already carry the strongest steer.
|
|
171
|
+
*/
|
|
172
|
+
export const THUMBNAIL_TRANSCRIPT_CHAR_CAP = 4000;
|
|
173
|
+
|
|
174
|
+
export interface ThumbnailConceptPromptArgs {
|
|
175
|
+
/** The producer's hook, when a beat sheet exists — the strongest claim. */
|
|
176
|
+
hook?: string;
|
|
177
|
+
/** `--intent`, when the run had one. */
|
|
178
|
+
intent?: string;
|
|
179
|
+
/** The repaired transcript's plain text — what the viewer actually hears. */
|
|
180
|
+
transcriptText: string;
|
|
181
|
+
/** Who watches the channel (`--audience` / config) — steers the angle. */
|
|
182
|
+
audience?: string;
|
|
183
|
+
/**
|
|
184
|
+
* The durable thumbnail steer (`--thumbnail-brief` / config). Marked
|
|
185
|
+
* must-honor in the prompt: this is the user's standing instruction, not a
|
|
186
|
+
* suggestion the model may trade away against its own craft rules.
|
|
187
|
+
*/
|
|
188
|
+
brief?: string;
|
|
189
|
+
/**
|
|
190
|
+
* The youtube pack's FIRST title, when the pack already exists — thumbnail
|
|
191
|
+
* and title must tell one story, and the first title is the pack's lead
|
|
192
|
+
* angle. The pre-render approval step runs BEFORE the pack is generated
|
|
193
|
+
* (the pack writes after render), so it passes the hook instead; only
|
|
194
|
+
* thumbnailStep's own post-pack concept call can supply this.
|
|
195
|
+
*/
|
|
196
|
+
titleAngle?: string;
|
|
197
|
+
/**
|
|
198
|
+
* A per-call creator note ("regenerate with a note" at the approval
|
|
199
|
+
* prompt). Must-honor like the brief, but transient — it lives in this one
|
|
200
|
+
* call and is never persisted to config or the concept cache.
|
|
201
|
+
*/
|
|
202
|
+
note?: string;
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/**
|
|
206
|
+
* Pure prompt builder for the concept call, separated from the provider so
|
|
207
|
+
* the include/omit matrix (hook, intent, audience, brief, titleAngle, note)
|
|
208
|
+
* and the transcript cap are testable without an LLM.
|
|
209
|
+
*/
|
|
210
|
+
export function buildThumbnailConceptPrompt(args: ThumbnailConceptPromptArgs): {
|
|
211
|
+
system: string;
|
|
212
|
+
user: string;
|
|
213
|
+
} {
|
|
214
|
+
const system =
|
|
215
|
+
"You are a YouTube thumbnail strategist designing a high-CTR thumbnail concept for a " +
|
|
216
|
+
"finished video.\n" +
|
|
217
|
+
"- scene: ONE vivid scene — a single concrete image, not a collage of ideas.\n" +
|
|
218
|
+
"- overlayText: 3-6 punchy words rendered on the image. A claim or a tension, never a " +
|
|
219
|
+
"full sentence, and never a clickbait claim the video does not deliver.\n" +
|
|
220
|
+
"- styleNotes: palette, lighting and mood direction for the image model — specific " +
|
|
221
|
+
"enough to constrain it, short enough to not fight the scene.\n" +
|
|
222
|
+
// Pose conflict fix (debugged 2026-08-16): the reference photo is frontal
|
|
223
|
+
// with arms crossed, but a concept saying "holding his head in
|
|
224
|
+
// frustration" made the image model follow the scene's choreography over
|
|
225
|
+
// the image prompt's keep-pose rule. The fix starts HERE — a concept
|
|
226
|
+
// that never choreographs the person cannot lose that fight downstream.
|
|
227
|
+
"- The creator appears from a fixed reference photo whose pose CANNOT change " +
|
|
228
|
+
"(frontal, natural). Design the scene AROUND the person — lighting, props, screen " +
|
|
229
|
+
"content, composition — never choreograph the person's body or hands.";
|
|
230
|
+
const capped =
|
|
231
|
+
args.transcriptText.length > THUMBNAIL_TRANSCRIPT_CHAR_CAP
|
|
232
|
+
? // Slice + say so (youtube.ts posture): the model must know it is
|
|
233
|
+
// reading an excerpt, or it will anchor the concept on the first half.
|
|
234
|
+
`${args.transcriptText.slice(0, THUMBNAIL_TRANSCRIPT_CHAR_CAP)}\n[transcript truncated — the video continues]`
|
|
235
|
+
: args.transcriptText;
|
|
236
|
+
const user =
|
|
237
|
+
(args.intent ? `Intent: ${args.intent}\n` : "") +
|
|
238
|
+
(args.hook ? `Hook (already chosen by the producer): ${args.hook}\n` : "") +
|
|
239
|
+
// One story across the upload: when the pack's lead title exists, the
|
|
240
|
+
// thumbnail must be its visual restatement, not a second pitch.
|
|
241
|
+
(args.titleAngle
|
|
242
|
+
? `Video title (already chosen — the thumbnail must tell the same story): ${args.titleAngle}\n`
|
|
243
|
+
: "") +
|
|
244
|
+
(args.audience ? `Audience: ${args.audience}\n` : "") +
|
|
245
|
+
(args.brief ? `Creator brief (must be honored): ${args.brief}\n` : "") +
|
|
246
|
+
(args.note ? `Creator note (must be honored): ${args.note}\n` : "") +
|
|
247
|
+
`\nTranscript:\n${capped}`;
|
|
248
|
+
return { system, user };
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
/** One editorial call → a validated concept. */
|
|
252
|
+
export async function generateThumbnailConcept(
|
|
253
|
+
provider: LlmProvider,
|
|
254
|
+
args: ThumbnailConceptPromptArgs,
|
|
255
|
+
): Promise<ThumbnailConcept> {
|
|
256
|
+
const { system, user } = buildThumbnailConceptPrompt(args);
|
|
257
|
+
return provider.complete({
|
|
258
|
+
system,
|
|
259
|
+
user,
|
|
260
|
+
schema: ThumbnailConceptSchema,
|
|
261
|
+
schemaName: "thumbnail_concept",
|
|
262
|
+
tier: "editorial",
|
|
263
|
+
});
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
/**
|
|
267
|
+
* The image-model prompt, with the craft rules baked in as fixed text rather
|
|
268
|
+
* than left to the concept call: the concept model picks WHAT to show, this
|
|
269
|
+
* function owns HOW a YouTube thumbnail is built (16:9, subject in a third,
|
|
270
|
+
* overlay verbatim, legible at grid size). Pure — the hasPortrait branch is
|
|
271
|
+
* a table test.
|
|
272
|
+
*
|
|
273
|
+
* `revisionNote` is the post-generation retry loop's one input ("regenerate
|
|
274
|
+
* with a note"): the concept is UNCHANGED, the note rides the image prompt
|
|
275
|
+
* only, marked must-honor and appended last so it reads as the final word.
|
|
276
|
+
*/
|
|
277
|
+
export function buildThumbnailPrompt(
|
|
278
|
+
concept: ThumbnailConcept,
|
|
279
|
+
hasPortrait: boolean,
|
|
280
|
+
revisionNote?: string,
|
|
281
|
+
): string {
|
|
282
|
+
const lines = [
|
|
283
|
+
"Create a YouTube thumbnail image.",
|
|
284
|
+
"",
|
|
285
|
+
...(hasPortrait
|
|
286
|
+
? [
|
|
287
|
+
// Identity FIRST, before Scene/Style (pose incident, debugged
|
|
288
|
+
// 2026-08-16): with the identity block buried under the scene, a
|
|
289
|
+
// concept that choreographed the person ("holding his head in
|
|
290
|
+
// frustration") beat the keep-pose rule — the image model weighs
|
|
291
|
+
// what it reads first. Identity is the HARD constraint, everything
|
|
292
|
+
// else is negotiable (user directive 2026-08-16: the first
|
|
293
|
+
// generation produced a face that "isn't me" — a thumbnail with
|
|
294
|
+
// someone else's face is worse than no thumbnail). The reference
|
|
295
|
+
// photo is ground truth: the model may relight and
|
|
296
|
+
// recontextualize, never redraw the person.
|
|
297
|
+
"Identity requirements — these override everything below:",
|
|
298
|
+
"- The person in the reference photo MUST appear with their face reproduced at " +
|
|
299
|
+
"100% accuracy — this is the single most important requirement. Treat the " +
|
|
300
|
+
"reference photo as ground truth for identity: exact same facial structure, " +
|
|
301
|
+
"eyes, nose, mouth, skin tone, facial hair, hairline and glasses. Do not " +
|
|
302
|
+
"idealize, de-age, slim, or otherwise alter any facial feature.",
|
|
303
|
+
"- Keep the head pose, angle and framing of the face as close to the reference " +
|
|
304
|
+
"photo as possible. Only the lighting and color grade may adapt to match the " +
|
|
305
|
+
"scene; the face itself must stay photorealistic and identical to the " +
|
|
306
|
+
"reference, never stylized or repainted.",
|
|
307
|
+
"- Place that person prominently in the left or right third of the frame, " +
|
|
308
|
+
"integrated naturally into the scene (matching light and color, no cut-out " +
|
|
309
|
+
"look). If any requirement conflicts with facial accuracy, facial accuracy " +
|
|
310
|
+
"wins.",
|
|
311
|
+
"- The scene adapts to the person; never re-pose, re-angle, or choreograph the " +
|
|
312
|
+
"person to fit the scene.",
|
|
313
|
+
"",
|
|
314
|
+
]
|
|
315
|
+
: []),
|
|
316
|
+
`Scene: ${concept.scene}`,
|
|
317
|
+
`Style: ${concept.styleNotes}`,
|
|
318
|
+
"",
|
|
319
|
+
"Requirements:",
|
|
320
|
+
"- 16:9 landscape, at least 1280x720.",
|
|
321
|
+
// Quotes on their own line: overlay text is the one part of the image
|
|
322
|
+
// with a right answer, and image models paraphrase anything stated
|
|
323
|
+
// loosely — but the instruction word itself must not touch the quoted
|
|
324
|
+
// string. The first field run (2026-08-16) used `reading EXACTLY: "…"`
|
|
325
|
+
// and the model painted "EXACTLY:" into the thumbnail as a headline.
|
|
326
|
+
"- Render bold, high-contrast overlay text. The overlay must contain this",
|
|
327
|
+
" text and nothing else:",
|
|
328
|
+
` ${concept.overlayText}`,
|
|
329
|
+
"- Vivid and eye-catching but not cluttered — one focal point.",
|
|
330
|
+
"- No watermarks, no logos, no borders.",
|
|
331
|
+
"- Every element must stay legible when the image is displayed 320px wide.",
|
|
332
|
+
...(revisionNote
|
|
333
|
+
? ["", `Revision note from the creator (must be honored): ${revisionNote}`]
|
|
334
|
+
: []),
|
|
335
|
+
];
|
|
336
|
+
return lines.join("\n");
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
/**
|
|
340
|
+
* Pull the first image part's bytes out of a generateContent response.
|
|
341
|
+
*
|
|
342
|
+
* Isolated in ONE small function on purpose: the @google/genai response
|
|
343
|
+
* shape is the SDK's to drift (plan risk note, 2026-08-16), and when it
|
|
344
|
+
* does, this is the only code that knows about `candidates[].content.parts`.
|
|
345
|
+
* Treats the response as `unknown` so the pure tests exercise it with plain
|
|
346
|
+
* objects and never import the SDK.
|
|
347
|
+
*/
|
|
348
|
+
export function extractImageBytes(response: unknown): Uint8Array {
|
|
349
|
+
const candidates = (response as { candidates?: unknown })?.candidates;
|
|
350
|
+
if (Array.isArray(candidates)) {
|
|
351
|
+
for (const candidate of candidates) {
|
|
352
|
+
const parts = (candidate as { content?: { parts?: unknown } })?.content?.parts;
|
|
353
|
+
if (!Array.isArray(parts)) continue;
|
|
354
|
+
for (const part of parts) {
|
|
355
|
+
const data = (part as { inlineData?: { data?: unknown } })?.inlineData?.data;
|
|
356
|
+
if (typeof data === "string" && data.length > 0) {
|
|
357
|
+
return new Uint8Array(Buffer.from(data, "base64"));
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
}
|
|
361
|
+
}
|
|
362
|
+
throw new Error(
|
|
363
|
+
"the model returned no image — the response had no inlineData part " +
|
|
364
|
+
"(model refusal or a text-only reply)",
|
|
365
|
+
);
|
|
366
|
+
}
|
|
367
|
+
|
|
368
|
+
export interface GenerateThumbnailImageOptions {
|
|
369
|
+
apiKey: string;
|
|
370
|
+
model: string;
|
|
371
|
+
prompt: string;
|
|
372
|
+
/** The creator's portrait as base64 `inlineData`, when the run has one. */
|
|
373
|
+
portrait?: { data: string; mimeType: string };
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
/**
|
|
377
|
+
* The ONE I/O function of this module: call Gemini image generation and
|
|
378
|
+
* return the image bytes.
|
|
379
|
+
*
|
|
380
|
+
* The SDK import is LAZY and lives here, nowhere else: core is near-zero-dep
|
|
381
|
+
* by design (only @anthropic-ai/sdk + zod), and every run that is not a
|
|
382
|
+
* `--youtube`-with-portrait-and-key run must never pay for loading
|
|
383
|
+
* @google/genai — a static import would tax every produce for a feature most
|
|
384
|
+
* runs never reach.
|
|
385
|
+
*
|
|
386
|
+
* Errors are NOT retried and surface the API's message verbatim — the
|
|
387
|
+
* isNonRetryableAgyFailure posture (§132): the model slug is user-specified,
|
|
388
|
+
* a rejection of it is deterministic, and a retry loop would only restate it.
|
|
389
|
+
*/
|
|
390
|
+
export async function generateThumbnailImage(
|
|
391
|
+
opts: GenerateThumbnailImageOptions,
|
|
392
|
+
): Promise<Uint8Array> {
|
|
393
|
+
const { GoogleGenAI } = await import("@google/genai");
|
|
394
|
+
const ai = new GoogleGenAI({ apiKey: opts.apiKey });
|
|
395
|
+
const response = await ai.models.generateContent({
|
|
396
|
+
model: opts.model,
|
|
397
|
+
contents: [
|
|
398
|
+
{
|
|
399
|
+
role: "user",
|
|
400
|
+
parts: [
|
|
401
|
+
// Portrait FIRST, prompt second — the prompt refers back to "the
|
|
402
|
+
// reference photo", so the photo must already be on the table.
|
|
403
|
+
...(opts.portrait
|
|
404
|
+
? [{ inlineData: { data: opts.portrait.data, mimeType: opts.portrait.mimeType } }]
|
|
405
|
+
: []),
|
|
406
|
+
{ text: opts.prompt },
|
|
407
|
+
],
|
|
408
|
+
},
|
|
409
|
+
],
|
|
410
|
+
});
|
|
411
|
+
return extractImageBytes(response);
|
|
412
|
+
}
|
package/src/transcribe.ts
CHANGED
|
@@ -169,6 +169,15 @@ export interface WhisperOptions {
|
|
|
169
169
|
* stay byte-identical to what English-suffixed models always got.
|
|
170
170
|
*/
|
|
171
171
|
language?: string;
|
|
172
|
+
/**
|
|
173
|
+
* Initial decoder prompt (`--prompt`), used to bias recognition toward the
|
|
174
|
+
* user's vocabulary (F4 dictionary, 2026-08-16: "Jason" for JSON). Left
|
|
175
|
+
* unset, the spawned args stay byte-identical to every pre-dictionary run.
|
|
176
|
+
* Known risk (documented in the flag's help): a whisper-cli built before
|
|
177
|
+
* the flag existed rejects it with its own loud error — accepted over
|
|
178
|
+
* silently dropping the user's terms.
|
|
179
|
+
*/
|
|
180
|
+
prompt?: string;
|
|
172
181
|
}
|
|
173
182
|
|
|
174
183
|
/**
|
|
@@ -186,9 +195,22 @@ export function whisperArgs(opts: WhisperOptions, wavPath: string): string[] {
|
|
|
186
195
|
"--no-prints",
|
|
187
196
|
];
|
|
188
197
|
if (opts.language !== undefined) args.push("-l", opts.language);
|
|
198
|
+
if (opts.prompt !== undefined) args.push("--prompt", opts.prompt);
|
|
189
199
|
return args;
|
|
190
200
|
}
|
|
191
201
|
|
|
202
|
+
/**
|
|
203
|
+
* The `--prompt` text for a user dictionary. whisper.cpp treats the prompt as
|
|
204
|
+
* preceding context, so a plain vocabulary list is enough to bias the decoder
|
|
205
|
+
* toward these spellings ("Jason" → "JSON", 2026-08-16 field report). Pure
|
|
206
|
+
* and undefined-for-empty so the no-dictionary invocation stays byte-identical
|
|
207
|
+
* to what every run before the flag got.
|
|
208
|
+
*/
|
|
209
|
+
export function whisperPromptFor(dictionary: readonly string[]): string | undefined {
|
|
210
|
+
if (dictionary.length === 0) return undefined;
|
|
211
|
+
return `Vocabulary: ${dictionary.join(", ")}.`;
|
|
212
|
+
}
|
|
213
|
+
|
|
192
214
|
export async function runWhisper(opts: WhisperOptions, wavPath: string): Promise<Transcript> {
|
|
193
215
|
await run(opts.whisperPath, whisperArgs(opts, wavPath));
|
|
194
216
|
// Bytes, not "utf8": the utf8 read is where the §130 split characters died.
|
package/src/zoom.ts
CHANGED
|
@@ -48,6 +48,13 @@ export interface ZoomPlan {
|
|
|
48
48
|
segments: ZoomSegment[];
|
|
49
49
|
/** How many cut-free clips the plan covers — logged, never inferred. */
|
|
50
50
|
clips: number;
|
|
51
|
+
/**
|
|
52
|
+
* The allowedClips split, carried on the plan so the CLI log reports the
|
|
53
|
+
* counts the plan actually acted on instead of re-deriving them (and
|
|
54
|
+
* possibly disagreeing after the boundary cleaning).
|
|
55
|
+
*/
|
|
56
|
+
zoomedClips: number;
|
|
57
|
+
staticClips: number;
|
|
51
58
|
/** The ramp actually used, so the log can't drift from the behaviour. */
|
|
52
59
|
rampSec: number;
|
|
53
60
|
}
|
|
@@ -63,6 +70,18 @@ export interface ZoomPlanOptions {
|
|
|
63
70
|
clipStarts?: readonly number[];
|
|
64
71
|
/** Seconds the push takes to arrive before it holds. */
|
|
65
72
|
rampSec?: number;
|
|
73
|
+
/**
|
|
74
|
+
* Per-clip zoom permission, PARALLEL TO `clipStarts` by index. User
|
|
75
|
+
* decision 2026-08-16 — "Face-only. If there's anything else, then no
|
|
76
|
+
* zoom": the always-on idle push visibly SLID screen-recording content,
|
|
77
|
+
* so a clip whose subject is not a face gets NO segments at all rather
|
|
78
|
+
* than a flat one. `zoomScaleAt` answers 1 outside the plan, and the
|
|
79
|
+
* premiere export's `zoomKeyframesFor` collapses a segment-free span to a
|
|
80
|
+
* plain scale-1 value, so downstream needs zero changes. Entries missing
|
|
81
|
+
* off the end of a short mask read as allowed, and an absent mask is
|
|
82
|
+
* today's plan exactly — pre-F1 callers are byte-identical.
|
|
83
|
+
*/
|
|
84
|
+
allowedClips?: readonly boolean[];
|
|
66
85
|
}
|
|
67
86
|
|
|
68
87
|
/**
|
|
@@ -83,13 +102,32 @@ export const ZOOM_MAX_SCALE = 1.05;
|
|
|
83
102
|
*/
|
|
84
103
|
export const ZOOM_RAMP_SEC = 8;
|
|
85
104
|
|
|
86
|
-
/**
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
105
|
+
/**
|
|
106
|
+
* Clip starts, cleaned — in range, unique, sorted, always including 0 — each
|
|
107
|
+
* carrying its `allowedClips` verdict. The verdict is paired with its start
|
|
108
|
+
* BY INDEX before any of the cleaning, so dedupe/sort can never shift a
|
|
109
|
+
* verdict onto a different clip; duplicated starts are one clip and OR their
|
|
110
|
+
* verdicts (any pairing that vouches "face" wins — losing the push on a face
|
|
111
|
+
* clip is the regression, holding still an extra clip is merely conservative
|
|
112
|
+
* the wrong way). The synthetic 0 boundary is allowed unless the caller's
|
|
113
|
+
* own list contains a 0 that says otherwise. Missing mask entries read as
|
|
114
|
+
* allowed (`!== false`), which is also what makes an absent mask identical
|
|
115
|
+
* to the pre-mask contract.
|
|
116
|
+
*/
|
|
117
|
+
function clipBoundaries(
|
|
118
|
+
starts: readonly number[] | undefined,
|
|
119
|
+
allowed: readonly boolean[] | undefined,
|
|
120
|
+
duration: number,
|
|
121
|
+
): Array<{ start: number; allowed: boolean }> {
|
|
122
|
+
const byStart = new Map<number, boolean>();
|
|
123
|
+
(starts ?? []).forEach((t, i) => {
|
|
124
|
+
if (!Number.isFinite(t) || t < 0 || t >= duration) return;
|
|
125
|
+
byStart.set(t, (byStart.get(t) ?? false) || allowed?.[i] !== false);
|
|
126
|
+
});
|
|
127
|
+
if (!byStart.has(0)) byStart.set(0, true);
|
|
128
|
+
return [...byStart.entries()]
|
|
129
|
+
.map(([start, ok]) => ({ start, allowed: ok }))
|
|
130
|
+
.sort((a, b) => a.start - b.start);
|
|
93
131
|
}
|
|
94
132
|
|
|
95
133
|
export function buildZoomPlan(
|
|
@@ -98,14 +136,21 @@ export function buildZoomPlan(
|
|
|
98
136
|
): ZoomPlan {
|
|
99
137
|
const maxScale = opts.maxScale ?? ZOOM_MAX_SCALE;
|
|
100
138
|
const rampSec = opts.rampSec ?? ZOOM_RAMP_SEC;
|
|
101
|
-
if (outputDurationSec <= 0)
|
|
139
|
+
if (outputDurationSec <= 0) {
|
|
140
|
+
return { segments: [], clips: 0, zoomedClips: 0, staticClips: 0, rampSec };
|
|
141
|
+
}
|
|
102
142
|
|
|
103
|
-
const starts = clipBoundaries(opts.clipStarts, outputDurationSec);
|
|
143
|
+
const starts = clipBoundaries(opts.clipStarts, opts.allowedClips, outputDurationSec);
|
|
104
144
|
const segments: ZoomSegment[] = [];
|
|
145
|
+
const staticClips = starts.filter((s) => !s.allowed).length;
|
|
105
146
|
|
|
106
147
|
for (let i = 0; i < starts.length; i++) {
|
|
107
|
-
const start = starts[i]!;
|
|
108
|
-
|
|
148
|
+
const { start, allowed } = starts[i]!;
|
|
149
|
+
// A disallowed clip emits NOTHING — not a flat segment. `zoomScaleAt`
|
|
150
|
+
// returns 1 for any instant no segment claims (zoom.ts's own "1 outside
|
|
151
|
+
// the plan" contract), so a hole in the plan IS the static camera.
|
|
152
|
+
if (!allowed) continue;
|
|
153
|
+
const end = i + 1 < starts.length ? starts[i + 1]!.start : outputDurationSec;
|
|
109
154
|
const length = end - start;
|
|
110
155
|
if (length <= 1e-9) continue;
|
|
111
156
|
|
|
@@ -120,7 +165,13 @@ export function buildZoomPlan(
|
|
|
120
165
|
}
|
|
121
166
|
}
|
|
122
167
|
|
|
123
|
-
return {
|
|
168
|
+
return {
|
|
169
|
+
segments,
|
|
170
|
+
clips: starts.length,
|
|
171
|
+
zoomedClips: starts.length - staticClips,
|
|
172
|
+
staticClips,
|
|
173
|
+
rampSec,
|
|
174
|
+
};
|
|
124
175
|
}
|
|
125
176
|
|
|
126
177
|
/**
|