@ossclip/core 0.1.24 → 0.1.26

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,412 @@
1
+ import { createHash } from "node:crypto";
2
+ import { z } from "zod/v4";
3
+ import type { LlmProvider } from "./producer/provider";
4
+ import { cappedText } from "./producer/beats";
5
+ import { coverHeadline } from "./cover";
6
+
7
+ /**
8
+ * The `--youtube` AI thumbnail (Y3, 2026-08-16): a Gemini-generated 16:9
9
+ * image built from the creator's portrait photo plus an LLM-written concept,
10
+ * written beside the video as `<out>.thumbnail.png`. Strictly additive — the
11
+ * frame-grab cover pipeline is untouched, and every failure here degrades to
12
+ * "the cover stands".
13
+ *
14
+ * Everything in this file except `generateThumbnailImage` is pure: the
15
+ * skip/generate decision, both prompt builders and the response-byte
16
+ * extraction are all testable without a network, an API key, or the SDK.
17
+ */
18
+
19
+ /**
20
+ * User-specified slug (2026-08-16); config `thumbnailModel` overrides. The
21
+ * slug is taken on faith — an API rejection surfaces VERBATIM and is never
22
+ * retried, the isNonRetryableAgyFailure posture (FINDINGS §132): a bad model
23
+ * name is deterministic, so a retry loop only burns quota restating it.
24
+ */
25
+ export const THUMBNAIL_MODEL_DEFAULT = "gemini-3.1-flash-lite-image";
26
+
27
+ export type ThumbnailDecision =
28
+ | "generate"
29
+ | "skip-no-youtube"
30
+ | "skip-no-portrait"
31
+ | "skip-no-key"
32
+ | "skip-portrait-missing";
33
+
34
+ /**
35
+ * Whether a run generates an AI thumbnail, as one pure function so the whole
36
+ * matrix is a table test. Ordered by how early the user could have known:
37
+ * the feature is off (no --youtube), never configured (no portrait path),
38
+ * unauthenticated (no GEMINI_API_KEY — env-only, secrets never live in
39
+ * config.json, env.ts rule), and only then the runtime surprise (a portrait
40
+ * path that points at nothing). The graceful-fallback contract is the user
41
+ * decision of 2026-08-16: portrait/key missing → the frame-grab cover stands.
42
+ */
43
+ export function thumbnailDecision(
44
+ youtube: boolean,
45
+ portraitPath: string | undefined,
46
+ hasKey: boolean,
47
+ portraitExists: boolean,
48
+ ): ThumbnailDecision {
49
+ if (!youtube) return "skip-no-youtube";
50
+ if (!portraitPath) return "skip-no-portrait";
51
+ if (!hasKey) return "skip-no-key";
52
+ if (!portraitExists) return "skip-portrait-missing";
53
+ return "generate";
54
+ }
55
+
56
+ /**
57
+ * Portrait formats the Gemini API accepts as `inlineData`, keyed by lowercase
58
+ * extension. A map, not a sniff: the file was pointed at by the user
59
+ * (`--portrait` / config `portrait`), and an extension outside this table is
60
+ * a loud skip at the call site rather than a guessed mime type the API
61
+ * rejects with a worse message.
62
+ */
63
+ export const PORTRAIT_MIME_TYPES: Readonly<Record<string, string>> = {
64
+ png: "image/png",
65
+ jpg: "image/jpeg",
66
+ jpeg: "image/jpeg",
67
+ webp: "image/webp",
68
+ };
69
+
70
+ /** `photo.PNG` → `image/png`; anything outside the table → undefined. */
71
+ export function portraitMimeType(path: string): string | undefined {
72
+ const ext = /\.([^./\\]+)$/.exec(path)?.[1]?.toLowerCase();
73
+ return ext ? PORTRAIT_MIME_TYPES[ext] : undefined;
74
+ }
75
+
76
+ /**
77
+ * The LLM-written concept the image prompt is built from. `cappedText`, not
78
+ * `.max()` — §112 as applied in beats.ts: LLM output is untrusted input, and
79
+ * a concept one word over budget must cost a word, never the thumbnail.
80
+ * `overlayText` is ADDITIONALLY passed through `coverHeadline` by the caller
81
+ * to cap WORDS — the schema caps characters, but overlay text at thumbnail
82
+ * size has the same 4-9 word ceiling a cover banner does (§35).
83
+ */
84
+ export const ThumbnailConceptSchema = z.object({
85
+ /** One vivid scene — what the image shows. */
86
+ scene: cappedText(300),
87
+ /** The 3-6 word text rendered ON the image, verbatim. */
88
+ overlayText: cappedText(60),
89
+ /** Palette, lighting, mood — craft direction for the image model. */
90
+ styleNotes: cappedText(300),
91
+ });
92
+ export type ThumbnailConcept = z.infer<typeof ThumbnailConceptSchema>;
93
+
94
+ /**
95
+ * The §35 WORD cap on overlay text, as the ONE helper every concept-accepting
96
+ * path calls (2026-08-17, editor thumbnail panel): thumbnailStep, the
97
+ * pre-render approval, the interactive edit parser and the editor's
98
+ * regenerate endpoint must all produce byte-identical text — the overlay
99
+ * feeds `thumbnailImageCacheName`, so two spellings of this cap would mint
100
+ * two cache keys for one concept. `|| raw` keeps the schema-capped text when
101
+ * coverHeadline rejects the whole line (e.g. all-stopword): an overlay the
102
+ * user typed must never silently become an EMPTY overlay.
103
+ */
104
+ export function approvedOverlayText(raw: string): string {
105
+ return coverHeadline(raw) || raw;
106
+ }
107
+
108
+ /**
109
+ * The workdir image cache's filename: model + the concept actually prompted +
110
+ * the portrait's CONTENT (sha1 of its bytes, not its path or mtime — a
111
+ * swapped portrait at the same path must regenerate). Extracted from
112
+ * thumbnailStep (2026-08-17) because the editor's regenerate endpoint writes
113
+ * the same cache, and a second spelling of the key would let the two callers
114
+ * cache past each other. The concept is serialized with the field order
115
+ * PINNED here rather than trusting the caller's object: both callers build
116
+ * concepts through `ThumbnailConceptSchema`, whose parse order this matches,
117
+ * so existing caches stay valid — and a future caller with a hand-built
118
+ * object cannot silently re-key everything by ordering its literal
119
+ * differently.
120
+ */
121
+ export function thumbnailImageCacheName(
122
+ model: string,
123
+ concept: ThumbnailConcept,
124
+ portraitSha1: string,
125
+ ): string {
126
+ const key = createHash("sha1")
127
+ .update(
128
+ JSON.stringify([
129
+ model,
130
+ {
131
+ scene: concept.scene,
132
+ overlayText: concept.overlayText,
133
+ styleNotes: concept.styleNotes,
134
+ },
135
+ portraitSha1,
136
+ ]),
137
+ )
138
+ .digest("hex")
139
+ .slice(0, 8);
140
+ return `thumbnail-${key}.png`;
141
+ }
142
+
143
+ /**
144
+ * The workdir file the pre-render approval step writes (thumbnail UX,
145
+ * 2026-08-16): once it exists, `thumbnailStep` uses it VERBATIM and never
146
+ * asks a model for a concept again — the user approved (and possibly edited)
147
+ * this exact text, and a fresh concept call would discard their edit.
148
+ */
149
+ export const THUMBNAIL_APPROVED_BASENAME = "thumbnail-concept-approved.json";
150
+
151
+ /**
152
+ * What the approved file holds: the approved concept, or an explicit
153
+ * `{skip: true}` — the user answered "skip thumbnail" at the approval prompt,
154
+ * and that decision must survive into the (non-interactive) replay exactly
155
+ * like an approval does, as a LOUD skip rather than a silent regeneration.
156
+ * Skip variant FIRST in the union: a concept object can never carry `skip`
157
+ * and a skip object can never carry the concept's required fields, so order
158
+ * only matters for error messages — but skip-first means a hand-added `skip`
159
+ * key wins over a leftover concept body instead of being ignored.
160
+ */
161
+ export const ThumbnailConceptApprovedSchema = z.union([
162
+ z.object({ skip: z.literal(true) }),
163
+ ThumbnailConceptSchema,
164
+ ]);
165
+ export type ThumbnailConceptApproved = z.infer<typeof ThumbnailConceptApprovedSchema>;
166
+
167
+ /**
168
+ * How much transcript the concept prompt carries. Half the youtube.ts cap: a
169
+ * thumbnail concept is about the video's ONE claim, not its coverage, and
170
+ * the hook/intent lines already carry the strongest steer.
171
+ */
172
+ export const THUMBNAIL_TRANSCRIPT_CHAR_CAP = 4000;
173
+
174
+ export interface ThumbnailConceptPromptArgs {
175
+ /** The producer's hook, when a beat sheet exists — the strongest claim. */
176
+ hook?: string;
177
+ /** `--intent`, when the run had one. */
178
+ intent?: string;
179
+ /** The repaired transcript's plain text — what the viewer actually hears. */
180
+ transcriptText: string;
181
+ /** Who watches the channel (`--audience` / config) — steers the angle. */
182
+ audience?: string;
183
+ /**
184
+ * The durable thumbnail steer (`--thumbnail-brief` / config). Marked
185
+ * must-honor in the prompt: this is the user's standing instruction, not a
186
+ * suggestion the model may trade away against its own craft rules.
187
+ */
188
+ brief?: string;
189
+ /**
190
+ * The youtube pack's FIRST title, when the pack already exists — thumbnail
191
+ * and title must tell one story, and the first title is the pack's lead
192
+ * angle. The pre-render approval step runs BEFORE the pack is generated
193
+ * (the pack writes after render), so it passes the hook instead; only
194
+ * thumbnailStep's own post-pack concept call can supply this.
195
+ */
196
+ titleAngle?: string;
197
+ /**
198
+ * A per-call creator note ("regenerate with a note" at the approval
199
+ * prompt). Must-honor like the brief, but transient — it lives in this one
200
+ * call and is never persisted to config or the concept cache.
201
+ */
202
+ note?: string;
203
+ }
204
+
205
+ /**
206
+ * Pure prompt builder for the concept call, separated from the provider so
207
+ * the include/omit matrix (hook, intent, audience, brief, titleAngle, note)
208
+ * and the transcript cap are testable without an LLM.
209
+ */
210
+ export function buildThumbnailConceptPrompt(args: ThumbnailConceptPromptArgs): {
211
+ system: string;
212
+ user: string;
213
+ } {
214
+ const system =
215
+ "You are a YouTube thumbnail strategist designing a high-CTR thumbnail concept for a " +
216
+ "finished video.\n" +
217
+ "- scene: ONE vivid scene — a single concrete image, not a collage of ideas.\n" +
218
+ "- overlayText: 3-6 punchy words rendered on the image. A claim or a tension, never a " +
219
+ "full sentence, and never a clickbait claim the video does not deliver.\n" +
220
+ "- styleNotes: palette, lighting and mood direction for the image model — specific " +
221
+ "enough to constrain it, short enough to not fight the scene.\n" +
222
+ // Pose conflict fix (debugged 2026-08-16): the reference photo is frontal
223
+ // with arms crossed, but a concept saying "holding his head in
224
+ // frustration" made the image model follow the scene's choreography over
225
+ // the image prompt's keep-pose rule. The fix starts HERE — a concept
226
+ // that never choreographs the person cannot lose that fight downstream.
227
+ "- The creator appears from a fixed reference photo whose pose CANNOT change " +
228
+ "(frontal, natural). Design the scene AROUND the person — lighting, props, screen " +
229
+ "content, composition — never choreograph the person's body or hands.";
230
+ const capped =
231
+ args.transcriptText.length > THUMBNAIL_TRANSCRIPT_CHAR_CAP
232
+ ? // Slice + say so (youtube.ts posture): the model must know it is
233
+ // reading an excerpt, or it will anchor the concept on the first half.
234
+ `${args.transcriptText.slice(0, THUMBNAIL_TRANSCRIPT_CHAR_CAP)}\n[transcript truncated — the video continues]`
235
+ : args.transcriptText;
236
+ const user =
237
+ (args.intent ? `Intent: ${args.intent}\n` : "") +
238
+ (args.hook ? `Hook (already chosen by the producer): ${args.hook}\n` : "") +
239
+ // One story across the upload: when the pack's lead title exists, the
240
+ // thumbnail must be its visual restatement, not a second pitch.
241
+ (args.titleAngle
242
+ ? `Video title (already chosen — the thumbnail must tell the same story): ${args.titleAngle}\n`
243
+ : "") +
244
+ (args.audience ? `Audience: ${args.audience}\n` : "") +
245
+ (args.brief ? `Creator brief (must be honored): ${args.brief}\n` : "") +
246
+ (args.note ? `Creator note (must be honored): ${args.note}\n` : "") +
247
+ `\nTranscript:\n${capped}`;
248
+ return { system, user };
249
+ }
250
+
251
+ /** One editorial call → a validated concept. */
252
+ export async function generateThumbnailConcept(
253
+ provider: LlmProvider,
254
+ args: ThumbnailConceptPromptArgs,
255
+ ): Promise<ThumbnailConcept> {
256
+ const { system, user } = buildThumbnailConceptPrompt(args);
257
+ return provider.complete({
258
+ system,
259
+ user,
260
+ schema: ThumbnailConceptSchema,
261
+ schemaName: "thumbnail_concept",
262
+ tier: "editorial",
263
+ });
264
+ }
265
+
266
+ /**
267
+ * The image-model prompt, with the craft rules baked in as fixed text rather
268
+ * than left to the concept call: the concept model picks WHAT to show, this
269
+ * function owns HOW a YouTube thumbnail is built (16:9, subject in a third,
270
+ * overlay verbatim, legible at grid size). Pure — the hasPortrait branch is
271
+ * a table test.
272
+ *
273
+ * `revisionNote` is the post-generation retry loop's one input ("regenerate
274
+ * with a note"): the concept is UNCHANGED, the note rides the image prompt
275
+ * only, marked must-honor and appended last so it reads as the final word.
276
+ */
277
+ export function buildThumbnailPrompt(
278
+ concept: ThumbnailConcept,
279
+ hasPortrait: boolean,
280
+ revisionNote?: string,
281
+ ): string {
282
+ const lines = [
283
+ "Create a YouTube thumbnail image.",
284
+ "",
285
+ ...(hasPortrait
286
+ ? [
287
+ // Identity FIRST, before Scene/Style (pose incident, debugged
288
+ // 2026-08-16): with the identity block buried under the scene, a
289
+ // concept that choreographed the person ("holding his head in
290
+ // frustration") beat the keep-pose rule — the image model weighs
291
+ // what it reads first. Identity is the HARD constraint, everything
292
+ // else is negotiable (user directive 2026-08-16: the first
293
+ // generation produced a face that "isn't me" — a thumbnail with
294
+ // someone else's face is worse than no thumbnail). The reference
295
+ // photo is ground truth: the model may relight and
296
+ // recontextualize, never redraw the person.
297
+ "Identity requirements — these override everything below:",
298
+ "- The person in the reference photo MUST appear with their face reproduced at " +
299
+ "100% accuracy — this is the single most important requirement. Treat the " +
300
+ "reference photo as ground truth for identity: exact same facial structure, " +
301
+ "eyes, nose, mouth, skin tone, facial hair, hairline and glasses. Do not " +
302
+ "idealize, de-age, slim, or otherwise alter any facial feature.",
303
+ "- Keep the head pose, angle and framing of the face as close to the reference " +
304
+ "photo as possible. Only the lighting and color grade may adapt to match the " +
305
+ "scene; the face itself must stay photorealistic and identical to the " +
306
+ "reference, never stylized or repainted.",
307
+ "- Place that person prominently in the left or right third of the frame, " +
308
+ "integrated naturally into the scene (matching light and color, no cut-out " +
309
+ "look). If any requirement conflicts with facial accuracy, facial accuracy " +
310
+ "wins.",
311
+ "- The scene adapts to the person; never re-pose, re-angle, or choreograph the " +
312
+ "person to fit the scene.",
313
+ "",
314
+ ]
315
+ : []),
316
+ `Scene: ${concept.scene}`,
317
+ `Style: ${concept.styleNotes}`,
318
+ "",
319
+ "Requirements:",
320
+ "- 16:9 landscape, at least 1280x720.",
321
+ // Quotes on their own line: overlay text is the one part of the image
322
+ // with a right answer, and image models paraphrase anything stated
323
+ // loosely — but the instruction word itself must not touch the quoted
324
+ // string. The first field run (2026-08-16) used `reading EXACTLY: "…"`
325
+ // and the model painted "EXACTLY:" into the thumbnail as a headline.
326
+ "- Render bold, high-contrast overlay text. The overlay must contain this",
327
+ " text and nothing else:",
328
+ ` ${concept.overlayText}`,
329
+ "- Vivid and eye-catching but not cluttered — one focal point.",
330
+ "- No watermarks, no logos, no borders.",
331
+ "- Every element must stay legible when the image is displayed 320px wide.",
332
+ ...(revisionNote
333
+ ? ["", `Revision note from the creator (must be honored): ${revisionNote}`]
334
+ : []),
335
+ ];
336
+ return lines.join("\n");
337
+ }
338
+
339
+ /**
340
+ * Pull the first image part's bytes out of a generateContent response.
341
+ *
342
+ * Isolated in ONE small function on purpose: the @google/genai response
343
+ * shape is the SDK's to drift (plan risk note, 2026-08-16), and when it
344
+ * does, this is the only code that knows about `candidates[].content.parts`.
345
+ * Treats the response as `unknown` so the pure tests exercise it with plain
346
+ * objects and never import the SDK.
347
+ */
348
+ export function extractImageBytes(response: unknown): Uint8Array {
349
+ const candidates = (response as { candidates?: unknown })?.candidates;
350
+ if (Array.isArray(candidates)) {
351
+ for (const candidate of candidates) {
352
+ const parts = (candidate as { content?: { parts?: unknown } })?.content?.parts;
353
+ if (!Array.isArray(parts)) continue;
354
+ for (const part of parts) {
355
+ const data = (part as { inlineData?: { data?: unknown } })?.inlineData?.data;
356
+ if (typeof data === "string" && data.length > 0) {
357
+ return new Uint8Array(Buffer.from(data, "base64"));
358
+ }
359
+ }
360
+ }
361
+ }
362
+ throw new Error(
363
+ "the model returned no image — the response had no inlineData part " +
364
+ "(model refusal or a text-only reply)",
365
+ );
366
+ }
367
+
368
+ export interface GenerateThumbnailImageOptions {
369
+ apiKey: string;
370
+ model: string;
371
+ prompt: string;
372
+ /** The creator's portrait as base64 `inlineData`, when the run has one. */
373
+ portrait?: { data: string; mimeType: string };
374
+ }
375
+
376
+ /**
377
+ * The ONE I/O function of this module: call Gemini image generation and
378
+ * return the image bytes.
379
+ *
380
+ * The SDK import is LAZY and lives here, nowhere else: core is near-zero-dep
381
+ * by design (only @anthropic-ai/sdk + zod), and every run that is not a
382
+ * `--youtube`-with-portrait-and-key run must never pay for loading
383
+ * @google/genai — a static import would tax every produce for a feature most
384
+ * runs never reach.
385
+ *
386
+ * Errors are NOT retried and surface the API's message verbatim — the
387
+ * isNonRetryableAgyFailure posture (§132): the model slug is user-specified,
388
+ * a rejection of it is deterministic, and a retry loop would only restate it.
389
+ */
390
+ export async function generateThumbnailImage(
391
+ opts: GenerateThumbnailImageOptions,
392
+ ): Promise<Uint8Array> {
393
+ const { GoogleGenAI } = await import("@google/genai");
394
+ const ai = new GoogleGenAI({ apiKey: opts.apiKey });
395
+ const response = await ai.models.generateContent({
396
+ model: opts.model,
397
+ contents: [
398
+ {
399
+ role: "user",
400
+ parts: [
401
+ // Portrait FIRST, prompt second — the prompt refers back to "the
402
+ // reference photo", so the photo must already be on the table.
403
+ ...(opts.portrait
404
+ ? [{ inlineData: { data: opts.portrait.data, mimeType: opts.portrait.mimeType } }]
405
+ : []),
406
+ { text: opts.prompt },
407
+ ],
408
+ },
409
+ ],
410
+ });
411
+ return extractImageBytes(response);
412
+ }
package/src/transcribe.ts CHANGED
@@ -78,6 +78,51 @@ function repairSplitSegments(json: WhisperJson): WhisperJson {
78
78
  };
79
79
  }
80
80
 
81
+ /**
82
+ * Run length at which a stack of zero-length words at ONE instant stops being
83
+ * a rounding artifact and becomes a repetition-loop hallucination. Real speech
84
+ * never emits 8 tokens at a single instant; the field case emitted 118.
85
+ */
86
+ export const REPETITION_BURST_MIN = 8;
87
+
88
+ /**
89
+ * Drop whisper repetition-loop bursts (field case 2026-08-18): an Urdu take
90
+ * re-decoded a whole phrase as 118 CONSECUTIVE tokens all stamped
91
+ * `from === to === 31040` — zero length, at one instant. The stamp repair
92
+ * below then fans such a burst out into 118 fabricated 50ms words marching
93
+ * forward from 31.04s, so the phrase ships TWICE in the captions (31.04s and
94
+ * 33.54s) and a fifth of the transcript carries the tell-tale exactly-0.05s
95
+ * duration. `-mc 0` in whisperArgs is the decoder-side mitigation for the same
96
+ * failure; it did not prevent this occurrence, and it can never repair an
97
+ * already-cached transcript.json — hence a parse-side guard too.
98
+ *
99
+ * A burst is a MAXIMAL run of consecutive zero-length/inverted words sharing
100
+ * one `start`. Equality is exact, not epsilon: these stamps are integer
101
+ * milliseconds divided by 1000, so members of one burst are the same double
102
+ * bit-for-bit, and a tolerance would only start swallowing real neighbors.
103
+ * Runs shorter than REPETITION_BURST_MIN fall through untouched — a lone
104
+ * zero-length stamp is a rounding artifact, not a hallucination. The drop is
105
+ * silent by design: this function is pure and total, and there is no logging
106
+ * channel in the parse path to warn on.
107
+ */
108
+ export function dropRepetitionBursts(words: readonly Word[]): Word[] {
109
+ const out: Word[] = [];
110
+ let i = 0;
111
+ while (i < words.length) {
112
+ const w = words[i]!;
113
+ if (w.end > w.start) {
114
+ out.push(w);
115
+ i++;
116
+ continue;
117
+ }
118
+ let j = i + 1;
119
+ while (j < words.length && words[j]!.end <= words[j]!.start && words[j]!.start === w.start) j++;
120
+ if (j - i < REPETITION_BURST_MIN) for (let k = i; k < j; k++) out.push(words[k]!);
121
+ i = j;
122
+ }
123
+ return out;
124
+ }
125
+
81
126
  const STRICT_UTF8 = new TextDecoder("utf-8", { fatal: true });
82
127
 
83
128
  /**
@@ -115,11 +160,19 @@ export function parseWhisperJson(json: WhisperJson): Transcript {
115
160
  if (!raw || !raw.trim()) continue;
116
161
  const text = raw.trim();
117
162
  if (NOISE_TOKEN.test(text)) continue;
163
+ // A word already CLOSED by Arabic-script sentence punctuation refuses
164
+ // continuations (field case 2026-08-18): whisper emits `۔` and the next
165
+ // sentence's first token with no leading whitespace, and the plain
166
+ // whitespace rule fused them into one unsplittable word ("ہوں۔اس").
167
+ // Deliberately NOT the Latin `.`/`!`/`?` — whisper tokenizes decimals
168
+ // ("3", ".", "5") and abbreviations as bare continuations too, and
169
+ // splitting those would shred "3.5" into two words. ۔ (U+06D4) and
170
+ // ؟ (U+061F) have no such second job.
118
171
  const startsWord = /^\s/.test(raw) || words.length === 0;
119
172
  const start = seg.offsets.from / 1000;
120
173
  const end = seg.offsets.to / 1000;
121
174
  const last = words[words.length - 1];
122
- if (!startsWord && last) {
175
+ if (!startsWord && last && !/[۔؟]$/.test(last.text)) {
123
176
  last.text += text;
124
177
  last.end = Math.max(last.end, end);
125
178
  } else {
@@ -146,14 +199,18 @@ export function parseWhisperJson(json: WhisperJson): Transcript {
146
199
  else if (next) next.start = Math.min(next.start, w.start);
147
200
  words.splice(i, 1);
148
201
  }
202
+ // BEFORE the repair loop, never after: the repair rewrites every burst
203
+ // member into a distinct monotone stamp, so once it has run the shared
204
+ // timestamp — the only evidence a burst existed — is gone.
205
+ const kept = dropRepetitionBursts(words);
149
206
  // Whisper occasionally emits zero-length or inverted stamps; repair minimally.
150
- for (let i = 0; i < words.length; i++) {
151
- const w = words[i]!;
207
+ for (let i = 0; i < kept.length; i++) {
208
+ const w = kept[i]!;
152
209
  if (w.end <= w.start) w.end = w.start + 0.05;
153
- const next = words[i + 1];
210
+ const next = kept[i + 1];
154
211
  if (next && next.start < w.end) next.start = w.end;
155
212
  }
156
- return { language: json.result?.language ?? "en", words };
213
+ return { language: json.result?.language ?? "en", words: kept };
157
214
  }
158
215
 
159
216
  export interface WhisperOptions {
@@ -169,6 +226,15 @@ export interface WhisperOptions {
169
226
  * stay byte-identical to what English-suffixed models always got.
170
227
  */
171
228
  language?: string;
229
+ /**
230
+ * Initial decoder prompt (`--prompt`), used to bias recognition toward the
231
+ * user's vocabulary (F4 dictionary, 2026-08-16: "Jason" for JSON). Left
232
+ * unset, the spawned args stay byte-identical to every pre-dictionary run.
233
+ * Known risk (documented in the flag's help): a whisper-cli built before
234
+ * the flag existed rejects it with its own loud error — accepted over
235
+ * silently dropping the user's terms.
236
+ */
237
+ prompt?: string;
172
238
  }
173
239
 
174
240
  /**
@@ -183,12 +249,34 @@ export function whisperArgs(opts: WhisperOptions, wavPath: string): string[] {
183
249
  "-oj",
184
250
  "-of", opts.outBase,
185
251
  "-ml", "1",
252
+ // No text context across 30s decode windows (field case 2026-08-18): an
253
+ // Urdu take hit whisper's repetition loop — a whole sentence re-decoded
254
+ // as 261 zero-duration tokens — and carrying the previous window's text
255
+ // into the decoder is the known trigger. `-mc 0` is the standard
256
+ // mitigation and leaves `--prompt` (the dictionary bias) untouched.
257
+ // Cached transcript.json files decoded without it are knowingly still
258
+ // reused (transcriptCacheReusable's no-spurious-retranscribe rule);
259
+ // delete a workdir's transcript.json to re-decode with it.
260
+ "-mc", "0",
186
261
  "--no-prints",
187
262
  ];
188
263
  if (opts.language !== undefined) args.push("-l", opts.language);
264
+ if (opts.prompt !== undefined) args.push("--prompt", opts.prompt);
189
265
  return args;
190
266
  }
191
267
 
268
+ /**
269
+ * The `--prompt` text for a user dictionary. whisper.cpp treats the prompt as
270
+ * preceding context, so a plain vocabulary list is enough to bias the decoder
271
+ * toward these spellings ("Jason" → "JSON", 2026-08-16 field report). Pure
272
+ * and undefined-for-empty so the no-dictionary invocation stays byte-identical
273
+ * to what every run before the flag got.
274
+ */
275
+ export function whisperPromptFor(dictionary: readonly string[]): string | undefined {
276
+ if (dictionary.length === 0) return undefined;
277
+ return `Vocabulary: ${dictionary.join(", ")}.`;
278
+ }
279
+
192
280
  export async function runWhisper(opts: WhisperOptions, wavPath: string): Promise<Transcript> {
193
281
  await run(opts.whisperPath, whisperArgs(opts, wavPath));
194
282
  // Bytes, not "utf8": the utf8 read is where the §130 split characters died.
package/src/zoom.ts CHANGED
@@ -48,6 +48,13 @@ export interface ZoomPlan {
48
48
  segments: ZoomSegment[];
49
49
  /** How many cut-free clips the plan covers — logged, never inferred. */
50
50
  clips: number;
51
+ /**
52
+ * The allowedClips split, carried on the plan so the CLI log reports the
53
+ * counts the plan actually acted on instead of re-deriving them (and
54
+ * possibly disagreeing after the boundary cleaning).
55
+ */
56
+ zoomedClips: number;
57
+ staticClips: number;
51
58
  /** The ramp actually used, so the log can't drift from the behaviour. */
52
59
  rampSec: number;
53
60
  }
@@ -63,6 +70,18 @@ export interface ZoomPlanOptions {
63
70
  clipStarts?: readonly number[];
64
71
  /** Seconds the push takes to arrive before it holds. */
65
72
  rampSec?: number;
73
+ /**
74
+ * Per-clip zoom permission, PARALLEL TO `clipStarts` by index. User
75
+ * decision 2026-08-16 — "Face-only. If there's anything else, then no
76
+ * zoom": the always-on idle push visibly SLID screen-recording content,
77
+ * so a clip whose subject is not a face gets NO segments at all rather
78
+ * than a flat one. `zoomScaleAt` answers 1 outside the plan, and the
79
+ * premiere export's `zoomKeyframesFor` collapses a segment-free span to a
80
+ * plain scale-1 value, so downstream needs zero changes. Entries missing
81
+ * off the end of a short mask read as allowed, and an absent mask is
82
+ * today's plan exactly — pre-F1 callers are byte-identical.
83
+ */
84
+ allowedClips?: readonly boolean[];
66
85
  }
67
86
 
68
87
  /**
@@ -83,13 +102,32 @@ export const ZOOM_MAX_SCALE = 1.05;
83
102
  */
84
103
  export const ZOOM_RAMP_SEC = 8;
85
104
 
86
- /** Clip starts, cleaned: in range, unique, sorted, and always including 0. */
87
- function clipBoundaries(starts: readonly number[] | undefined, duration: number): number[] {
88
- const seen = new Set<number>([0]);
89
- for (const t of starts ?? []) {
90
- if (Number.isFinite(t) && t > 0 && t < duration) seen.add(t);
91
- }
92
- return [...seen].sort((a, b) => a - b);
105
+ /**
106
+ * Clip starts, cleaned — in range, unique, sorted, always including 0 — each
107
+ * carrying its `allowedClips` verdict. The verdict is paired with its start
108
+ * BY INDEX before any of the cleaning, so dedupe/sort can never shift a
109
+ * verdict onto a different clip; duplicated starts are one clip and OR their
110
+ * verdicts (any pairing that vouches "face" wins — losing the push on a face
111
+ * clip is the regression, holding still an extra clip is merely conservative
112
+ * the wrong way). The synthetic 0 boundary is allowed unless the caller's
113
+ * own list contains a 0 that says otherwise. Missing mask entries read as
114
+ * allowed (`!== false`), which is also what makes an absent mask identical
115
+ * to the pre-mask contract.
116
+ */
117
+ function clipBoundaries(
118
+ starts: readonly number[] | undefined,
119
+ allowed: readonly boolean[] | undefined,
120
+ duration: number,
121
+ ): Array<{ start: number; allowed: boolean }> {
122
+ const byStart = new Map<number, boolean>();
123
+ (starts ?? []).forEach((t, i) => {
124
+ if (!Number.isFinite(t) || t < 0 || t >= duration) return;
125
+ byStart.set(t, (byStart.get(t) ?? false) || allowed?.[i] !== false);
126
+ });
127
+ if (!byStart.has(0)) byStart.set(0, true);
128
+ return [...byStart.entries()]
129
+ .map(([start, ok]) => ({ start, allowed: ok }))
130
+ .sort((a, b) => a.start - b.start);
93
131
  }
94
132
 
95
133
  export function buildZoomPlan(
@@ -98,14 +136,21 @@ export function buildZoomPlan(
98
136
  ): ZoomPlan {
99
137
  const maxScale = opts.maxScale ?? ZOOM_MAX_SCALE;
100
138
  const rampSec = opts.rampSec ?? ZOOM_RAMP_SEC;
101
- if (outputDurationSec <= 0) return { segments: [], clips: 0, rampSec };
139
+ if (outputDurationSec <= 0) {
140
+ return { segments: [], clips: 0, zoomedClips: 0, staticClips: 0, rampSec };
141
+ }
102
142
 
103
- const starts = clipBoundaries(opts.clipStarts, outputDurationSec);
143
+ const starts = clipBoundaries(opts.clipStarts, opts.allowedClips, outputDurationSec);
104
144
  const segments: ZoomSegment[] = [];
145
+ const staticClips = starts.filter((s) => !s.allowed).length;
105
146
 
106
147
  for (let i = 0; i < starts.length; i++) {
107
- const start = starts[i]!;
108
- const end = i + 1 < starts.length ? starts[i + 1]! : outputDurationSec;
148
+ const { start, allowed } = starts[i]!;
149
+ // A disallowed clip emits NOTHING — not a flat segment. `zoomScaleAt`
150
+ // returns 1 for any instant no segment claims (zoom.ts's own "1 outside
151
+ // the plan" contract), so a hole in the plan IS the static camera.
152
+ if (!allowed) continue;
153
+ const end = i + 1 < starts.length ? starts[i + 1]!.start : outputDurationSec;
109
154
  const length = end - start;
110
155
  if (length <= 1e-9) continue;
111
156
 
@@ -120,7 +165,13 @@ export function buildZoomPlan(
120
165
  }
121
166
  }
122
167
 
123
- return { segments, clips: starts.length, rampSec };
168
+ return {
169
+ segments,
170
+ clips: starts.length,
171
+ zoomedClips: starts.length - staticClips,
172
+ staticClips,
173
+ rampSec,
174
+ };
124
175
  }
125
176
 
126
177
  /**