@ossclip/core 0.1.24 → 0.1.25

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,434 @@
1
+ import { z } from "zod/v4";
2
+ import type { Word } from "../schema";
3
+ import type { TimeMap } from "../timemap";
4
+ import { isSentenceEnd } from "../clip";
5
+ import type { LlmProvider } from "./provider";
6
+ import { cappedText } from "./beats";
7
+
8
+ /**
9
+ * The `--youtube` SEO pack (Y2, 2026-08-16; prompt v2 2026-08-17): title
10
+ * options, a description with measured chapter timestamps spliced in,
11
+ * hashtags and tags, written beside the video as `<out>.youtube.md` so the
12
+ * upload form is a paste job instead of a second writing session.
13
+ *
14
+ * Free text is `cappedText`, not `.max()` — the §112 doctrine as applied in
15
+ * beats.ts: LLM output is untrusted input, validated where the pipeline can
16
+ * still degrade instead of at the point where it can only die. A title one
17
+ * word over budget must cost a word, never the run.
18
+ *
19
+ * Two sections a human YouTube-strategist workflow would include are
20
+ * deliberately NOT here (prompt v2 decision):
21
+ * - competitor analysis — the provider has no browsing, so "competitor
22
+ * research" would be confident hallucination dressed up as strategy;
23
+ * - a thumbnail prompt — ossclip GENERATES the actual thumbnail from the
24
+ * real portrait (thumbnail.ts), which beats prose describing an imaginary
25
+ * one.
26
+ */
27
+
28
+ /**
29
+ * Bump whenever buildYoutubePrompt changes what it asks for: a prompt change
30
+ * changes the answer, so the Y2 pack cache key carries this (the §78
31
+ * cache-key posture) — an old cached pack must not survive a new prompt.
32
+ */
33
+ export const YOUTUBE_PROMPT_VERSION = "v2";
34
+
35
+ export const YoutubeChapterSchema = z.object({
36
+ /** Output-timeline seconds — the produced video's clock, not the source's. */
37
+ atSec: z.number().nonnegative(),
38
+ title: cappedText(80),
39
+ });
40
+ export type YoutubeChapter = z.infer<typeof YoutubeChapterSchema>;
41
+
42
+ /**
43
+ * The three labeled title angles for YouTube's Test & Compare A/B feature
44
+ * (prompt v2): browse = story/curiosity for home-page CTR, search =
45
+ * keyword/tool-name for evergreen search, benefit = the ROI claim. "alt"
46
+ * covers any extra option past the three.
47
+ */
48
+ export const TitleAngleSchema = z.enum(["browse", "search", "benefit", "alt"]);
49
+ export type TitleAngle = z.infer<typeof TitleAngleSchema>;
50
+
51
+ export const YoutubePackSchema = z.object({
52
+ /** Distinct angles on the same video, not five rewordings of one. */
53
+ titles: z.array(cappedText(100)).min(3).max(5),
54
+ /**
55
+ * Parallel to `titles`, one angle label each. OPTIONAL for back-compat:
56
+ * packs approved before prompt v2 carry bare titles, and the approved-file
57
+ * contract (YOUTUBE_APPROVED_BASENAME) means those files must keep parsing
58
+ * verbatim forever.
59
+ */
60
+ titleAngles: z.array(TitleAngleSchema).optional(),
61
+ /** Prompted for keyword-first-lines, hashtags-at-end; free-form otherwise.
62
+ * Chapter timestamps are NOT the model's to write — the formatter splices
63
+ * the measured ones in (spliceTimestampsIntoDescription). */
64
+ description: z.string(),
65
+ hashtags: z.array(z.string()),
66
+ tags: z.array(z.string()),
67
+ chapters: z.array(YoutubeChapterSchema).optional(),
68
+ /** Advisory first-60-seconds retention note for the CREATOR — never pasted
69
+ * into YouTube. Optional: pre-v2 packs never carried it. */
70
+ hook60: cappedText(400).optional(),
71
+ /** A ready-to-post LinkedIn announcement. Optional, pre-v2 back-compat. */
72
+ linkedinPost: cappedText(1500).optional(),
73
+ /** A short YouTube community post for existing subscribers. Optional. */
74
+ communityPost: cappedText(400).optional(),
75
+ });
76
+ export type YoutubePack = z.infer<typeof YoutubePackSchema>;
77
+
78
+ /** YouTube's hard cap on the tags field, counted over the comma-joined text. */
79
+ export const YOUTUBE_TAGS_LIMIT = 500;
80
+
81
+ /**
82
+ * Post-parse guard for the tags field: drop tags from the END until the
83
+ * comma-joined line fits YouTube's limit. From the end because the prompt
84
+ * asks for relevance order — the least relevant tag is the one a budget cut
85
+ * should cost. Degrade-don't-die (§112): the schema cannot express a
86
+ * joined-length cap, and refusing the whole pack over an over-enthusiastic
87
+ * tag list would discard the titles and description that were fine.
88
+ */
89
+ export function trimTagsToLimit(tags: string[], limit: number = YOUTUBE_TAGS_LIMIT): string[] {
90
+ const kept = [...tags];
91
+ while (kept.length > 0 && kept.join(", ").length > limit) {
92
+ kept.pop();
93
+ }
94
+ return kept;
95
+ }
96
+
97
+ /**
98
+ * The workdir file the editor's SEO panel writes (2026-08-17), mirroring the
99
+ * thumbnail approval contract (THUMBNAIL_APPROVED_BASENAME, thumbnail.ts):
100
+ * once it exists, produce's Y2 block uses the pack VERBATIM — no cache
101
+ * lookup, no LLM call — because an edited pack is the user's decision, and a
102
+ * fresh generation would silently discard it. Holds a plain `YoutubePack`,
103
+ * no skip variant: the `--youtube` flag itself gates the feature, so there
104
+ * is no "opted in then declined" state for a pack the way there is for a
105
+ * thumbnail. Deleting the file is the regenerate gesture.
106
+ */
107
+ export const YOUTUBE_APPROVED_BASENAME = "youtube-pack-approved.json";
108
+
109
+ /**
110
+ * How much transcript the prompt carries. Field-measured 2026-08-17: at the
111
+ * old 10k cap a 21-minute video's chapters STOPPED AT 8:32 — the model can
112
+ * only chapter what it reads, and a truncated transcript silently produces
113
+ * chapters for the excerpt while looking complete. 60k chars ≈ 15k tokens
114
+ * covers ~45 minutes of stamped talk for pennies; past that the truncation
115
+ * note at least tells the model (and the chapters) where the coverage ends.
116
+ * `stampedTranscript` pre-caps to this same constant, so
117
+ * `buildYoutubePrompt`'s own cap is a backstop for raw-text callers, never a
118
+ * second cut of stamped input.
119
+ */
120
+ export const YOUTUBE_TRANSCRIPT_CHAR_CAP = 60_000;
121
+
122
+ /** 125 → "2:05" — YouTube's own chapter-list spelling. */
123
+ function chapterStamp(sec: number): string {
124
+ const whole = Math.max(0, Math.floor(sec));
125
+ const m = Math.floor(whole / 60);
126
+ const s = whole % 60;
127
+ return `${m}:${String(s).padStart(2, "0")}`;
128
+ }
129
+
130
+ /** What a truncated stamped transcript ends with — the model must know it is
131
+ * reading an excerpt, or it will write a description that promises only the
132
+ * first half. */
133
+ const TRUNCATION_NOTE = "[transcript truncated — the video continues]";
134
+
135
+ /**
136
+ * The transcript as the prompt's chapter evidence (prompt v2): one line per
137
+ * sentence, prefixed with the sentence's first surviving word's OUTPUT time
138
+ * as `[m:ss]`. This is the pack's unique advantage over paste-a-transcript
139
+ * prompt tools — words carry SOURCE times and the TimeMap converts them to
140
+ * the produced video's clock, so chapters can be measured instead of the
141
+ * guessed "0:00 Intro" everyone else ships.
142
+ *
143
+ * Fully-cut sentences (every word `mapWord` → null) are skipped: they are
144
+ * not in the video, and stamping them would hand the model timestamps that
145
+ * point at content the viewer never sees. Pure — words + map in, string
146
+ * out — so the mapping, the cut-skip and the cap are testable without an
147
+ * LLM or a filesystem.
148
+ *
149
+ * The cap reserves room for TRUNCATION_NOTE up front, so the result NEVER
150
+ * exceeds `maxChars` — truncated or not — and whole sentences are dropped
151
+ * rather than sliced mid-line, which would leave a dangling stamp.
152
+ */
153
+ export function stampedTranscript(
154
+ words: readonly Word[],
155
+ map: TimeMap,
156
+ opts: { maxChars?: number } = {},
157
+ ): string {
158
+ const maxChars = opts.maxChars ?? YOUTUBE_TRANSCRIPT_CHAR_CAP;
159
+ // isSentenceEnd reads punctuation off Transcript-shaped input.
160
+ const t = { language: "en", words: [...words] };
161
+ const budget = maxChars - TRUNCATION_NOTE.length - 1;
162
+ const lines: string[] = [];
163
+ let used = 0;
164
+ let truncated = false;
165
+ let start = 0;
166
+ for (let i = 0; i < words.length; i++) {
167
+ if (i < words.length - 1 && !isSentenceEnd(t, i)) continue;
168
+ const sentence = words.slice(start, i + 1);
169
+ start = i + 1;
170
+ // The stamp is the first SURVIVING word's output time, not the first
171
+ // word's: a sentence whose opening filler was cut starts, on the
172
+ // viewer's clock, at the first word they hear.
173
+ let stampSec: number | null = null;
174
+ for (const w of sentence) {
175
+ const mapped = map.mapWord(w);
176
+ if (mapped) {
177
+ stampSec = mapped.start;
178
+ break;
179
+ }
180
+ }
181
+ if (stampSec === null) continue; // the whole sentence was cut
182
+ const line = `[${chapterStamp(stampSec)}] ${sentence.map((w) => w.text).join(" ")}`;
183
+ const cost = lines.length === 0 ? line.length : line.length + 1;
184
+ if (used + cost > budget) {
185
+ truncated = true;
186
+ break;
187
+ }
188
+ lines.push(line);
189
+ used += cost;
190
+ }
191
+ if (truncated) lines.push(TRUNCATION_NOTE);
192
+ return lines.join("\n");
193
+ }
194
+
195
+ /** YouTube ignores a chapter list whose entries run shorter than this. */
196
+ export const YOUTUBE_CHAPTER_MIN_GAP_SEC = 10;
197
+ /** ...or that carries fewer timestamps than this. */
198
+ export const YOUTUBE_CHAPTER_MIN_COUNT = 3;
199
+
200
+ /**
201
+ * Post-parse guard for the chapters field — trimTagsToLimit's posture
202
+ * applied to YouTube's chapter rules, which the schema cannot express:
203
+ * the list must start at 0:00, hold ≥3 entries, and give every chapter
204
+ * ≥10 seconds, or YouTube silently ignores the WHOLE list. Degrade-don't-die
205
+ * (§112): repair what can be repaired, drop what cannot, and return [] —
206
+ * "no chapters", which the formatter renders as no timestamp block — rather
207
+ * than ship a list YouTube would reject wholesale anyway.
208
+ *
209
+ * - unsorted input sorts ascending (the model's list order is untrusted);
210
+ * - chapters at or past the video's end are dropped — a stamp the video
211
+ * never reaches is a hallucinated one;
212
+ * - a first chapter within the 10s minimum of 0 snaps TO 0 (it IS the
213
+ * intro, mis-stamped); one further out gets a synthetic 0:00 "Intro"
214
+ * prepended — the unnamed opening is real content, and 0:00 is mandatory;
215
+ * - of a pair closer than 10s, the LATER one drops (the earlier stamp is
216
+ * the section's true start).
217
+ */
218
+ export function normalizeChapters(
219
+ chapters: readonly YoutubeChapter[],
220
+ durationSec: number,
221
+ ): YoutubeChapter[] {
222
+ const sorted = [...chapters]
223
+ .sort((a, b) => a.atSec - b.atSec)
224
+ .filter((c) => c.atSec < durationSec);
225
+ if (sorted.length === 0) return [];
226
+ const first = sorted[0]!;
227
+ const list: YoutubeChapter[] =
228
+ first.atSec === 0
229
+ ? sorted
230
+ : first.atSec < YOUTUBE_CHAPTER_MIN_GAP_SEC
231
+ ? [{ ...first, atSec: 0 }, ...sorted.slice(1)]
232
+ : [{ atSec: 0, title: "Intro" }, ...sorted];
233
+ const kept: YoutubeChapter[] = [];
234
+ for (const c of list) {
235
+ const prev = kept[kept.length - 1];
236
+ if (prev === undefined || c.atSec - prev.atSec >= YOUTUBE_CHAPTER_MIN_GAP_SEC) kept.push(c);
237
+ }
238
+ return kept.length >= YOUTUBE_CHAPTER_MIN_COUNT ? kept : [];
239
+ }
240
+
241
+ export interface YoutubePromptArgs {
242
+ /** The stamped transcript (stampedTranscript) — sentences prefixed with
243
+ * their `[m:ss]` output-clock time. Plain text still works (raw-text
244
+ * callers just get no measured chapters worth trusting). */
245
+ transcriptText: string;
246
+ /** `--intent`, when the run had one. */
247
+ intent?: string;
248
+ /** The producer's hook, when a beat sheet exists — the strongest claim. */
249
+ hook?: string;
250
+ /** The cover banner text, when a beat sheet wrote one. */
251
+ coverText?: string;
252
+ /**
253
+ * Who watches the channel (`--audience` / config `audience`, 2026-08-16):
254
+ * titles and tags for "junior devs learning agents" are a different pack
255
+ * than for "engineering managers", and only the user knows which one this
256
+ * channel is.
257
+ */
258
+ audience?: string;
259
+ /** Output duration in seconds — the ceiling no chapter may pass. */
260
+ durationSec: number;
261
+ }
262
+
263
+ /**
264
+ * Pure prompt builder, separated from the provider call so the
265
+ * include/omit matrix (intent, hook, coverText, audience) and the transcript
266
+ * cap are testable without an LLM.
267
+ */
268
+ export function buildYoutubePrompt(args: YoutubePromptArgs): { system: string; user: string } {
269
+ const system =
270
+ "You are an expert YouTube growth strategist writing upload metadata for a finished video. " +
271
+ "The transcript's sentences are each prefixed with a [m:ss] timestamp on the FINAL video's " +
272
+ "clock — measured from the actual edit, not estimated. Match the creator's tone from the " +
273
+ "transcript, and never make a claim the video does not deliver: click-drivers that deliver, " +
274
+ "not clickbait.\n" +
275
+ "- titles: 3-5 options for YouTube's Test & Compare A/B feature, each a DISTINCT angle, " +
276
+ "labeled in the parallel titleAngles array: \"browse\" — story/curiosity, zero jargon, " +
277
+ "before/after tension, built for home-page CTR; \"search\" — keyword-heavy, names the exact " +
278
+ "tools and technologies, built for evergreen search; \"benefit\" — the ROI, \"do X in N " +
279
+ "minutes\". Include at least one of each of those three; label any extra option \"alt\". " +
280
+ "Target at most 70 characters (search results truncate around 70, browse cards earlier).\n" +
281
+ "- description: the FIRST TWO lines are the search snippet — the primary keyword plus a " +
282
+ "curiosity hook. Then a body of short paragraphs or bullets carrying the secondary keywords, " +
283
+ "a call to action near the end, and 3-5 hashtags as the LAST line. Do NOT write timestamp " +
284
+ "or chapter lines in the description — the measured chapter timestamps are spliced in " +
285
+ "automatically.\n" +
286
+ "- hashtags: the same 3-5, as an array.\n" +
287
+ "- tags: search keywords and phrases a viewer would type, ordered by relevance — the most " +
288
+ "important first.\n" +
289
+ "- chapters: 4-10, USING ONLY the [m:ss] stamps that appear in the transcript — atSec is " +
290
+ "that stamp converted to seconds, never a time you estimated. The first chapter is at 0. " +
291
+ "Titles at most 50 characters, keyword-bearing but written for a human scanning the list.\n" +
292
+ "- hook60: a first-60-seconds retention note for the CREATOR (it is never published) — e.g. " +
293
+ "\"show the final result at 0:05, then rewind to how\" — grounded in what this transcript " +
294
+ "actually opens with.\n" +
295
+ "- linkedinPost: a story-driven LinkedIn post about this video: short lines with line " +
296
+ "breaks, a curiosity gap, no hashtag spam, ending by pointing to the link in the comments " +
297
+ "(the LinkedIn convention for off-platform links).\n" +
298
+ "- communityPost: a short, casual YouTube community post for existing subscribers.";
299
+ const capped =
300
+ args.transcriptText.length > YOUTUBE_TRANSCRIPT_CHAR_CAP
301
+ ? // Slice + say so: the model must know it is reading an excerpt, or it
302
+ // will write a description that promises only the first half.
303
+ `${args.transcriptText.slice(0, YOUTUBE_TRANSCRIPT_CHAR_CAP)}\n${TRUNCATION_NOTE}`
304
+ : args.transcriptText;
305
+ const user =
306
+ (args.intent ? `Intent: ${args.intent}\n` : "") +
307
+ (args.hook ? `Hook (already chosen by the producer): ${args.hook}\n` : "") +
308
+ (args.coverText ? `Cover headline: ${args.coverText}\n` : "") +
309
+ (args.audience ? `Audience: ${args.audience}\n` : "") +
310
+ `Total runtime: ${chapterStamp(args.durationSec)} — chapters must not exceed it.\n\n` +
311
+ `Transcript:\n${capped}`;
312
+ return { system, user };
313
+ }
314
+
315
+ /** One editorial call → a validated, tag-trimmed, chapter-normalized pack. */
316
+ export async function generateYoutubePack(
317
+ provider: LlmProvider,
318
+ args: YoutubePromptArgs,
319
+ ): Promise<YoutubePack> {
320
+ const { system, user } = buildYoutubePrompt(args);
321
+ const pack = await provider.complete({
322
+ system,
323
+ user,
324
+ schema: YoutubePackSchema,
325
+ schemaName: "youtube_pack",
326
+ tier: "editorial",
327
+ });
328
+ // Both post-parse guards live HERE, beside the call, so every path that
329
+ // stores the pack (the Y2 cache, the approved file) already holds a
330
+ // publishable one. An empty normalized list degrades to "no chapters"
331
+ // (undefined), which the formatter renders as no timestamp block.
332
+ const chapters = normalizeChapters(pack.chapters ?? [], args.durationSec);
333
+ return {
334
+ ...pack,
335
+ tags: trimTagsToLimit(pack.tags),
336
+ chapters: chapters.length > 0 ? chapters : undefined,
337
+ };
338
+ }
339
+
340
+ /** The heading the timestamp block opens with inside the description. */
341
+ export const YOUTUBE_TIMESTAMPS_HEADING = "⏱️ Timestamps:";
342
+
343
+ /** A line that is nothing but hashtags — the description's prompted last
344
+ * line, which the timestamp splice must stay above. */
345
+ function isHashtagLine(line: string): boolean {
346
+ const tokens = line.trim().split(/\s+/);
347
+ return tokens.length > 0 && tokens.every((t) => t.startsWith("#") && t.length > 1);
348
+ }
349
+
350
+ /**
351
+ * Splice the measured chapter list INTO the description text — YouTube
352
+ * parses chapters FROM the description, so a separate list beside it would
353
+ * make the user hand-merge two blocks the file exists to spare them. The
354
+ * block lands after the body but ABOVE a trailing hashtag line when the
355
+ * description has one (the prompt asks for hashtags last, and search treats
356
+ * that placement specially). Pure and exported so the above/below matrix is
357
+ * testable on its own.
358
+ */
359
+ export function spliceTimestampsIntoDescription(
360
+ description: string,
361
+ chapters: readonly YoutubeChapter[],
362
+ ): string {
363
+ const body = description.trimEnd();
364
+ if (chapters.length === 0) return body;
365
+ const block = [
366
+ YOUTUBE_TIMESTAMPS_HEADING,
367
+ ...chapters.map((c) => `${chapterStamp(c.atSec)} ${c.title}`),
368
+ ].join("\n");
369
+ const lines = body.split("\n");
370
+ const last = lines[lines.length - 1] ?? "";
371
+ if (lines.length > 1 && isHashtagLine(last)) {
372
+ const head = lines.slice(0, -1).join("\n").trimEnd();
373
+ return `${head}\n\n${block}\n\n${last}`;
374
+ }
375
+ return `${body}\n\n${block}`;
376
+ }
377
+
378
+ /** How a titleAngle spells inside the markdown's numbered list. */
379
+ const ANGLE_LABELS: Record<TitleAngle, string> = {
380
+ browse: "Browse",
381
+ search: "Search",
382
+ benefit: "Benefit",
383
+ alt: "Alt",
384
+ };
385
+
386
+ /**
387
+ * The pack as the markdown file a user pastes from. Pure — the produce
388
+ * orchestration owns the write.
389
+ *
390
+ * Chapters render INSIDE the description (spliceTimestampsIntoDescription),
391
+ * not as their own section: the description is the one paste target, and a
392
+ * separate `## Chapters` block was a second copy of the same lines. The
393
+ * optional prompt-v2 sections (hook strategy, LinkedIn, community) render
394
+ * only when the pack carries them, so a pre-v2 pack formats exactly as it
395
+ * always did.
396
+ *
397
+ * `chaptersFromSec` rebases chapter stamps by subtracting it (dropping any
398
+ * chapter that would land before 0), for a caller whose chapters are on some
399
+ * other clock than the output's. Default 0 — stamps pass through.
400
+ */
401
+ export function formatYoutubeMarkdown(
402
+ pack: YoutubePack,
403
+ opts: { chaptersFromSec?: number } = {},
404
+ ): string {
405
+ const lines: string[] = ["# YouTube pack", "", "## Title options", ""];
406
+ pack.titles.forEach((t, i) => {
407
+ // Labels only when the pack carries angles (prompt v2); a mismatched or
408
+ // missing entry renders the bare title — the label is a hint, not data
409
+ // the paste depends on.
410
+ const angle = pack.titleAngles?.[i];
411
+ lines.push(`${i + 1}. ${angle !== undefined ? `[${ANGLE_LABELS[angle]}] ` : ""}${t}`);
412
+ });
413
+ const from = opts.chaptersFromSec ?? 0;
414
+ const chapters = (pack.chapters ?? [])
415
+ .filter((c) => c.atSec - from >= 0)
416
+ .map((c) => ({ ...c, atSec: c.atSec - from }));
417
+ lines.push(
418
+ "",
419
+ "## Description",
420
+ "",
421
+ spliceTimestampsIntoDescription(pack.description, chapters),
422
+ "",
423
+ "## Hashtags",
424
+ "",
425
+ );
426
+ // Paste-ready: one line, every entry carrying its # exactly once whether or
427
+ // not the model already spelled it.
428
+ lines.push(pack.hashtags.map((h) => (h.startsWith("#") ? h : `#${h}`)).join(" "));
429
+ lines.push("", "## Tags (comma-separated)", "", pack.tags.join(", "));
430
+ if (pack.hook60) lines.push("", "## First-60s hook strategy", "", pack.hook60.trimEnd());
431
+ if (pack.linkedinPost) lines.push("", "## LinkedIn post", "", pack.linkedinPost.trimEnd());
432
+ if (pack.communityPost) lines.push("", "## Community post", "", pack.communityPost.trimEnd());
433
+ return `${lines.join("\n")}\n`;
434
+ }
package/src/retake.ts CHANGED
@@ -20,6 +20,21 @@ import type { Analysis, Span, Transcript } from "./schema";
20
20
  * (edit distance over the token stream, not the letters) rather than a single
21
21
  * fuzzy-string threshold: it tolerates the fuzzy word or two `soundsSimilar`
22
22
  * chased, without opening the same phonetic false-positive channel.
23
+ *
24
+ * The exact-prefix restart pass (2026-08-16 field case) deliberately NARROWS
25
+ * a §128 protection rather than repealing it. §128 (PHASE1-FINDINGS.md,
26
+ * "128. Retake collapse without a spoken marker") records the verified repro
27
+ * where "Let me show you this." scored a spurious 1.0 against its own longer
28
+ * continuation and the LONGER, more complete side got cut — which is why
29
+ * complete-vs-complete comparison is full-sequence and a shorter complete
30
+ * sentence never prefix-matches forward. The field flub this pass exists for
31
+ * is the opposite shape: "You can use OpenAI." → [pause] → "You can use
32
+ * OpenAI models and whatnot." — ASR punctuated the abandoned attempt as
33
+ * complete, so the protection above (correctly) refuses the match, and the
34
+ * flub ships. The pass cuts ONLY the strictly-shorter EARLIER sentence, only
35
+ * on an EXACT token prefix, and only with a real pause at the boundary —
36
+ * three conditions the §128 incident's cut (of the longer side, on a fuzzy
37
+ * score, with no pause requirement) would fail.
23
38
  */
24
39
 
25
40
  /** Sequence similarity floor for two attempts to count as the same line. */
@@ -58,6 +73,25 @@ export const HALLUCINATION_SILENCE_FRAC = 0.65;
58
73
  * be at this boundary, so the same threshold gates both: not tuned twice.
59
74
  */
60
75
  export const RESTART_SPLIT_MIN_SIL = 0.35;
76
+ /**
77
+ * How far INTO the next sentence's stamped words the exact-prefix pass looks
78
+ * for the restart pause. Stamp-stretch physics again (§18): the pause between
79
+ * the abandoned "You can use OpenAI." and its restart was the silence
80
+ * 201.52–201.93 — INSIDE the stamped span of the restart sentence
81
+ * (201.0–203.08), not in the inter-sentence gap, because whisper stretched
82
+ * the restart's opening words back over the dead air. So the window is
83
+ * [A.endSec, B.startSec + this], measured with `silenceOverlap` against real
84
+ * silence spans, never against word-stamp gaps (which are ~always zero).
85
+ */
86
+ export const PREFIX_RESTART_SIL_WINDOW = 1.0;
87
+ /**
88
+ * Confidence `produce.ts` assigns an exact-prefix restart cut in the
89
+ * cutlist: below the 0.9 an ordinary similarity-matched retake earns,
90
+ * because the evidence is structural (prefix + pause) rather than a scored
91
+ * near-identical repeat — the report and any confidence-gated behavior can
92
+ * tell the two rules apart.
93
+ */
94
+ export const RESTART_PREFIX_CONFIDENCE = 0.85;
61
95
 
62
96
  /** A span of transcript, in word indices (inclusive) and source seconds. */
63
97
  export interface RetakeInstance {
@@ -119,6 +153,13 @@ export interface RetakeGroup {
119
153
  cuts: RetakeCut[];
120
154
  hallucinated: RetakeHallucination[];
121
155
  undecided: RetakeUndecided[];
156
+ /**
157
+ * Which pass decided this group, when it was not the ordinary similarity
158
+ * chain: "exact-prefix" is the restart pass (module doc) — the report
159
+ * prints the structural evidence instead of a match percentage, because
160
+ * "100% match" would misdescribe a rule that never scored anything.
161
+ */
162
+ rule?: "exact-prefix";
122
163
  }
123
164
 
124
165
  // ---- token comparison -------------------------------------------------
@@ -577,9 +618,63 @@ export function findRetakeGroups(
577
618
  }
578
619
  if (overlaps) continue;
579
620
  groups.push(g);
621
+ // The sentence pass's members join the claimed set too: the exact-prefix
622
+ // pass below must not re-decide words either earlier pass already
623
+ // decided — including spares, so a declined cut stays declined.
624
+ for (const m of members) {
625
+ for (let w = m.startWord; w <= m.endWord; w++) claimed.add(w);
626
+ }
627
+ }
628
+
629
+ // Exact-prefix restart pass (2026-08-16 field case; module doc has the
630
+ // §128-narrowing rationale). The flub: sentence A is an EXACT token prefix
631
+ // of the immediately following sentence B, strictly shorter, with a real
632
+ // pause (>= RESTART_SPLIT_MIN_SIL of silence) at the restart boundary. ASR
633
+ // punctuated A as complete ("You can use OpenAI."), so the similarity
634
+ // passes above score A-vs-B honestly divergent (1 - 3/7 ≈ 0.57) and
635
+ // correctly decline — completeness is no defense here, which is why A may
636
+ // be complete. Adjacency skips zero-token sentences, the same
637
+ // filler/marker transparency the chaining rule grants.
638
+ const live = sentences.filter((s) => s.tokens.length > 0);
639
+ for (let k = 0; k + 1 < live.length; k++) {
640
+ const a = live[k]!;
641
+ const b = live[k + 1]!;
642
+ if (a.hallucinated || b.hallucinated) continue;
643
+ if (a.tokens.length < RETAKE_MIN_TOKENS) continue;
644
+ if (a.tokens.length >= b.tokens.length) continue;
645
+ if (!a.tokens.every((tok, j) => tokensEqual(tok, b.tokens[j]!))) continue;
646
+ // The pause: ONE silence overlapping [A.end, B.start + window] by at
647
+ // least RESTART_SPLIT_MIN_SIL — see PREFIX_RESTART_SIL_WINDOW for why
648
+ // the window reaches INTO B's stamped words and why word gaps are
649
+ // useless here. Per-silence, not summed: three 0.1s breath gaps across
650
+ // the window are ordinary delivery, not a restart pause.
651
+ const windowEnd = b.startSec + PREFIX_RESTART_SIL_WINDOW;
652
+ const paused = analysis.silences.some(
653
+ (s) => silenceOverlap([s], a.endSec, windowEnd) >= RESTART_SPLIT_MIN_SIL,
654
+ );
655
+ if (!paused) continue;
656
+ // Claimed dedupe, same set as the sentence pass: a word either earlier
657
+ // pass already cut, kept OR deliberately spared is not re-decidable.
658
+ let overlaps = false;
659
+ for (const m of [a, b]) {
660
+ for (let w = m.startWord; w <= m.endWord && !overlaps; w++) {
661
+ if (claimed.has(w)) overlaps = true;
662
+ }
663
+ }
664
+ if (overlaps) continue;
665
+ groups.push({
666
+ kept: toPublic(b),
667
+ cuts: [{ ...toPublic(a), similarity: prefixSimilarity(a.tokens, b.tokens) }],
668
+ hallucinated: [],
669
+ undecided: [],
670
+ rule: "exact-prefix",
671
+ });
672
+ for (const m of [a, b]) {
673
+ for (let w = m.startWord; w <= m.endWord; w++) claimed.add(w);
674
+ }
580
675
  }
581
676
 
582
- // Two passes can interleave in transcript order — the report reads in
677
+ // The passes can interleave in transcript order — the report reads in
583
678
  // word order, whichever pass found the group.
584
679
  groups.sort((a, b) => {
585
680
  const first = (g: RetakeGroup): number =>
@@ -667,7 +762,14 @@ export function formatRetakeGroup(transcript: Transcript, group: RetakeGroup): s
667
762
  } else {
668
763
  lines.push(`kept: "${said(group.kept)}"`);
669
764
  for (const c of group.cuts) {
670
- lines.push(` cut (${Math.round(c.similarity * 100)}% match): "${said(c)}"`);
765
+ // An exact-prefix cut was never scored — printing a percentage would
766
+ // claim evidence the rule doesn't have. Name the structural evidence
767
+ // (prefix + pause) instead.
768
+ lines.push(
769
+ group.rule === "exact-prefix"
770
+ ? ` cut (exact-prefix restart, ≥0.35s pause): "${said(c)}"`
771
+ : ` cut (${Math.round(c.similarity * 100)}% match): "${said(c)}"`,
772
+ );
671
773
  }
672
774
  for (const u of group.undecided) {
673
775
  const pct = Math.round((u.similarity ?? 0) * 100);
@@ -186,6 +186,16 @@ export const FaceCropSchema = z.object({
186
186
  * phone footage, wrong for a webcam recording or a screen capture.
187
187
  */
188
188
  sourceAspect: z.number().positive().optional(),
189
+ /**
190
+ * What the measured face IS: the frame's subject, or an incidental face in
191
+ * a frame that is really about something else (2026-08-16 incident: a
192
+ * screen recording's camera PiP, 12% of frame height at the bottom-right,
193
+ * dragged the cover crop to the frame bottom and decapitated the speaker
194
+ * everywhere they appeared full-frame). "screen" tells the stage to center
195
+ * the cover instead of biasing toward the face. Absent means "face" — every
196
+ * pre-existing render-props keeps its old behavior.
197
+ */
198
+ subject: z.enum(["face", "screen"]).optional(),
189
199
  });
190
200
  export type FaceCrop = z.infer<typeof FaceCropSchema>;
191
201