@ossclip/core 0.1.24 → 0.1.26
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/fonts/NotoNastaliqUrdu-Bold.ttf +0 -0
- package/assets/fonts/OFL.txt +93 -0
- package/assets/fonts/README.md +15 -0
- package/package.json +2 -1
- package/src/blooper.ts +91 -8
- package/src/browser.ts +8 -0
- package/src/captions.ts +45 -2
- package/src/concat.ts +95 -8
- package/src/config.ts +110 -0
- package/src/content-rect-detect.ts +16 -4
- package/src/content-rect.ts +211 -0
- package/src/cover.ts +21 -5
- package/src/cutlist.ts +38 -6
- package/src/dictionary.ts +56 -0
- package/src/export-premiere-project.ts +26 -9
- package/src/fonts.ts +17 -0
- package/src/index.ts +3 -0
- package/src/ingest.ts +88 -2
- package/src/normalize.ts +273 -127
- package/src/overrides.ts +728 -0
- package/src/phonetics.ts +101 -1
- package/src/producer/index.ts +1 -0
- package/src/producer/repair.ts +55 -23
- package/src/producer/youtube.ts +434 -0
- package/src/recut.ts +61 -0
- package/src/retake.ts +104 -2
- package/src/scene-schema.ts +10 -0
- package/src/thumbnail.ts +412 -0
- package/src/transcribe.ts +93 -5
- package/src/zoom.ts +63 -12
|
@@ -0,0 +1,434 @@
|
|
|
1
|
+
import { z } from "zod/v4";
|
|
2
|
+
import type { Word } from "../schema";
|
|
3
|
+
import type { TimeMap } from "../timemap";
|
|
4
|
+
import { isSentenceEnd } from "../clip";
|
|
5
|
+
import type { LlmProvider } from "./provider";
|
|
6
|
+
import { cappedText } from "./beats";
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* The `--youtube` SEO pack (Y2, 2026-08-16; prompt v2 2026-08-17): title
|
|
10
|
+
* options, a description with measured chapter timestamps spliced in,
|
|
11
|
+
* hashtags and tags, written beside the video as `<out>.youtube.md` so the
|
|
12
|
+
* upload form is a paste job instead of a second writing session.
|
|
13
|
+
*
|
|
14
|
+
* Free text is `cappedText`, not `.max()` — the §112 doctrine as applied in
|
|
15
|
+
* beats.ts: LLM output is untrusted input, validated where the pipeline can
|
|
16
|
+
* still degrade instead of at the point where it can only die. A title one
|
|
17
|
+
* word over budget must cost a word, never the run.
|
|
18
|
+
*
|
|
19
|
+
* Two sections a human YouTube-strategist workflow would include are
|
|
20
|
+
* deliberately NOT here (prompt v2 decision):
|
|
21
|
+
* - competitor analysis — the provider has no browsing, so "competitor
|
|
22
|
+
* research" would be confident hallucination dressed up as strategy;
|
|
23
|
+
* - a thumbnail prompt — ossclip GENERATES the actual thumbnail from the
|
|
24
|
+
* real portrait (thumbnail.ts), which beats prose describing an imaginary
|
|
25
|
+
* one.
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Bump whenever buildYoutubePrompt changes what it asks for: a prompt change
|
|
30
|
+
* changes the answer, so the Y2 pack cache key carries this (the §78
|
|
31
|
+
* cache-key posture) — an old cached pack must not survive a new prompt.
|
|
32
|
+
*/
|
|
33
|
+
export const YOUTUBE_PROMPT_VERSION = "v2";
|
|
34
|
+
|
|
35
|
+
export const YoutubeChapterSchema = z.object({
|
|
36
|
+
/** Output-timeline seconds — the produced video's clock, not the source's. */
|
|
37
|
+
atSec: z.number().nonnegative(),
|
|
38
|
+
title: cappedText(80),
|
|
39
|
+
});
|
|
40
|
+
export type YoutubeChapter = z.infer<typeof YoutubeChapterSchema>;
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* The three labeled title angles for YouTube's Test & Compare A/B feature
|
|
44
|
+
* (prompt v2): browse = story/curiosity for home-page CTR, search =
|
|
45
|
+
* keyword/tool-name for evergreen search, benefit = the ROI claim. "alt"
|
|
46
|
+
* covers any extra option past the three.
|
|
47
|
+
*/
|
|
48
|
+
export const TitleAngleSchema = z.enum(["browse", "search", "benefit", "alt"]);
|
|
49
|
+
export type TitleAngle = z.infer<typeof TitleAngleSchema>;
|
|
50
|
+
|
|
51
|
+
export const YoutubePackSchema = z.object({
|
|
52
|
+
/** Distinct angles on the same video, not five rewordings of one. */
|
|
53
|
+
titles: z.array(cappedText(100)).min(3).max(5),
|
|
54
|
+
/**
|
|
55
|
+
* Parallel to `titles`, one angle label each. OPTIONAL for back-compat:
|
|
56
|
+
* packs approved before prompt v2 carry bare titles, and the approved-file
|
|
57
|
+
* contract (YOUTUBE_APPROVED_BASENAME) means those files must keep parsing
|
|
58
|
+
* verbatim forever.
|
|
59
|
+
*/
|
|
60
|
+
titleAngles: z.array(TitleAngleSchema).optional(),
|
|
61
|
+
/** Prompted for keyword-first-lines, hashtags-at-end; free-form otherwise.
|
|
62
|
+
* Chapter timestamps are NOT the model's to write — the formatter splices
|
|
63
|
+
* the measured ones in (spliceTimestampsIntoDescription). */
|
|
64
|
+
description: z.string(),
|
|
65
|
+
hashtags: z.array(z.string()),
|
|
66
|
+
tags: z.array(z.string()),
|
|
67
|
+
chapters: z.array(YoutubeChapterSchema).optional(),
|
|
68
|
+
/** Advisory first-60-seconds retention note for the CREATOR — never pasted
|
|
69
|
+
* into YouTube. Optional: pre-v2 packs never carried it. */
|
|
70
|
+
hook60: cappedText(400).optional(),
|
|
71
|
+
/** A ready-to-post LinkedIn announcement. Optional, pre-v2 back-compat. */
|
|
72
|
+
linkedinPost: cappedText(1500).optional(),
|
|
73
|
+
/** A short YouTube community post for existing subscribers. Optional. */
|
|
74
|
+
communityPost: cappedText(400).optional(),
|
|
75
|
+
});
|
|
76
|
+
export type YoutubePack = z.infer<typeof YoutubePackSchema>;
|
|
77
|
+
|
|
78
|
+
/** YouTube's hard cap on the tags field, counted over the comma-joined text. */
|
|
79
|
+
export const YOUTUBE_TAGS_LIMIT = 500;
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Post-parse guard for the tags field: drop tags from the END until the
|
|
83
|
+
* comma-joined line fits YouTube's limit. From the end because the prompt
|
|
84
|
+
* asks for relevance order — the least relevant tag is the one a budget cut
|
|
85
|
+
* should cost. Degrade-don't-die (§112): the schema cannot express a
|
|
86
|
+
* joined-length cap, and refusing the whole pack over an over-enthusiastic
|
|
87
|
+
* tag list would discard the titles and description that were fine.
|
|
88
|
+
*/
|
|
89
|
+
export function trimTagsToLimit(tags: string[], limit: number = YOUTUBE_TAGS_LIMIT): string[] {
|
|
90
|
+
const kept = [...tags];
|
|
91
|
+
while (kept.length > 0 && kept.join(", ").length > limit) {
|
|
92
|
+
kept.pop();
|
|
93
|
+
}
|
|
94
|
+
return kept;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* The workdir file the editor's SEO panel writes (2026-08-17), mirroring the
|
|
99
|
+
* thumbnail approval contract (THUMBNAIL_APPROVED_BASENAME, thumbnail.ts):
|
|
100
|
+
* once it exists, produce's Y2 block uses the pack VERBATIM — no cache
|
|
101
|
+
* lookup, no LLM call — because an edited pack is the user's decision, and a
|
|
102
|
+
* fresh generation would silently discard it. Holds a plain `YoutubePack`,
|
|
103
|
+
* no skip variant: the `--youtube` flag itself gates the feature, so there
|
|
104
|
+
* is no "opted in then declined" state for a pack the way there is for a
|
|
105
|
+
* thumbnail. Deleting the file is the regenerate gesture.
|
|
106
|
+
*/
|
|
107
|
+
export const YOUTUBE_APPROVED_BASENAME = "youtube-pack-approved.json";
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* How much transcript the prompt carries. Field-measured 2026-08-17: at the
|
|
111
|
+
* old 10k cap a 21-minute video's chapters STOPPED AT 8:32 — the model can
|
|
112
|
+
* only chapter what it reads, and a truncated transcript silently produces
|
|
113
|
+
* chapters for the excerpt while looking complete. 60k chars ≈ 15k tokens
|
|
114
|
+
* covers ~45 minutes of stamped talk for pennies; past that the truncation
|
|
115
|
+
* note at least tells the model (and the chapters) where the coverage ends.
|
|
116
|
+
* `stampedTranscript` pre-caps to this same constant, so
|
|
117
|
+
* `buildYoutubePrompt`'s own cap is a backstop for raw-text callers, never a
|
|
118
|
+
* second cut of stamped input.
|
|
119
|
+
*/
|
|
120
|
+
export const YOUTUBE_TRANSCRIPT_CHAR_CAP = 60_000;
|
|
121
|
+
|
|
122
|
+
/** 125 → "2:05" — YouTube's own chapter-list spelling. */
|
|
123
|
+
function chapterStamp(sec: number): string {
|
|
124
|
+
const whole = Math.max(0, Math.floor(sec));
|
|
125
|
+
const m = Math.floor(whole / 60);
|
|
126
|
+
const s = whole % 60;
|
|
127
|
+
return `${m}:${String(s).padStart(2, "0")}`;
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/** What a truncated stamped transcript ends with — the model must know it is
|
|
131
|
+
* reading an excerpt, or it will write a description that promises only the
|
|
132
|
+
* first half. */
|
|
133
|
+
const TRUNCATION_NOTE = "[transcript truncated — the video continues]";
|
|
134
|
+
|
|
135
|
+
/**
|
|
136
|
+
* The transcript as the prompt's chapter evidence (prompt v2): one line per
|
|
137
|
+
* sentence, prefixed with the sentence's first surviving word's OUTPUT time
|
|
138
|
+
* as `[m:ss]`. This is the pack's unique advantage over paste-a-transcript
|
|
139
|
+
* prompt tools — words carry SOURCE times and the TimeMap converts them to
|
|
140
|
+
* the produced video's clock, so chapters can be measured instead of the
|
|
141
|
+
* guessed "0:00 Intro" everyone else ships.
|
|
142
|
+
*
|
|
143
|
+
* Fully-cut sentences (every word `mapWord` → null) are skipped: they are
|
|
144
|
+
* not in the video, and stamping them would hand the model timestamps that
|
|
145
|
+
* point at content the viewer never sees. Pure — words + map in, string
|
|
146
|
+
* out — so the mapping, the cut-skip and the cap are testable without an
|
|
147
|
+
* LLM or a filesystem.
|
|
148
|
+
*
|
|
149
|
+
* The cap reserves room for TRUNCATION_NOTE up front, so the result NEVER
|
|
150
|
+
* exceeds `maxChars` — truncated or not — and whole sentences are dropped
|
|
151
|
+
* rather than sliced mid-line, which would leave a dangling stamp.
|
|
152
|
+
*/
|
|
153
|
+
export function stampedTranscript(
|
|
154
|
+
words: readonly Word[],
|
|
155
|
+
map: TimeMap,
|
|
156
|
+
opts: { maxChars?: number } = {},
|
|
157
|
+
): string {
|
|
158
|
+
const maxChars = opts.maxChars ?? YOUTUBE_TRANSCRIPT_CHAR_CAP;
|
|
159
|
+
// isSentenceEnd reads punctuation off Transcript-shaped input.
|
|
160
|
+
const t = { language: "en", words: [...words] };
|
|
161
|
+
const budget = maxChars - TRUNCATION_NOTE.length - 1;
|
|
162
|
+
const lines: string[] = [];
|
|
163
|
+
let used = 0;
|
|
164
|
+
let truncated = false;
|
|
165
|
+
let start = 0;
|
|
166
|
+
for (let i = 0; i < words.length; i++) {
|
|
167
|
+
if (i < words.length - 1 && !isSentenceEnd(t, i)) continue;
|
|
168
|
+
const sentence = words.slice(start, i + 1);
|
|
169
|
+
start = i + 1;
|
|
170
|
+
// The stamp is the first SURVIVING word's output time, not the first
|
|
171
|
+
// word's: a sentence whose opening filler was cut starts, on the
|
|
172
|
+
// viewer's clock, at the first word they hear.
|
|
173
|
+
let stampSec: number | null = null;
|
|
174
|
+
for (const w of sentence) {
|
|
175
|
+
const mapped = map.mapWord(w);
|
|
176
|
+
if (mapped) {
|
|
177
|
+
stampSec = mapped.start;
|
|
178
|
+
break;
|
|
179
|
+
}
|
|
180
|
+
}
|
|
181
|
+
if (stampSec === null) continue; // the whole sentence was cut
|
|
182
|
+
const line = `[${chapterStamp(stampSec)}] ${sentence.map((w) => w.text).join(" ")}`;
|
|
183
|
+
const cost = lines.length === 0 ? line.length : line.length + 1;
|
|
184
|
+
if (used + cost > budget) {
|
|
185
|
+
truncated = true;
|
|
186
|
+
break;
|
|
187
|
+
}
|
|
188
|
+
lines.push(line);
|
|
189
|
+
used += cost;
|
|
190
|
+
}
|
|
191
|
+
if (truncated) lines.push(TRUNCATION_NOTE);
|
|
192
|
+
return lines.join("\n");
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/** YouTube ignores a chapter list whose entries run shorter than this. */
|
|
196
|
+
export const YOUTUBE_CHAPTER_MIN_GAP_SEC = 10;
|
|
197
|
+
/** ...or that carries fewer timestamps than this. */
|
|
198
|
+
export const YOUTUBE_CHAPTER_MIN_COUNT = 3;
|
|
199
|
+
|
|
200
|
+
/**
|
|
201
|
+
* Post-parse guard for the chapters field — trimTagsToLimit's posture
|
|
202
|
+
* applied to YouTube's chapter rules, which the schema cannot express:
|
|
203
|
+
* the list must start at 0:00, hold ≥3 entries, and give every chapter
|
|
204
|
+
* ≥10 seconds, or YouTube silently ignores the WHOLE list. Degrade-don't-die
|
|
205
|
+
* (§112): repair what can be repaired, drop what cannot, and return [] —
|
|
206
|
+
* "no chapters", which the formatter renders as no timestamp block — rather
|
|
207
|
+
* than ship a list YouTube would reject wholesale anyway.
|
|
208
|
+
*
|
|
209
|
+
* - unsorted input sorts ascending (the model's list order is untrusted);
|
|
210
|
+
* - chapters at or past the video's end are dropped — a stamp the video
|
|
211
|
+
* never reaches is a hallucinated one;
|
|
212
|
+
* - a first chapter within the 10s minimum of 0 snaps TO 0 (it IS the
|
|
213
|
+
* intro, mis-stamped); one further out gets a synthetic 0:00 "Intro"
|
|
214
|
+
* prepended — the unnamed opening is real content, and 0:00 is mandatory;
|
|
215
|
+
* - of a pair closer than 10s, the LATER one drops (the earlier stamp is
|
|
216
|
+
* the section's true start).
|
|
217
|
+
*/
|
|
218
|
+
export function normalizeChapters(
|
|
219
|
+
chapters: readonly YoutubeChapter[],
|
|
220
|
+
durationSec: number,
|
|
221
|
+
): YoutubeChapter[] {
|
|
222
|
+
const sorted = [...chapters]
|
|
223
|
+
.sort((a, b) => a.atSec - b.atSec)
|
|
224
|
+
.filter((c) => c.atSec < durationSec);
|
|
225
|
+
if (sorted.length === 0) return [];
|
|
226
|
+
const first = sorted[0]!;
|
|
227
|
+
const list: YoutubeChapter[] =
|
|
228
|
+
first.atSec === 0
|
|
229
|
+
? sorted
|
|
230
|
+
: first.atSec < YOUTUBE_CHAPTER_MIN_GAP_SEC
|
|
231
|
+
? [{ ...first, atSec: 0 }, ...sorted.slice(1)]
|
|
232
|
+
: [{ atSec: 0, title: "Intro" }, ...sorted];
|
|
233
|
+
const kept: YoutubeChapter[] = [];
|
|
234
|
+
for (const c of list) {
|
|
235
|
+
const prev = kept[kept.length - 1];
|
|
236
|
+
if (prev === undefined || c.atSec - prev.atSec >= YOUTUBE_CHAPTER_MIN_GAP_SEC) kept.push(c);
|
|
237
|
+
}
|
|
238
|
+
return kept.length >= YOUTUBE_CHAPTER_MIN_COUNT ? kept : [];
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
export interface YoutubePromptArgs {
|
|
242
|
+
/** The stamped transcript (stampedTranscript) — sentences prefixed with
|
|
243
|
+
* their `[m:ss]` output-clock time. Plain text still works (raw-text
|
|
244
|
+
* callers just get no measured chapters worth trusting). */
|
|
245
|
+
transcriptText: string;
|
|
246
|
+
/** `--intent`, when the run had one. */
|
|
247
|
+
intent?: string;
|
|
248
|
+
/** The producer's hook, when a beat sheet exists — the strongest claim. */
|
|
249
|
+
hook?: string;
|
|
250
|
+
/** The cover banner text, when a beat sheet wrote one. */
|
|
251
|
+
coverText?: string;
|
|
252
|
+
/**
|
|
253
|
+
* Who watches the channel (`--audience` / config `audience`, 2026-08-16):
|
|
254
|
+
* titles and tags for "junior devs learning agents" are a different pack
|
|
255
|
+
* than for "engineering managers", and only the user knows which one this
|
|
256
|
+
* channel is.
|
|
257
|
+
*/
|
|
258
|
+
audience?: string;
|
|
259
|
+
/** Output duration in seconds — the ceiling no chapter may pass. */
|
|
260
|
+
durationSec: number;
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
/**
|
|
264
|
+
* Pure prompt builder, separated from the provider call so the
|
|
265
|
+
* include/omit matrix (intent, hook, coverText, audience) and the transcript
|
|
266
|
+
* cap are testable without an LLM.
|
|
267
|
+
*/
|
|
268
|
+
export function buildYoutubePrompt(args: YoutubePromptArgs): { system: string; user: string } {
|
|
269
|
+
const system =
|
|
270
|
+
"You are an expert YouTube growth strategist writing upload metadata for a finished video. " +
|
|
271
|
+
"The transcript's sentences are each prefixed with a [m:ss] timestamp on the FINAL video's " +
|
|
272
|
+
"clock — measured from the actual edit, not estimated. Match the creator's tone from the " +
|
|
273
|
+
"transcript, and never make a claim the video does not deliver: click-drivers that deliver, " +
|
|
274
|
+
"not clickbait.\n" +
|
|
275
|
+
"- titles: 3-5 options for YouTube's Test & Compare A/B feature, each a DISTINCT angle, " +
|
|
276
|
+
"labeled in the parallel titleAngles array: \"browse\" — story/curiosity, zero jargon, " +
|
|
277
|
+
"before/after tension, built for home-page CTR; \"search\" — keyword-heavy, names the exact " +
|
|
278
|
+
"tools and technologies, built for evergreen search; \"benefit\" — the ROI, \"do X in N " +
|
|
279
|
+
"minutes\". Include at least one of each of those three; label any extra option \"alt\". " +
|
|
280
|
+
"Target at most 70 characters (search results truncate around 70, browse cards earlier).\n" +
|
|
281
|
+
"- description: the FIRST TWO lines are the search snippet — the primary keyword plus a " +
|
|
282
|
+
"curiosity hook. Then a body of short paragraphs or bullets carrying the secondary keywords, " +
|
|
283
|
+
"a call to action near the end, and 3-5 hashtags as the LAST line. Do NOT write timestamp " +
|
|
284
|
+
"or chapter lines in the description — the measured chapter timestamps are spliced in " +
|
|
285
|
+
"automatically.\n" +
|
|
286
|
+
"- hashtags: the same 3-5, as an array.\n" +
|
|
287
|
+
"- tags: search keywords and phrases a viewer would type, ordered by relevance — the most " +
|
|
288
|
+
"important first.\n" +
|
|
289
|
+
"- chapters: 4-10, USING ONLY the [m:ss] stamps that appear in the transcript — atSec is " +
|
|
290
|
+
"that stamp converted to seconds, never a time you estimated. The first chapter is at 0. " +
|
|
291
|
+
"Titles at most 50 characters, keyword-bearing but written for a human scanning the list.\n" +
|
|
292
|
+
"- hook60: a first-60-seconds retention note for the CREATOR (it is never published) — e.g. " +
|
|
293
|
+
"\"show the final result at 0:05, then rewind to how\" — grounded in what this transcript " +
|
|
294
|
+
"actually opens with.\n" +
|
|
295
|
+
"- linkedinPost: a story-driven LinkedIn post about this video: short lines with line " +
|
|
296
|
+
"breaks, a curiosity gap, no hashtag spam, ending by pointing to the link in the comments " +
|
|
297
|
+
"(the LinkedIn convention for off-platform links).\n" +
|
|
298
|
+
"- communityPost: a short, casual YouTube community post for existing subscribers.";
|
|
299
|
+
const capped =
|
|
300
|
+
args.transcriptText.length > YOUTUBE_TRANSCRIPT_CHAR_CAP
|
|
301
|
+
? // Slice + say so: the model must know it is reading an excerpt, or it
|
|
302
|
+
// will write a description that promises only the first half.
|
|
303
|
+
`${args.transcriptText.slice(0, YOUTUBE_TRANSCRIPT_CHAR_CAP)}\n${TRUNCATION_NOTE}`
|
|
304
|
+
: args.transcriptText;
|
|
305
|
+
const user =
|
|
306
|
+
(args.intent ? `Intent: ${args.intent}\n` : "") +
|
|
307
|
+
(args.hook ? `Hook (already chosen by the producer): ${args.hook}\n` : "") +
|
|
308
|
+
(args.coverText ? `Cover headline: ${args.coverText}\n` : "") +
|
|
309
|
+
(args.audience ? `Audience: ${args.audience}\n` : "") +
|
|
310
|
+
`Total runtime: ${chapterStamp(args.durationSec)} — chapters must not exceed it.\n\n` +
|
|
311
|
+
`Transcript:\n${capped}`;
|
|
312
|
+
return { system, user };
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
/** One editorial call → a validated, tag-trimmed, chapter-normalized pack. */
|
|
316
|
+
export async function generateYoutubePack(
|
|
317
|
+
provider: LlmProvider,
|
|
318
|
+
args: YoutubePromptArgs,
|
|
319
|
+
): Promise<YoutubePack> {
|
|
320
|
+
const { system, user } = buildYoutubePrompt(args);
|
|
321
|
+
const pack = await provider.complete({
|
|
322
|
+
system,
|
|
323
|
+
user,
|
|
324
|
+
schema: YoutubePackSchema,
|
|
325
|
+
schemaName: "youtube_pack",
|
|
326
|
+
tier: "editorial",
|
|
327
|
+
});
|
|
328
|
+
// Both post-parse guards live HERE, beside the call, so every path that
|
|
329
|
+
// stores the pack (the Y2 cache, the approved file) already holds a
|
|
330
|
+
// publishable one. An empty normalized list degrades to "no chapters"
|
|
331
|
+
// (undefined), which the formatter renders as no timestamp block.
|
|
332
|
+
const chapters = normalizeChapters(pack.chapters ?? [], args.durationSec);
|
|
333
|
+
return {
|
|
334
|
+
...pack,
|
|
335
|
+
tags: trimTagsToLimit(pack.tags),
|
|
336
|
+
chapters: chapters.length > 0 ? chapters : undefined,
|
|
337
|
+
};
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
/** The heading the timestamp block opens with inside the description. */
|
|
341
|
+
export const YOUTUBE_TIMESTAMPS_HEADING = "⏱️ Timestamps:";
|
|
342
|
+
|
|
343
|
+
/** A line that is nothing but hashtags — the description's prompted last
|
|
344
|
+
* line, which the timestamp splice must stay above. */
|
|
345
|
+
function isHashtagLine(line: string): boolean {
|
|
346
|
+
const tokens = line.trim().split(/\s+/);
|
|
347
|
+
return tokens.length > 0 && tokens.every((t) => t.startsWith("#") && t.length > 1);
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
/**
|
|
351
|
+
* Splice the measured chapter list INTO the description text — YouTube
|
|
352
|
+
* parses chapters FROM the description, so a separate list beside it would
|
|
353
|
+
* make the user hand-merge two blocks the file exists to spare them. The
|
|
354
|
+
* block lands after the body but ABOVE a trailing hashtag line when the
|
|
355
|
+
* description has one (the prompt asks for hashtags last, and search treats
|
|
356
|
+
* that placement specially). Pure and exported so the above/below matrix is
|
|
357
|
+
* testable on its own.
|
|
358
|
+
*/
|
|
359
|
+
export function spliceTimestampsIntoDescription(
|
|
360
|
+
description: string,
|
|
361
|
+
chapters: readonly YoutubeChapter[],
|
|
362
|
+
): string {
|
|
363
|
+
const body = description.trimEnd();
|
|
364
|
+
if (chapters.length === 0) return body;
|
|
365
|
+
const block = [
|
|
366
|
+
YOUTUBE_TIMESTAMPS_HEADING,
|
|
367
|
+
...chapters.map((c) => `${chapterStamp(c.atSec)} ${c.title}`),
|
|
368
|
+
].join("\n");
|
|
369
|
+
const lines = body.split("\n");
|
|
370
|
+
const last = lines[lines.length - 1] ?? "";
|
|
371
|
+
if (lines.length > 1 && isHashtagLine(last)) {
|
|
372
|
+
const head = lines.slice(0, -1).join("\n").trimEnd();
|
|
373
|
+
return `${head}\n\n${block}\n\n${last}`;
|
|
374
|
+
}
|
|
375
|
+
return `${body}\n\n${block}`;
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
/** How a titleAngle spells inside the markdown's numbered list. */
|
|
379
|
+
const ANGLE_LABELS: Record<TitleAngle, string> = {
|
|
380
|
+
browse: "Browse",
|
|
381
|
+
search: "Search",
|
|
382
|
+
benefit: "Benefit",
|
|
383
|
+
alt: "Alt",
|
|
384
|
+
};
|
|
385
|
+
|
|
386
|
+
/**
|
|
387
|
+
* The pack as the markdown file a user pastes from. Pure — the produce
|
|
388
|
+
* orchestration owns the write.
|
|
389
|
+
*
|
|
390
|
+
* Chapters render INSIDE the description (spliceTimestampsIntoDescription),
|
|
391
|
+
* not as their own section: the description is the one paste target, and a
|
|
392
|
+
* separate `## Chapters` block was a second copy of the same lines. The
|
|
393
|
+
* optional prompt-v2 sections (hook strategy, LinkedIn, community) render
|
|
394
|
+
* only when the pack carries them, so a pre-v2 pack formats exactly as it
|
|
395
|
+
* always did.
|
|
396
|
+
*
|
|
397
|
+
* `chaptersFromSec` rebases chapter stamps by subtracting it (dropping any
|
|
398
|
+
* chapter that would land before 0), for a caller whose chapters are on some
|
|
399
|
+
* other clock than the output's. Default 0 — stamps pass through.
|
|
400
|
+
*/
|
|
401
|
+
export function formatYoutubeMarkdown(
|
|
402
|
+
pack: YoutubePack,
|
|
403
|
+
opts: { chaptersFromSec?: number } = {},
|
|
404
|
+
): string {
|
|
405
|
+
const lines: string[] = ["# YouTube pack", "", "## Title options", ""];
|
|
406
|
+
pack.titles.forEach((t, i) => {
|
|
407
|
+
// Labels only when the pack carries angles (prompt v2); a mismatched or
|
|
408
|
+
// missing entry renders the bare title — the label is a hint, not data
|
|
409
|
+
// the paste depends on.
|
|
410
|
+
const angle = pack.titleAngles?.[i];
|
|
411
|
+
lines.push(`${i + 1}. ${angle !== undefined ? `[${ANGLE_LABELS[angle]}] ` : ""}${t}`);
|
|
412
|
+
});
|
|
413
|
+
const from = opts.chaptersFromSec ?? 0;
|
|
414
|
+
const chapters = (pack.chapters ?? [])
|
|
415
|
+
.filter((c) => c.atSec - from >= 0)
|
|
416
|
+
.map((c) => ({ ...c, atSec: c.atSec - from }));
|
|
417
|
+
lines.push(
|
|
418
|
+
"",
|
|
419
|
+
"## Description",
|
|
420
|
+
"",
|
|
421
|
+
spliceTimestampsIntoDescription(pack.description, chapters),
|
|
422
|
+
"",
|
|
423
|
+
"## Hashtags",
|
|
424
|
+
"",
|
|
425
|
+
);
|
|
426
|
+
// Paste-ready: one line, every entry carrying its # exactly once whether or
|
|
427
|
+
// not the model already spelled it.
|
|
428
|
+
lines.push(pack.hashtags.map((h) => (h.startsWith("#") ? h : `#${h}`)).join(" "));
|
|
429
|
+
lines.push("", "## Tags (comma-separated)", "", pack.tags.join(", "));
|
|
430
|
+
if (pack.hook60) lines.push("", "## First-60s hook strategy", "", pack.hook60.trimEnd());
|
|
431
|
+
if (pack.linkedinPost) lines.push("", "## LinkedIn post", "", pack.linkedinPost.trimEnd());
|
|
432
|
+
if (pack.communityPost) lines.push("", "## Community post", "", pack.communityPost.trimEnd());
|
|
433
|
+
return `${lines.join("\n")}\n`;
|
|
434
|
+
}
|
package/src/recut.ts
CHANGED
|
@@ -391,3 +391,64 @@ export function applyUserCuts(
|
|
|
391
391
|
removedSec,
|
|
392
392
|
};
|
|
393
393
|
}
|
|
394
|
+
|
|
395
|
+
/** What `pruneHidesInsideCuts` hands back: the doc (same reference when
|
|
396
|
+
* nothing was pruned — the caller's changed-gate reads `pruned.length`), and
|
|
397
|
+
* the retired keys so produce can SAY what it retired. */
|
|
398
|
+
export interface PrunedHides {
|
|
399
|
+
doc: OverrideDoc;
|
|
400
|
+
pruned: string[];
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
/**
|
|
404
|
+
* Retire `captionWordsHidden` entries whose word the final cutlist REMOVES
|
|
405
|
+
* (§59b revisited 2026-08-18 — the "captions + video" delete gesture writes
|
|
406
|
+
* both a hide and a cut in one commit).
|
|
407
|
+
*
|
|
408
|
+
* Once the cut lands, `buildCaptionLines` drops the word before the hide
|
|
409
|
+
* layer ever sees it, so the hide key would report `found: null` ("the cut
|
|
410
|
+
* removed it", `captionHideDropLine`) on every subsequent run forever — the
|
|
411
|
+
* cut SUPERSEDES the hide, the same superseded philosophy `overrides.ts`'s
|
|
412
|
+
* caption-key migration applies. Hides whose source instant is OUTSIDE every
|
|
413
|
+
* removed segment are kept verbatim — as are keys that are not §137 `w<ms>`
|
|
414
|
+
* anchors at all, which name no instant this can test (see the guard below).
|
|
415
|
+
*
|
|
416
|
+
* HALF-OPEN interval (`srcIn <= src < srcOut`), on purpose — the two edges
|
|
417
|
+
* are NOT symmetric. A word starting exactly at `srcIn` IS cut: that is
|
|
418
|
+
* precisely where the FIRST word of a captions+video delete lands (its
|
|
419
|
+
* srcStart round-trips through the TimeMap to the resolved cut's own srcIn),
|
|
420
|
+
* and `mapWord` clamps that instant into the removal and drops the word — a
|
|
421
|
+
* strictly-inside test never retired the gesture's own first hide, leaving
|
|
422
|
+
* it a permanent `found: null` drop report. A word starting exactly at
|
|
423
|
+
* `srcOut` belongs to the NEXT kept span (`buildCaptionLines` still emits it
|
|
424
|
+
* — a seam instant has a kept-side preimage, `timemap.ts`), so its hide is
|
|
425
|
+
* still doing work and must survive.
|
|
426
|
+
*/
|
|
427
|
+
export function pruneHidesInsideCuts(doc: OverrideDoc, cutlist: readonly Segment[]): PrunedHides {
|
|
428
|
+
const pruned: string[] = [];
|
|
429
|
+
const kept: OverrideDoc["captionWordsHidden"] = {};
|
|
430
|
+
for (const [key, entry] of Object.entries(doc.captionWordsHidden)) {
|
|
431
|
+
// SOURCE-KEYED ONLY, parsed and not coerced. `captionWordsHidden` is an
|
|
432
|
+
// unpinned `z.record` (unlike `CaptionRangeEditSchema`'s `/^w\d+$/`), so a
|
|
433
|
+
// hand-edited or legacy-keyed doc reaches here: a POSITIONAL key like "17"
|
|
434
|
+
// would slice to "7", parse as 7ms, land inside any early cut and be
|
|
435
|
+
// deleted with nothing said. The editor guards the identical case and
|
|
436
|
+
// states the rule (`apps/editor/src/useEdits.ts:587-600`): only §137
|
|
437
|
+
// `w<ms>` keys carry an interval-testable instant, and an entry this
|
|
438
|
+
// function cannot honestly locate is KEPT.
|
|
439
|
+
if (!/^w\d+$/.test(key)) {
|
|
440
|
+
kept[key] = entry;
|
|
441
|
+
continue;
|
|
442
|
+
}
|
|
443
|
+
// `captionKeyFor`'s quantization inverted (`w${Math.round(sec * 1000)}`,
|
|
444
|
+
// overrides.ts): the key IS the word's source instant, ms-quantized.
|
|
445
|
+
const srcSec = parseInt(key.slice(1), 10) / 1000;
|
|
446
|
+
const removed = cutlist.some(
|
|
447
|
+
(seg) => seg.kind === "remove" && srcSec >= seg.srcIn && srcSec < seg.srcOut,
|
|
448
|
+
);
|
|
449
|
+
if (removed) pruned.push(key);
|
|
450
|
+
else kept[key] = entry;
|
|
451
|
+
}
|
|
452
|
+
if (pruned.length === 0) return { doc, pruned };
|
|
453
|
+
return { doc: { ...doc, captionWordsHidden: kept }, pruned };
|
|
454
|
+
}
|
package/src/retake.ts
CHANGED
|
@@ -20,6 +20,21 @@ import type { Analysis, Span, Transcript } from "./schema";
|
|
|
20
20
|
* (edit distance over the token stream, not the letters) rather than a single
|
|
21
21
|
* fuzzy-string threshold: it tolerates the fuzzy word or two `soundsSimilar`
|
|
22
22
|
* chased, without opening the same phonetic false-positive channel.
|
|
23
|
+
*
|
|
24
|
+
* The exact-prefix restart pass (2026-08-16 field case) deliberately NARROWS
|
|
25
|
+
* a §128 protection rather than repealing it. §128 (PHASE1-FINDINGS.md,
|
|
26
|
+
* "128. Retake collapse without a spoken marker") records the verified repro
|
|
27
|
+
* where "Let me show you this." scored a spurious 1.0 against its own longer
|
|
28
|
+
* continuation and the LONGER, more complete side got cut — which is why
|
|
29
|
+
* complete-vs-complete comparison is full-sequence and a shorter complete
|
|
30
|
+
* sentence never prefix-matches forward. The field flub this pass exists for
|
|
31
|
+
* is the opposite shape: "You can use OpenAI." → [pause] → "You can use
|
|
32
|
+
* OpenAI models and whatnot." — ASR punctuated the abandoned attempt as
|
|
33
|
+
* complete, so the protection above (correctly) refuses the match, and the
|
|
34
|
+
* flub ships. The pass cuts ONLY the strictly-shorter EARLIER sentence, only
|
|
35
|
+
* on an EXACT token prefix, and only with a real pause at the boundary —
|
|
36
|
+
* three conditions the §128 incident's cut (of the longer side, on a fuzzy
|
|
37
|
+
* score, with no pause requirement) would fail.
|
|
23
38
|
*/
|
|
24
39
|
|
|
25
40
|
/** Sequence similarity floor for two attempts to count as the same line. */
|
|
@@ -58,6 +73,25 @@ export const HALLUCINATION_SILENCE_FRAC = 0.65;
|
|
|
58
73
|
* be at this boundary, so the same threshold gates both: not tuned twice.
|
|
59
74
|
*/
|
|
60
75
|
export const RESTART_SPLIT_MIN_SIL = 0.35;
|
|
76
|
+
/**
|
|
77
|
+
* How far INTO the next sentence's stamped words the exact-prefix pass looks
|
|
78
|
+
* for the restart pause. Stamp-stretch physics again (§18): the pause between
|
|
79
|
+
* the abandoned "You can use OpenAI." and its restart was the silence
|
|
80
|
+
* 201.52–201.93 — INSIDE the stamped span of the restart sentence
|
|
81
|
+
* (201.0–203.08), not in the inter-sentence gap, because whisper stretched
|
|
82
|
+
* the restart's opening words back over the dead air. So the window is
|
|
83
|
+
* [A.endSec, B.startSec + this], measured with `silenceOverlap` against real
|
|
84
|
+
* silence spans, never against word-stamp gaps (which are ~always zero).
|
|
85
|
+
*/
|
|
86
|
+
export const PREFIX_RESTART_SIL_WINDOW = 1.0;
|
|
87
|
+
/**
|
|
88
|
+
* Confidence `produce.ts` assigns an exact-prefix restart cut in the
|
|
89
|
+
* cutlist: below the 0.9 an ordinary similarity-matched retake earns,
|
|
90
|
+
* because the evidence is structural (prefix + pause) rather than a scored
|
|
91
|
+
* near-identical repeat — the report and any confidence-gated behavior can
|
|
92
|
+
* tell the two rules apart.
|
|
93
|
+
*/
|
|
94
|
+
export const RESTART_PREFIX_CONFIDENCE = 0.85;
|
|
61
95
|
|
|
62
96
|
/** A span of transcript, in word indices (inclusive) and source seconds. */
|
|
63
97
|
export interface RetakeInstance {
|
|
@@ -119,6 +153,13 @@ export interface RetakeGroup {
|
|
|
119
153
|
cuts: RetakeCut[];
|
|
120
154
|
hallucinated: RetakeHallucination[];
|
|
121
155
|
undecided: RetakeUndecided[];
|
|
156
|
+
/**
|
|
157
|
+
* Which pass decided this group, when it was not the ordinary similarity
|
|
158
|
+
* chain: "exact-prefix" is the restart pass (module doc) — the report
|
|
159
|
+
* prints the structural evidence instead of a match percentage, because
|
|
160
|
+
* "100% match" would misdescribe a rule that never scored anything.
|
|
161
|
+
*/
|
|
162
|
+
rule?: "exact-prefix";
|
|
122
163
|
}
|
|
123
164
|
|
|
124
165
|
// ---- token comparison -------------------------------------------------
|
|
@@ -577,9 +618,63 @@ export function findRetakeGroups(
|
|
|
577
618
|
}
|
|
578
619
|
if (overlaps) continue;
|
|
579
620
|
groups.push(g);
|
|
621
|
+
// The sentence pass's members join the claimed set too: the exact-prefix
|
|
622
|
+
// pass below must not re-decide words either earlier pass already
|
|
623
|
+
// decided — including spares, so a declined cut stays declined.
|
|
624
|
+
for (const m of members) {
|
|
625
|
+
for (let w = m.startWord; w <= m.endWord; w++) claimed.add(w);
|
|
626
|
+
}
|
|
627
|
+
}
|
|
628
|
+
|
|
629
|
+
// Exact-prefix restart pass (2026-08-16 field case; module doc has the
|
|
630
|
+
// §128-narrowing rationale). The flub: sentence A is an EXACT token prefix
|
|
631
|
+
// of the immediately following sentence B, strictly shorter, with a real
|
|
632
|
+
// pause (>= RESTART_SPLIT_MIN_SIL of silence) at the restart boundary. ASR
|
|
633
|
+
// punctuated A as complete ("You can use OpenAI."), so the similarity
|
|
634
|
+
// passes above score A-vs-B honestly divergent (1 - 3/7 ≈ 0.57) and
|
|
635
|
+
// correctly decline — completeness is no defense here, which is why A may
|
|
636
|
+
// be complete. Adjacency skips zero-token sentences, the same
|
|
637
|
+
// filler/marker transparency the chaining rule grants.
|
|
638
|
+
const live = sentences.filter((s) => s.tokens.length > 0);
|
|
639
|
+
for (let k = 0; k + 1 < live.length; k++) {
|
|
640
|
+
const a = live[k]!;
|
|
641
|
+
const b = live[k + 1]!;
|
|
642
|
+
if (a.hallucinated || b.hallucinated) continue;
|
|
643
|
+
if (a.tokens.length < RETAKE_MIN_TOKENS) continue;
|
|
644
|
+
if (a.tokens.length >= b.tokens.length) continue;
|
|
645
|
+
if (!a.tokens.every((tok, j) => tokensEqual(tok, b.tokens[j]!))) continue;
|
|
646
|
+
// The pause: ONE silence overlapping [A.end, B.start + window] by at
|
|
647
|
+
// least RESTART_SPLIT_MIN_SIL — see PREFIX_RESTART_SIL_WINDOW for why
|
|
648
|
+
// the window reaches INTO B's stamped words and why word gaps are
|
|
649
|
+
// useless here. Per-silence, not summed: three 0.1s breath gaps across
|
|
650
|
+
// the window are ordinary delivery, not a restart pause.
|
|
651
|
+
const windowEnd = b.startSec + PREFIX_RESTART_SIL_WINDOW;
|
|
652
|
+
const paused = analysis.silences.some(
|
|
653
|
+
(s) => silenceOverlap([s], a.endSec, windowEnd) >= RESTART_SPLIT_MIN_SIL,
|
|
654
|
+
);
|
|
655
|
+
if (!paused) continue;
|
|
656
|
+
// Claimed dedupe, same set as the sentence pass: a word either earlier
|
|
657
|
+
// pass already cut, kept OR deliberately spared is not re-decidable.
|
|
658
|
+
let overlaps = false;
|
|
659
|
+
for (const m of [a, b]) {
|
|
660
|
+
for (let w = m.startWord; w <= m.endWord && !overlaps; w++) {
|
|
661
|
+
if (claimed.has(w)) overlaps = true;
|
|
662
|
+
}
|
|
663
|
+
}
|
|
664
|
+
if (overlaps) continue;
|
|
665
|
+
groups.push({
|
|
666
|
+
kept: toPublic(b),
|
|
667
|
+
cuts: [{ ...toPublic(a), similarity: prefixSimilarity(a.tokens, b.tokens) }],
|
|
668
|
+
hallucinated: [],
|
|
669
|
+
undecided: [],
|
|
670
|
+
rule: "exact-prefix",
|
|
671
|
+
});
|
|
672
|
+
for (const m of [a, b]) {
|
|
673
|
+
for (let w = m.startWord; w <= m.endWord; w++) claimed.add(w);
|
|
674
|
+
}
|
|
580
675
|
}
|
|
581
676
|
|
|
582
|
-
//
|
|
677
|
+
// The passes can interleave in transcript order — the report reads in
|
|
583
678
|
// word order, whichever pass found the group.
|
|
584
679
|
groups.sort((a, b) => {
|
|
585
680
|
const first = (g: RetakeGroup): number =>
|
|
@@ -667,7 +762,14 @@ export function formatRetakeGroup(transcript: Transcript, group: RetakeGroup): s
|
|
|
667
762
|
} else {
|
|
668
763
|
lines.push(`kept: "${said(group.kept)}"`);
|
|
669
764
|
for (const c of group.cuts) {
|
|
670
|
-
|
|
765
|
+
// An exact-prefix cut was never scored — printing a percentage would
|
|
766
|
+
// claim evidence the rule doesn't have. Name the structural evidence
|
|
767
|
+
// (prefix + pause) instead.
|
|
768
|
+
lines.push(
|
|
769
|
+
group.rule === "exact-prefix"
|
|
770
|
+
? ` cut (exact-prefix restart, ≥0.35s pause): "${said(c)}"`
|
|
771
|
+
: ` cut (${Math.round(c.similarity * 100)}% match): "${said(c)}"`,
|
|
772
|
+
);
|
|
671
773
|
}
|
|
672
774
|
for (const u of group.undecided) {
|
|
673
775
|
const pct = Math.round((u.similarity ?? 0) * 100);
|
package/src/scene-schema.ts
CHANGED
|
@@ -186,6 +186,16 @@ export const FaceCropSchema = z.object({
|
|
|
186
186
|
* phone footage, wrong for a webcam recording or a screen capture.
|
|
187
187
|
*/
|
|
188
188
|
sourceAspect: z.number().positive().optional(),
|
|
189
|
+
/**
|
|
190
|
+
* What the measured face IS: the frame's subject, or an incidental face in
|
|
191
|
+
* a frame that is really about something else (2026-08-16 incident: a
|
|
192
|
+
* screen recording's camera PiP, 12% of frame height at the bottom-right,
|
|
193
|
+
* dragged the cover crop to the frame bottom and decapitated the speaker
|
|
194
|
+
* everywhere they appeared full-frame). "screen" tells the stage to center
|
|
195
|
+
* the cover instead of biasing toward the face. Absent means "face" — every
|
|
196
|
+
* pre-existing render-props keeps its old behavior.
|
|
197
|
+
*/
|
|
198
|
+
subject: z.enum(["face", "screen"]).optional(),
|
|
189
199
|
});
|
|
190
200
|
export type FaceCrop = z.infer<typeof FaceCropSchema>;
|
|
191
201
|
|