@ossclip/core 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +27 -0
- package/README.md +20 -0
- package/package.json +29 -0
- package/src/analyze.ts +299 -0
- package/src/assemble.ts +124 -0
- package/src/browser.ts +24 -0
- package/src/captions.ts +92 -0
- package/src/clip.ts +306 -0
- package/src/config.ts +66 -0
- package/src/content-rect-detect.ts +162 -0
- package/src/content-rect.ts +324 -0
- package/src/cover.ts +216 -0
- package/src/cta.ts +68 -0
- package/src/cutlist.ts +170 -0
- package/src/exec.ts +36 -0
- package/src/face.ts +519 -0
- package/src/fill.ts +110 -0
- package/src/framing.ts +277 -0
- package/src/grounding.ts +130 -0
- package/src/index.ts +27 -0
- package/src/ingest.ts +83 -0
- package/src/normalize.ts +397 -0
- package/src/overrides.ts +509 -0
- package/src/phonetics.ts +129 -0
- package/src/producer/anthropic.ts +73 -0
- package/src/producer/beats.ts +330 -0
- package/src/producer/claude-cli.ts +150 -0
- package/src/producer/gemini.ts +197 -0
- package/src/producer/index.ts +217 -0
- package/src/producer/mock.ts +101 -0
- package/src/producer/provider.ts +42 -0
- package/src/producer/repair.ts +474 -0
- package/src/producer/scene-props.ts +212 -0
- package/src/producer/tiered.ts +56 -0
- package/src/producer/usage.ts +426 -0
- package/src/report.ts +36 -0
- package/src/scene-registry.ts +246 -0
- package/src/scene-schema.ts +203 -0
- package/src/schema.ts +177 -0
- package/src/source-text.ts +348 -0
- package/src/timemap.ts +115 -0
- package/src/transcribe.ts +67 -0
- package/src/zoom.ts +154 -0
|
@@ -0,0 +1,474 @@
|
|
|
1
|
+
import { z } from "zod/v4";
|
|
2
|
+
import type { Transcript, Word } from "../schema";
|
|
3
|
+
import type { Scene } from "../scene-schema";
|
|
4
|
+
import { soundsSimilar } from "../phonetics";
|
|
5
|
+
import type { LlmProvider } from "./provider";
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Transcript repair (FINDINGS §17/§21).
|
|
9
|
+
*
|
|
10
|
+
* ASR mishearings used to reach the screen twice over: once in the captions
|
|
11
|
+
* (raw ASR), and once as pressure on the producer, which was told to repair
|
|
12
|
+
* mishearings in its copy and then had `checkGrounding` report the repair as
|
|
13
|
+
* an invention. Worse, the two halves disagreed *in the same frame* — a
|
|
14
|
+
* graphic reading "Orchestration Tax" over a caption reading "Orchestration
|
|
15
|
+
* text".
|
|
16
|
+
*
|
|
17
|
+
* The fix is to repair ONCE, up front, and let one transcript feed captions,
|
|
18
|
+
* the producer and the grounding check. Everything downstream then agrees by
|
|
19
|
+
* construction rather than by luck.
|
|
20
|
+
*
|
|
21
|
+
* The pass is deliberately narrow: it may only swap words for words that
|
|
22
|
+
* SOUND THE SAME. These words end up on screen, so an LLM must not be able to
|
|
23
|
+
* paraphrase, censor, or tidy up what the speaker actually said.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
export const TranscriptRepairSchema = z.object({
|
|
27
|
+
repairs: z
|
|
28
|
+
.array(
|
|
29
|
+
z.object({
|
|
30
|
+
startWord: z.number().int().nonnegative(),
|
|
31
|
+
endWord: z.number().int().nonnegative(),
|
|
32
|
+
/** The ASR text being replaced — quoted back as the anchor we verify. */
|
|
33
|
+
heard: z.string().max(80),
|
|
34
|
+
correction: z.string().max(80),
|
|
35
|
+
}),
|
|
36
|
+
)
|
|
37
|
+
// A ceiling, not a target: rewriting a large fraction of a take under a
|
|
38
|
+
// guard that cannot hear the audio is not repair, it is redrafting.
|
|
39
|
+
.max(12),
|
|
40
|
+
});
|
|
41
|
+
|
|
42
|
+
/** A repair may not restructure a sentence — these bound "swapped a word". */
|
|
43
|
+
const MAX_SPAN_WORDS = 4;
|
|
44
|
+
const MAX_TOKEN_DELTA = 1;
|
|
45
|
+
const MAX_LENGTH_RATIO = 2;
|
|
46
|
+
/** …but ratios are meaningless at 2-3 characters, so allow a small absolute gap. */
|
|
47
|
+
const MAX_LENGTH_DELTA = 3;
|
|
48
|
+
/** Words shorter than this get dropped by TimeMap.mapWord — never emit them. */
|
|
49
|
+
const MIN_WORD_SEC = 0.04;
|
|
50
|
+
/** How far from the claimed index to look for the quoted text. */
|
|
51
|
+
const INDEX_SEARCH_RADIUS = 3;
|
|
52
|
+
export type TranscriptRepair = z.infer<typeof TranscriptRepairSchema>["repairs"][number];
|
|
53
|
+
|
|
54
|
+
export interface AppliedRepair {
|
|
55
|
+
startWord: number;
|
|
56
|
+
endWord: number;
|
|
57
|
+
heard: string;
|
|
58
|
+
correction: string;
|
|
59
|
+
applied: boolean;
|
|
60
|
+
/** Why a proposal was refused — surfaced, never swallowed. */
|
|
61
|
+
rejected?: string;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
export const REPAIR_SYSTEM = `You correct speech-recognition errors in a transcript. The corrected words are shown on screen as captions, so this job is narrow and literal.
|
|
65
|
+
|
|
66
|
+
Fix ONLY mishearings: places where the recognizer produced words that sound like what was said but are the wrong words. Typical cases: a common phrase turned into an unfamiliar proper noun ("code churn" heard as "CodeChun" or "coach and"), or a word swapped for a near-homophone ("tax" heard as "text").
|
|
67
|
+
|
|
68
|
+
You MUST NOT:
|
|
69
|
+
- paraphrase, reword, shorten or improve anything;
|
|
70
|
+
- fix grammar, punctuation or capitalisation;
|
|
71
|
+
- remove filler words, stutters or repetition — they are part of the take;
|
|
72
|
+
- censor or soften anything;
|
|
73
|
+
- "correct" a word that is merely unusual but plausible as spoken.
|
|
74
|
+
|
|
75
|
+
A correction must sound essentially identical to what was heard. If a span is not clearly a mishearing, leave it alone. Returning an empty list is the correct answer for a clean transcript.
|
|
76
|
+
|
|
77
|
+
For each fix give the word-index span, the exact text you are replacing (\`heard\`), and the corrected text.`;
|
|
78
|
+
|
|
79
|
+
export function buildRepairUserPrompt(transcript: Transcript, speaker?: string): string {
|
|
80
|
+
const words = transcript.words.map((w, i) => `[${i}]${w.text}`).join(" ");
|
|
81
|
+
return (
|
|
82
|
+
(speaker
|
|
83
|
+
? // The failure class a hint fixes: recovering a proper noun needs to
|
|
84
|
+
// know WHO is talking, not what the phonemes were. Measured on a real
|
|
85
|
+
// reel, "code with SM" became "Code with Ahsan" (the speaker's own
|
|
86
|
+
// channel) with one model and the invented name "Sam" with another —
|
|
87
|
+
// and "Sam" genuinely does sound like "SM", so no phonetic gate can
|
|
88
|
+
// catch it. Naming the speaker turns a guess into a lookup.
|
|
89
|
+
`About the speaker (use this to recognise names the recognizer mangled, ` +
|
|
90
|
+
`never to introduce facts): ${speaker}\n\n`
|
|
91
|
+
: "") +
|
|
92
|
+
`Word-indexed transcript (indices refer to THIS list):\n${words}\n\n` +
|
|
93
|
+
`Report only spans that are clearly mishearings.`
|
|
94
|
+
);
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
/** Comparable form: case- and punctuation-insensitive. */
|
|
98
|
+
function norm(s: string): string {
|
|
99
|
+
return s
|
|
100
|
+
.toLowerCase()
|
|
101
|
+
.replace(/[^a-z0-9\s]/g, "")
|
|
102
|
+
.split(/\s+/)
|
|
103
|
+
.filter(Boolean)
|
|
104
|
+
.join(" ");
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
function spanText(transcript: Transcript, startWord: number, endWord: number): string {
|
|
108
|
+
return transcript.words
|
|
109
|
+
.slice(startWord, endWord + 1)
|
|
110
|
+
.map((w) => w.text)
|
|
111
|
+
.join(" ");
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* Re-time a correction over the span it replaces.
|
|
116
|
+
*
|
|
117
|
+
* When the token count is unchanged — the common case, "coach and" → "code
|
|
118
|
+
* churn" — the ASR's own word boundaries are MEASURED onsets, so they are
|
|
119
|
+
* kept and only the text changes. Guessing new boundaries there would throw
|
|
120
|
+
* away real information. Only a count change forces a split, and then the
|
|
121
|
+
* span is divided proportionally to token length.
|
|
122
|
+
*
|
|
123
|
+
* Either way timings stay inside the original span and strictly increasing,
|
|
124
|
+
* which is what `TimeMap.mapWord` and `buildCaptionLines` rely on.
|
|
125
|
+
*/
|
|
126
|
+
function retime(tokens: string[], originals: readonly Word[]): Word[] {
|
|
127
|
+
if (tokens.length === originals.length) {
|
|
128
|
+
return tokens.map((text, i) => ({ ...originals[i]!, text }));
|
|
129
|
+
}
|
|
130
|
+
const start = originals[0]!.start;
|
|
131
|
+
const end = originals[originals.length - 1]!.end;
|
|
132
|
+
const weights = tokens.map((t) => t.length + 1);
|
|
133
|
+
const total = weights.reduce((a, b) => a + b, 0);
|
|
134
|
+
const out: Word[] = [];
|
|
135
|
+
let cursor = start;
|
|
136
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
137
|
+
const share = ((end - start) * weights[i]!) / total;
|
|
138
|
+
const wordEnd = i === tokens.length - 1 ? end : cursor + share;
|
|
139
|
+
out.push({ text: tokens[i]!, start: cursor, end: wordEnd });
|
|
140
|
+
cursor = wordEnd;
|
|
141
|
+
}
|
|
142
|
+
return out;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/** Would a split produce words too short to survive the TimeMap? */
|
|
146
|
+
function tooShort(tokens: string[], originals: readonly Word[]): boolean {
|
|
147
|
+
if (tokens.length === originals.length) return false;
|
|
148
|
+
const span = originals[originals.length - 1]!.end - originals[0]!.start;
|
|
149
|
+
return span / tokens.length < MIN_WORD_SEC;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
export interface RepairOptions extends ApplyRepairsOptions {
|
|
153
|
+
/** Who is speaking — lets the model recognise mangled proper nouns. */
|
|
154
|
+
speaker?: string;
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* Apply proposed repairs, refusing anything that isn't demonstrably a
|
|
159
|
+
* mishearing of the span it claims to fix. Pure — the LLM is only a source of
|
|
160
|
+
* proposals; every guard below is deterministic and testable.
|
|
161
|
+
*/
|
|
162
|
+
export interface ApplyRepairsOptions {
|
|
163
|
+
/**
|
|
164
|
+
* True when the given source-time span contains a cut. A repair straddling
|
|
165
|
+
* a removal would merge words across the cut, so it is refused.
|
|
166
|
+
*/
|
|
167
|
+
isCut?: (startSec: number, endSec: number) => boolean;
|
|
168
|
+
/**
|
|
169
|
+
* The `--speaker` hint. Words the USER supplied are vouched-for evidence,
|
|
170
|
+
* not model invention, so a correction made entirely of them may pass the
|
|
171
|
+
* phonetic gate. Everything else about it is still checked.
|
|
172
|
+
*/
|
|
173
|
+
speaker?: string;
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
export function applyRepairs(
|
|
177
|
+
transcript: Transcript,
|
|
178
|
+
repairs: readonly TranscriptRepair[],
|
|
179
|
+
opts: ApplyRepairsOptions = {},
|
|
180
|
+
): { transcript: Transcript; applied: AppliedRepair[] } {
|
|
181
|
+
const maxIndex = transcript.words.length - 1;
|
|
182
|
+
/**
|
|
183
|
+
* Names the user vouched for via `--speaker`. A recognizer that turned
|
|
184
|
+
* "Ahsan" into the initialism "SM" produces a correction no phonetic
|
|
185
|
+
* measure will accept — and refusing it leaves the wrong name in the
|
|
186
|
+
* captions, which is the failure the hint exists to prevent. So a correction
|
|
187
|
+
* built ENTIRELY from words the user supplied is exempt from that one gate.
|
|
188
|
+
* Every other guard still applies, and the exemption can only ever
|
|
189
|
+
* substitute text the user typed themselves.
|
|
190
|
+
*/
|
|
191
|
+
const speakerWords = new Set(norm(opts.speaker ?? "").split(" ").filter(Boolean));
|
|
192
|
+
const speakerVouched = (correction: string): boolean => {
|
|
193
|
+
const tokens = norm(correction).split(" ").filter(Boolean);
|
|
194
|
+
return tokens.length > 0 && tokens.every((t) => speakerWords.has(t));
|
|
195
|
+
};
|
|
196
|
+
const results: AppliedRepair[] = [];
|
|
197
|
+
const accepted: Array<{ startWord: number; endWord: number; tokens: string[] }> = [];
|
|
198
|
+
const claimed: Array<[number, number]> = [];
|
|
199
|
+
|
|
200
|
+
/**
|
|
201
|
+
* The model quoted `heard` by copying from the prompt, so it is reliable;
|
|
202
|
+
* what it gets wrong is index arithmetic over hundreds of `[i]word` tokens.
|
|
203
|
+
* So trust the text and re-derive the index: look for the quoted words near
|
|
204
|
+
* the claimed position and use where they actually are.
|
|
205
|
+
*/
|
|
206
|
+
const locate = (r: TranscriptRepair): { startWord: number; endWord: number } | null => {
|
|
207
|
+
const want = norm(r.heard);
|
|
208
|
+
// Widths to try, in order of trust. The QUOTED TEXT is the reliable part
|
|
209
|
+
// of a proposal, so its own token count leads; the claimed span is a
|
|
210
|
+
// fallback for a quote whose normalisation splits differently. Trusting
|
|
211
|
+
// only the claimed width used to reject correct repairs outright: a model
|
|
212
|
+
// proposed `"SM" → "Ahsan"` at 64-65, so the search only ever tested the
|
|
213
|
+
// two-word span "SM which" and never the one word it had quoted.
|
|
214
|
+
const widths = [...new Set([want.split(" ").length - 1, r.endWord - r.startWord])];
|
|
215
|
+
for (const width of widths) {
|
|
216
|
+
if (width < 0) continue;
|
|
217
|
+
for (let delta = 0; delta <= INDEX_SEARCH_RADIUS; delta++) {
|
|
218
|
+
const starts = delta === 0 ? [r.startWord] : [r.startWord - delta, r.startWord + delta];
|
|
219
|
+
for (const start of starts) {
|
|
220
|
+
const end = start + width;
|
|
221
|
+
if (start < 0 || end > maxIndex) continue;
|
|
222
|
+
if (norm(spanText(transcript, start, end)) === want) {
|
|
223
|
+
return { startWord: start, endWord: end };
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
return null;
|
|
229
|
+
};
|
|
230
|
+
|
|
231
|
+
for (const r of repairs) {
|
|
232
|
+
let located: { startWord: number; endWord: number } | null = null;
|
|
233
|
+
const record = (rejected?: string): void => {
|
|
234
|
+
results.push({
|
|
235
|
+
startWord: located?.startWord ?? r.startWord,
|
|
236
|
+
endWord: located?.endWord ?? r.endWord,
|
|
237
|
+
heard: r.heard,
|
|
238
|
+
correction: r.correction,
|
|
239
|
+
applied: rejected === undefined,
|
|
240
|
+
...(rejected === undefined ? {} : { rejected }),
|
|
241
|
+
});
|
|
242
|
+
};
|
|
243
|
+
|
|
244
|
+
if (r.endWord < r.startWord) {
|
|
245
|
+
record("span ends before it starts");
|
|
246
|
+
continue;
|
|
247
|
+
}
|
|
248
|
+
if (r.endWord - r.startWord + 1 > MAX_SPAN_WORDS) {
|
|
249
|
+
record(`span of ${r.endWord - r.startWord + 1} words is a rewrite, not a mishearing`);
|
|
250
|
+
continue;
|
|
251
|
+
}
|
|
252
|
+
located = locate(r);
|
|
253
|
+
if (located === null) {
|
|
254
|
+
record(`quoted "${r.heard}" but no span near ${r.startWord} matches it`);
|
|
255
|
+
continue;
|
|
256
|
+
}
|
|
257
|
+
if (claimed.some(([s, e]) => located!.startWord <= e && located!.endWord >= s)) {
|
|
258
|
+
record("span overlaps an earlier repair");
|
|
259
|
+
continue;
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
const originals = transcript.words.slice(located.startWord, located.endWord + 1);
|
|
263
|
+
const actual = originals.map((w) => w.text).join(" ");
|
|
264
|
+
const tokens = r.correction.split(/\s+/).filter(Boolean);
|
|
265
|
+
if (tokens.length === 0) {
|
|
266
|
+
record("empty correction");
|
|
267
|
+
continue;
|
|
268
|
+
}
|
|
269
|
+
if (norm(actual) === norm(r.correction)) {
|
|
270
|
+
record("correction is identical to what was heard");
|
|
271
|
+
continue;
|
|
272
|
+
}
|
|
273
|
+
if (Math.abs(tokens.length - originals.length) > MAX_TOKEN_DELTA) {
|
|
274
|
+
record(`${originals.length} words → ${tokens.length} restructures the sentence`);
|
|
275
|
+
continue;
|
|
276
|
+
}
|
|
277
|
+
// A ratio is a poor measure at small magnitudes: "SM" → "Ahsan" is 2.5x
|
|
278
|
+
// by ratio but three characters, and an initialism the recognizer coined
|
|
279
|
+
// out of a real name is exactly the case the speaker hint exists to fix.
|
|
280
|
+
// So allow a small ABSOLUTE difference regardless of ratio — the phonetic
|
|
281
|
+
// gate below is what actually separates a repair from a rewrite.
|
|
282
|
+
const ratio = r.correction.length / Math.max(1, actual.length);
|
|
283
|
+
const absDelta = Math.abs(r.correction.length - actual.length);
|
|
284
|
+
if (absDelta > MAX_LENGTH_DELTA && (ratio > MAX_LENGTH_RATIO || ratio < 1 / MAX_LENGTH_RATIO)) {
|
|
285
|
+
record(`"${r.correction}" is too different in length from "${actual}"`);
|
|
286
|
+
continue;
|
|
287
|
+
}
|
|
288
|
+
if (!soundsSimilar(actual, r.correction) && !speakerVouched(r.correction)) {
|
|
289
|
+
// The gate that keeps this a repair pass and not a rewrite pass.
|
|
290
|
+
record(`"${r.correction}" does not sound like "${actual}" — rewrite, not a repair`);
|
|
291
|
+
continue;
|
|
292
|
+
}
|
|
293
|
+
if (tooShort(tokens, originals)) {
|
|
294
|
+
record("split would produce words too short to survive the time map");
|
|
295
|
+
continue;
|
|
296
|
+
}
|
|
297
|
+
if (opts.isCut?.(originals[0]!.start, originals[originals.length - 1]!.end)) {
|
|
298
|
+
record("span straddles a cut");
|
|
299
|
+
continue;
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
claimed.push([located.startWord, located.endWord]);
|
|
303
|
+
accepted.push({ ...located, tokens });
|
|
304
|
+
record();
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
// Back to front: a repair may change the word count, and every span index
|
|
308
|
+
// ahead of the edit point must stay valid while the rest are applied.
|
|
309
|
+
const words = [...transcript.words];
|
|
310
|
+
for (const r of [...accepted].sort((a, b) => b.startWord - a.startWord)) {
|
|
311
|
+
const originals = words.slice(r.startWord, r.endWord + 1);
|
|
312
|
+
words.splice(r.startWord, originals.length, ...retime(r.tokens, originals));
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
return { transcript: { ...transcript, words }, applied: results };
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
/**
|
|
319
|
+
* One LLM call, then the deterministic guards. Fail-soft: a provider that
|
|
320
|
+
* throws (or returns nothing usable) yields zero repairs and the raw
|
|
321
|
+
* transcript, never a failed render — the same degrade-don't-fail policy the
|
|
322
|
+
* scene-props call follows.
|
|
323
|
+
*/
|
|
324
|
+
export async function repairTranscript(
|
|
325
|
+
provider: LlmProvider,
|
|
326
|
+
transcript: Transcript,
|
|
327
|
+
opts: RepairOptions = {},
|
|
328
|
+
): Promise<{ transcript: Transcript; applied: AppliedRepair[]; error?: string }> {
|
|
329
|
+
if (transcript.words.length === 0) return { transcript, applied: [] };
|
|
330
|
+
// The budget scales with the transcript (R20 §98): the flat 4000 was sized
|
|
331
|
+
// for 30-70s takes, and the first long-form run blew straight through it —
|
|
332
|
+
// a thinking model's thought tokens draw from the SAME budget, so the
|
|
333
|
+
// visible JSON was truncated mid-string after burning half the run's cost.
|
|
334
|
+
const maxTokens = Math.min(32000, 4000 + transcript.words.length * 10);
|
|
335
|
+
let lastError: unknown;
|
|
336
|
+
// Two attempts (R20 §98): repair is the gate that keeps a mishearing off
|
|
337
|
+
// the screen (§17), and one malformed response was measured costing the
|
|
338
|
+
// whole pass — a single retry is cheap against that. Still fail-soft after.
|
|
339
|
+
for (let attempt = 0; attempt < 2; attempt++) {
|
|
340
|
+
try {
|
|
341
|
+
const result = await provider.complete({
|
|
342
|
+
system: REPAIR_SYSTEM,
|
|
343
|
+
user: buildRepairUserPrompt(transcript, opts.speaker),
|
|
344
|
+
schema: TranscriptRepairSchema,
|
|
345
|
+
schemaName: "transcript_repair",
|
|
346
|
+
// EDITORIAL on purpose, despite looking mechanical. Measured on the real
|
|
347
|
+
// reel: the small model returned zero repairs where the large one
|
|
348
|
+
// recovers "code with SM" → "Code with Ahsan" every time. Deciding what
|
|
349
|
+
// a person actually said is semantic work, and it is the gate that keeps
|
|
350
|
+
// a mishearing off the screen (§17) — the wrong place to save $0.20.
|
|
351
|
+
tier: "editorial",
|
|
352
|
+
maxTokens,
|
|
353
|
+
});
|
|
354
|
+
return applyRepairs(transcript, result.repairs, opts);
|
|
355
|
+
} catch (err) {
|
|
356
|
+
lastError = err;
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
return {
|
|
360
|
+
transcript,
|
|
361
|
+
applied: [],
|
|
362
|
+
error: lastError instanceof Error ? lastError.message : String(lastError),
|
|
363
|
+
};
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
// ---- §21: the producer's copy and the captions must spell words the same ---
|
|
367
|
+
|
|
368
|
+
/** Props whose text is lifted from what the speaker said (not stylised copy). */
|
|
369
|
+
const COPY_FIELDS: Record<string, string[]> = {
|
|
370
|
+
TitleCard: ["eyebrow", "title", "sub"],
|
|
371
|
+
StatCard: ["label", "caption"],
|
|
372
|
+
RuleCard: ["kicker", "text", "struck"],
|
|
373
|
+
StrikethroughReveal: ["lines"],
|
|
374
|
+
FlowDiagram: ["nodes"],
|
|
375
|
+
ScreenshotFrame: ["label"],
|
|
376
|
+
};
|
|
377
|
+
|
|
378
|
+
function stringsOf(value: unknown): string[] {
|
|
379
|
+
if (typeof value === "string") return [value];
|
|
380
|
+
if (Array.isArray(value)) return value.flatMap(stringsOf);
|
|
381
|
+
if (value && typeof value === "object") return Object.values(value).flatMap(stringsOf);
|
|
382
|
+
return [];
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
/**
|
|
386
|
+
* Same word, different inflection — "SHIP" for "shipped", "AGENTS" for
|
|
387
|
+
* "agent". These sound alike by construction, but the producer shortening a
|
|
388
|
+
* word for the screen is EDITING, not a recognizer error, and rewriting the
|
|
389
|
+
* caption to match would put words in the speaker's mouth.
|
|
390
|
+
*/
|
|
391
|
+
function isInflection(a: string, b: string): boolean {
|
|
392
|
+
const [short, long] = a.length <= b.length ? [a, b] : [b, a];
|
|
393
|
+
if (!long.startsWith(short)) return false;
|
|
394
|
+
return long.length - short.length <= 3;
|
|
395
|
+
}
|
|
396
|
+
|
|
397
|
+
/**
|
|
398
|
+
* Copy the original word's capitalisation onto the replacement, so a caption
|
|
399
|
+
* corrected from an ALL-CAPS on-screen label still reads as speech.
|
|
400
|
+
*/
|
|
401
|
+
function matchCase(original: string, replacement: string): string {
|
|
402
|
+
if (original === original.toUpperCase() && /[A-Z]/.test(original)) return replacement.toUpperCase();
|
|
403
|
+
if (/^[A-Z]/.test(original)) {
|
|
404
|
+
return replacement.charAt(0).toUpperCase() + replacement.slice(1).toLowerCase();
|
|
405
|
+
}
|
|
406
|
+
return replacement.toLowerCase();
|
|
407
|
+
}
|
|
408
|
+
|
|
409
|
+
/**
|
|
410
|
+
* Reconciliation demands a closer match than the repair pass: this is a
|
|
411
|
+
* single-word swap with no resegmentation to excuse a low score.
|
|
412
|
+
*/
|
|
413
|
+
const RECONCILE_FLOOR = 0.6;
|
|
414
|
+
|
|
415
|
+
/**
|
|
416
|
+
* Reconcile caption text with the on-screen copy (FINDINGS §21).
|
|
417
|
+
*
|
|
418
|
+
* The repair pass runs before the producer, so normally both sides already
|
|
419
|
+
* agree. This catches the residue: where a scene's copy contains a word that
|
|
420
|
+
* SOUNDS LIKE — but isn't spelled like — a word the speaker says underneath
|
|
421
|
+
* it, the producer's spelling wins and the caption is corrected to match.
|
|
422
|
+
* That is the "Orchestration Tax" / "Orchestration text" case.
|
|
423
|
+
*
|
|
424
|
+
* Strictly a 1:1 word substitution: word count and all timings are untouched,
|
|
425
|
+
* so scene anchors stay valid and this is safe to run after the producer.
|
|
426
|
+
*/
|
|
427
|
+
export function reconcileCopy(
|
|
428
|
+
transcript: Transcript,
|
|
429
|
+
scenes: readonly Scene[],
|
|
430
|
+
): { transcript: Transcript; applied: AppliedRepair[] } {
|
|
431
|
+
const words = [...transcript.words];
|
|
432
|
+
const applied: AppliedRepair[] = [];
|
|
433
|
+
const spokenExactly = new Set(words.map((w) => norm(w.text)).filter(Boolean));
|
|
434
|
+
|
|
435
|
+
for (const scene of scenes) {
|
|
436
|
+
const fields = COPY_FIELDS[scene.component] ?? [];
|
|
437
|
+
const merged = { ...scene.props, ...scene.overrides };
|
|
438
|
+
const copyTokens = fields
|
|
439
|
+
.flatMap((f) => stringsOf(merged[f]))
|
|
440
|
+
.flatMap((s) => s.split(/\s+/))
|
|
441
|
+
.map((t) => t.replace(/[^A-Za-z]/g, ""))
|
|
442
|
+
.filter((t) => t.length >= 3);
|
|
443
|
+
|
|
444
|
+
for (let i = scene.anchor.startWord; i <= Math.min(scene.anchor.endWord, words.length - 1); i++) {
|
|
445
|
+
const spoken = words[i]!;
|
|
446
|
+
const key = norm(spoken.text);
|
|
447
|
+
if (!key || key.length < 3) continue;
|
|
448
|
+
for (const token of copyTokens) {
|
|
449
|
+
const candidate = norm(token);
|
|
450
|
+
if (candidate === key) break; // already agrees
|
|
451
|
+
// Only correct toward a word the take never says: if the copy's word
|
|
452
|
+
// appears verbatim elsewhere in the transcript, the speaker used both
|
|
453
|
+
// and this is not a mishearing.
|
|
454
|
+
if (spokenExactly.has(candidate)) continue;
|
|
455
|
+
// "SHIP" for "shipped" is the producer editing for the screen, not
|
|
456
|
+
// the recognizer erring — the caption keeps what was actually said.
|
|
457
|
+
if (isInflection(candidate, key)) continue;
|
|
458
|
+
if (!soundsSimilar(spoken.text, token, RECONCILE_FLOOR)) continue;
|
|
459
|
+
const correction = matchCase(spoken.text, token);
|
|
460
|
+
applied.push({
|
|
461
|
+
startWord: i,
|
|
462
|
+
endWord: i,
|
|
463
|
+
heard: spoken.text,
|
|
464
|
+
correction,
|
|
465
|
+
applied: true,
|
|
466
|
+
});
|
|
467
|
+
words[i] = { ...spoken, text: correction };
|
|
468
|
+
break;
|
|
469
|
+
}
|
|
470
|
+
}
|
|
471
|
+
}
|
|
472
|
+
|
|
473
|
+
return { transcript: { ...transcript, words }, applied };
|
|
474
|
+
}
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
import { z } from "zod/v4";
|
|
2
|
+
import type { Transcript } from "../schema";
|
|
3
|
+
import type { Layout, Scene, SceneComponentId } from "../scene-schema";
|
|
4
|
+
import { SCENE_REGISTRY } from "../scene-registry";
|
|
5
|
+
import type { LlmProvider } from "./provider";
|
|
6
|
+
import type { Moment } from "./beats";
|
|
7
|
+
import {
|
|
8
|
+
layoutFeasible,
|
|
9
|
+
momentSourceWindow,
|
|
10
|
+
worstFaceFrac,
|
|
11
|
+
type FramingContext,
|
|
12
|
+
} from "../framing";
|
|
13
|
+
|
|
14
|
+
export interface ScenePropsFailure {
|
|
15
|
+
momentIndex: number;
|
|
16
|
+
component: SceneComponentId;
|
|
17
|
+
error: string;
|
|
18
|
+
fellBackTo: "TitleCard" | "dropped";
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
const PROPS_SYSTEM = `You fill the props for ONE scene component of a short-form video, from the transcript slice it accompanies. Copy is SHORT and punchy: numbers over adjectives, ALL-CAPS reads fine for labels/kickers, never full sentences. Output only what the schema asks for.
|
|
22
|
+
|
|
23
|
+
GROUNDING — hard rules:
|
|
24
|
+
- Every label, noun and claim must be supported by the transcript slice. NEVER introduce an entity, metric name or brand the slice does not contain — a number gets the noun the speaker attached to it, not a plausible-sounding one. If the slice offers no supporting noun, use the number alone or fall back to the suggested copy verbatim.
|
|
25
|
+
- The transcript is automatic speech recognition output and may contain mishearings. An unfamiliar proper noun is more likely a mistranscription of a common phrase than a real company or product — prefer the common-sense reading ("code churn", not "CodeChun") and never promote a suspected mishearing into a name or label.`;
|
|
26
|
+
|
|
27
|
+
function buildPropsPrompt(moment: Moment, transcript: Transcript): string {
|
|
28
|
+
const slice = transcript.words
|
|
29
|
+
.slice(moment.startWord, moment.endWord + 1)
|
|
30
|
+
.map((w) => w.text)
|
|
31
|
+
.join(" ");
|
|
32
|
+
return (
|
|
33
|
+
`Component: ${moment.sceneKind}\n` +
|
|
34
|
+
`Purpose of this moment: ${moment.purpose}\n` +
|
|
35
|
+
`Suggested on-screen copy: ${moment.onScreenCopy}\n` +
|
|
36
|
+
`Transcript slice: "${slice}"`
|
|
37
|
+
);
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* One call for ALL the graphic moments, returning props keyed by moment index.
|
|
42
|
+
*
|
|
43
|
+
* Call count is the only cost lever that matters here. Measured against the
|
|
44
|
+
* Claude Code CLI: every invocation carries ~25-45k tokens of harness prefix
|
|
45
|
+
* before ossclip's own prompt (~1-2k) is even considered, and that prefix is
|
|
46
|
+
* re-sent per call rather than reused across separate CLI sessions. A 32s clip
|
|
47
|
+
* spent 6 calls / 270k tokens, of which the content was a rounding error.
|
|
48
|
+
*
|
|
49
|
+
* Isolation is preserved by validating each item SEPARATELY on the way out:
|
|
50
|
+
* anything malformed simply isn't returned, and the caller retries that moment
|
|
51
|
+
* on its own. So the batch is a fast path, never a new failure mode.
|
|
52
|
+
*/
|
|
53
|
+
async function generateScenePropsBatch(
|
|
54
|
+
provider: LlmProvider,
|
|
55
|
+
moments: readonly Moment[],
|
|
56
|
+
indices: readonly number[],
|
|
57
|
+
transcript: Transcript,
|
|
58
|
+
): Promise<Map<number, Record<string, unknown>>> {
|
|
59
|
+
const out = new Map<number, Record<string, unknown>>();
|
|
60
|
+
if (indices.length < 2) return out; // nothing to amortise
|
|
61
|
+
const schema = z.object({
|
|
62
|
+
scenes: z.array(z.object({ index: z.number().int(), props: z.record(z.string(), z.unknown()) })),
|
|
63
|
+
});
|
|
64
|
+
const blocks = indices.map((i) => {
|
|
65
|
+
const moment = moments[i]!;
|
|
66
|
+
const meta = SCENE_REGISTRY[moment.sceneKind as SceneComponentId];
|
|
67
|
+
return (
|
|
68
|
+
`--- moment ${i} ---\n` +
|
|
69
|
+
`${buildPropsPrompt(moment, transcript)}\n` +
|
|
70
|
+
`Props schema: ${JSON.stringify(z.toJSONSchema(meta.propsSchema as z.ZodType))}`
|
|
71
|
+
);
|
|
72
|
+
});
|
|
73
|
+
let raw: z.infer<typeof schema>;
|
|
74
|
+
try {
|
|
75
|
+
raw = await provider.complete({
|
|
76
|
+
system: PROPS_SYSTEM,
|
|
77
|
+
user:
|
|
78
|
+
`Fill the props for EACH moment below. Reply with one entry per moment, ` +
|
|
79
|
+
`echoing its index. Each entry's props must satisfy that moment's own schema.\n\n` +
|
|
80
|
+
blocks.join("\n\n"),
|
|
81
|
+
schema,
|
|
82
|
+
schemaName: "scene_props_batch",
|
|
83
|
+
// Mechanical: filling a schema from a transcript slice, with every
|
|
84
|
+
// field validated on the way out.
|
|
85
|
+
tier: "mechanical",
|
|
86
|
+
});
|
|
87
|
+
} catch {
|
|
88
|
+
return out; // the per-moment path takes over
|
|
89
|
+
}
|
|
90
|
+
for (const entry of raw.scenes) {
|
|
91
|
+
const moment = moments[entry.index];
|
|
92
|
+
if (!moment || !indices.includes(entry.index)) continue;
|
|
93
|
+
const meta = SCENE_REGISTRY[moment.sceneKind as SceneComponentId];
|
|
94
|
+
const parsed = (meta.propsSchema as z.ZodType<Record<string, unknown>>).safeParse(entry.props);
|
|
95
|
+
if (parsed.success) out.set(entry.index, parsed.data);
|
|
96
|
+
}
|
|
97
|
+
return out;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* Call 2 — scene props (PHASE1 §4). One batched call covers the moments that
|
|
102
|
+
* behave; anything it fails to produce falls back to a per-moment call, so one
|
|
103
|
+
* bad scene still can't poison the rest. Validation loop per moment: schema
|
|
104
|
+
* parse → one retry with the error appended → TitleCard fallback with the
|
|
105
|
+
* moment's onScreenCopy. Bounded retries, accepted residuals — the Opus-log
|
|
106
|
+
* policy.
|
|
107
|
+
*/
|
|
108
|
+
export async function generateScenes(
|
|
109
|
+
provider: LlmProvider,
|
|
110
|
+
moments: readonly Moment[],
|
|
111
|
+
transcript: Transcript,
|
|
112
|
+
opts: { framing?: FramingContext } = {},
|
|
113
|
+
): Promise<{ scenes: Scene[]; failures: ScenePropsFailure[] }> {
|
|
114
|
+
const scenes: Scene[] = [];
|
|
115
|
+
const failures: ScenePropsFailure[] = [];
|
|
116
|
+
/**
|
|
117
|
+
* Layout resolution, in priority order (PLAN Task B):
|
|
118
|
+
*
|
|
119
|
+
* 1. The PRODUCER's explicit choice — that is the point of Task B, and it
|
|
120
|
+
* arrives already checked by `repairMomentLayouts`.
|
|
121
|
+
* 2. The §20 variety rotation: repeats of a component get an alternate
|
|
122
|
+
* layout, because two identical card treatments read as a template.
|
|
123
|
+
* 3. Feasibility outranks variety (Task B4): a rotation candidate that
|
|
124
|
+
* would crop the head at this moment's framing is skipped for the
|
|
125
|
+
* next feasible candidate — variety picks among what is left, and
|
|
126
|
+
* with nothing left the least-bad candidate stands (the same verdict
|
|
127
|
+
* `repairMomentLayouts` would reach).
|
|
128
|
+
*/
|
|
129
|
+
const seen = new Map<SceneComponentId, number>();
|
|
130
|
+
const layoutFor = (component: SceneComponentId, moment: Moment): Layout => {
|
|
131
|
+
const meta = SCENE_REGISTRY[component];
|
|
132
|
+
const n = seen.get(component) ?? 0;
|
|
133
|
+
seen.set(component, n + 1);
|
|
134
|
+
if (moment.layout) return moment.layout;
|
|
135
|
+
const rotation =
|
|
136
|
+
n === 0 || meta.altLayouts.length === 0
|
|
137
|
+
? [meta.defaultLayout, ...meta.altLayouts]
|
|
138
|
+
: [
|
|
139
|
+
meta.altLayouts[(n - 1) % meta.altLayouts.length]!,
|
|
140
|
+
meta.defaultLayout,
|
|
141
|
+
...meta.altLayouts,
|
|
142
|
+
];
|
|
143
|
+
const framing = opts.framing;
|
|
144
|
+
if (!framing) return rotation[0]!;
|
|
145
|
+
const window = momentSourceWindow(transcript, moment.startWord, moment.endWord);
|
|
146
|
+
const faceFrac = window
|
|
147
|
+
? worstFaceFrac(framing.windows, window.startSec, window.endSec)
|
|
148
|
+
: 0;
|
|
149
|
+
return rotation.find((l) => layoutFeasible(framing, l, faceFrac)) ?? rotation[0]!;
|
|
150
|
+
};
|
|
151
|
+
|
|
152
|
+
const graphicIndices = moments.flatMap((m, i) => (m.sceneKind === "none" ? [] : [i]));
|
|
153
|
+
const batched = await generateScenePropsBatch(provider, moments, graphicIndices, transcript);
|
|
154
|
+
|
|
155
|
+
for (let i = 0; i < moments.length; i++) {
|
|
156
|
+
const moment = moments[i]!;
|
|
157
|
+
if (moment.sceneKind === "none") continue;
|
|
158
|
+
const component = moment.sceneKind;
|
|
159
|
+
const meta = SCENE_REGISTRY[component];
|
|
160
|
+
const schema = meta.propsSchema as z.ZodType<Record<string, unknown>>;
|
|
161
|
+
const layout = layoutFor(component, moment);
|
|
162
|
+
|
|
163
|
+
let props: Record<string, unknown> | null = batched.get(i) ?? null;
|
|
164
|
+
let lastError = "";
|
|
165
|
+
const basePrompt = buildPropsPrompt(moment, transcript);
|
|
166
|
+
for (let attempt = 0; attempt < 2 && props === null; attempt++) {
|
|
167
|
+
const user =
|
|
168
|
+
attempt === 0
|
|
169
|
+
? basePrompt
|
|
170
|
+
: `${basePrompt}\n\nYour previous attempt failed validation:\n${lastError}\nFix it.`;
|
|
171
|
+
try {
|
|
172
|
+
props = await provider.complete({
|
|
173
|
+
system: PROPS_SYSTEM,
|
|
174
|
+
user,
|
|
175
|
+
schema,
|
|
176
|
+
schemaName: `${component}_props`,
|
|
177
|
+
tier: "mechanical",
|
|
178
|
+
});
|
|
179
|
+
} catch (err) {
|
|
180
|
+
lastError = err instanceof Error ? err.message : String(err);
|
|
181
|
+
}
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
if (props === null) {
|
|
185
|
+
// Degrade, don't fail the render (PHASE1 acceptance #4).
|
|
186
|
+
const fallbackTitle = moment.onScreenCopy.slice(0, 48) || "—";
|
|
187
|
+
failures.push({ momentIndex: i, component, error: lastError, fellBackTo: "TitleCard" });
|
|
188
|
+
scenes.push({
|
|
189
|
+
id: `scene-${i}`,
|
|
190
|
+
anchor: { startWord: moment.startWord, endWord: moment.endWord },
|
|
191
|
+
layout: SCENE_REGISTRY.TitleCard.defaultLayout,
|
|
192
|
+
component: "TitleCard",
|
|
193
|
+
props: { title: fallbackTitle },
|
|
194
|
+
overrides: {},
|
|
195
|
+
rationale: `fallback after ${component} props failed: ${lastError.slice(0, 120)}`,
|
|
196
|
+
});
|
|
197
|
+
continue;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
scenes.push({
|
|
201
|
+
id: `scene-${i}`,
|
|
202
|
+
anchor: { startWord: moment.startWord, endWord: moment.endWord },
|
|
203
|
+
layout,
|
|
204
|
+
component,
|
|
205
|
+
props,
|
|
206
|
+
overrides: {},
|
|
207
|
+
rationale: moment.rationale ?? moment.purpose,
|
|
208
|
+
});
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
return { scenes, failures };
|
|
212
|
+
}
|