@ossclip/core 0.1.7 → 0.1.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/blooper.ts +82 -6
- package/src/concat.ts +461 -0
- package/src/cutlist.ts +77 -7
- package/src/index.ts +3 -0
- package/src/overrides.ts +157 -2
- package/src/phonetics.ts +6 -1
- package/src/recut.ts +335 -0
- package/src/retake.ts +607 -0
- package/src/scene-schema.ts +9 -1
- package/src/timemap.ts +37 -0
package/src/retake.ts
ADDED
|
@@ -0,0 +1,607 @@
|
|
|
1
|
+
import { normalizeToken } from "./analyze";
|
|
2
|
+
import { isSentenceEnd, isSentenceStart } from "./clip";
|
|
3
|
+
import { levenshtein } from "./phonetics";
|
|
4
|
+
import type { Analysis, Span, Transcript } from "./schema";
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Deterministic retake collapse (R27 §128) — the sibling of `findBloopSpans`
|
|
8
|
+
* (§122) for the flub the speaker did NOT mark out loud. Consecutive
|
|
9
|
+
* near-identical sentences in the raw transcript are a retake; keep the last
|
|
10
|
+
* complete attempt, cut the rest. See PHASE1-FINDINGS.md §128 for the worked
|
|
11
|
+
* examples and the guard rationale below.
|
|
12
|
+
*
|
|
13
|
+
* Deliberately NOT `soundsSimilar` (§125: `soundsSimilar("builds", "blooper")`
|
|
14
|
+
* scored 0.500 on shared onset alone and cut 86.8% of a real video). A retake
|
|
15
|
+
* pair needs to be *the same words*, not phonetically adjacent ones — token
|
|
16
|
+
* equality here is exact-or-tiny-edit-distance on the ASR text itself, never
|
|
17
|
+
* a sound-alike heuristic. Two independently recorded takes of one line are
|
|
18
|
+
* also NOT guaranteed to differ by an edit distance of 1 or 2 at the phrase
|
|
19
|
+
* level, which is why the comparison is a normalized SEQUENCE similarity
|
|
20
|
+
* (edit distance over the token stream, not the letters) rather than a single
|
|
21
|
+
* fuzzy-string threshold: it tolerates the fuzzy word or two `soundsSimilar`
|
|
22
|
+
* chased, without opening the same phonetic false-positive channel.
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
/** Sequence similarity floor for two attempts to count as the same line. */
|
|
26
|
+
export const RETAKE_SIM_THRESHOLD = 0.8;
|
|
27
|
+
/**
|
|
28
|
+
* Below this many compared tokens, similarity is not evidence of a retake —
|
|
29
|
+
* it's coincidence. "Yes. Yes. Yes." is deliberate emphasis, not three
|
|
30
|
+
* attempts at one line, and at one token apiece it would otherwise clear
|
|
31
|
+
* RETAKE_SIM_THRESHOLD trivially (identical single-token "sequences").
|
|
32
|
+
*/
|
|
33
|
+
export const RETAKE_MIN_TOKENS = 3;
|
|
34
|
+
/** A token must be at least this long before edit-distance fuzz applies. */
|
|
35
|
+
export const TOKEN_FUZZ_MIN_LEN = 5;
|
|
36
|
+
/** Max Levenshtein distance for two long tokens to still count equal. */
|
|
37
|
+
export const TOKEN_FUZZ_MAX_DIST = 1;
|
|
38
|
+
/**
|
|
39
|
+
* Fraction of an instance's own span that must be covered by `analysis.silences`
|
|
40
|
+
* before it is presumed a whisper hallucination rather than a real attempt —
|
|
41
|
+
* the 2026-08-05 field failure: a real take early, then whisper repeating it
|
|
42
|
+
* near-verbatim over dead air later. Read from `analysis.silences`, not
|
|
43
|
+
* `cuttable`: `cuttable`'s transcript veto is exactly the thing that would
|
|
44
|
+
* suppress the signal here (a "word" whisper invented over silence is the
|
|
45
|
+
* conflict the veto exists to resolve the OTHER way), and it is defeated
|
|
46
|
+
* outright in the `windowsDb: []` fallback path (`analyze.ts`) — `silences`
|
|
47
|
+
* is measured straight from the audio and carries neither problem. Not
|
|
48
|
+
* `LevelStats` either: it isn't persisted onto `Analysis`.
|
|
49
|
+
*/
|
|
50
|
+
export const HALLUCINATION_SILENCE_FRAC = 0.65;
|
|
51
|
+
/**
|
|
52
|
+
* Two roles, one number, deliberately: (a) the minimum silence a mid-sentence
|
|
53
|
+
* gap needs before it is treated as a candidate restart boundary — an
|
|
54
|
+
* ordinary breath pause must not fragment one sentence into two "attempts";
|
|
55
|
+
* (b) the max silenceFrac the KEPT survivor itself may carry (stricter than
|
|
56
|
+
* the 0.65 hallucination bar — risk item 7 of the design). A last "complete"
|
|
57
|
+
* instance that is this gappy is exactly as suspect as a fresh restart would
|
|
58
|
+
* be at this boundary, so the same threshold gates both: not tuned twice.
|
|
59
|
+
*/
|
|
60
|
+
export const RESTART_SPLIT_MIN_SIL = 0.35;
|
|
61
|
+
|
|
62
|
+
/** A span of transcript, in word indices (inclusive) and source seconds. */
|
|
63
|
+
export interface RetakeInstance {
|
|
64
|
+
startWord: number;
|
|
65
|
+
endWord: number;
|
|
66
|
+
startSec: number;
|
|
67
|
+
endSec: number;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
export interface RetakeCut extends RetakeInstance {
|
|
71
|
+
/** Sequence similarity (0..1) to the kept instance — the audit trail. */
|
|
72
|
+
similarity: number;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
export interface RetakeHallucination extends RetakeInstance {
|
|
76
|
+
/** Fraction of this span covered by `analysis.silences` — why it was spared. */
|
|
77
|
+
silenceFrac: number;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/** A real instance reported without a cut/keep decision — see `RetakeGroup.kept`. */
|
|
81
|
+
export interface RetakeUndecided extends RetakeInstance {
|
|
82
|
+
silenceFrac: number;
|
|
83
|
+
/**
|
|
84
|
+
* Present when the instance sat in a chain WITH a kept survivor but was
|
|
85
|
+
* spared (§128): the report needs the number to say what the match to the
|
|
86
|
+
* kept instance actually was, whichever rule spared it.
|
|
87
|
+
*/
|
|
88
|
+
similarity?: number;
|
|
89
|
+
/**
|
|
90
|
+
* Why a chain member with a kept survivor was spared (§128):
|
|
91
|
+
* `below-threshold` — scored under RETAKE_SIM_THRESHOLD against the kept
|
|
92
|
+
* instance (the cut-validation rule); `clause-boundary` — matched, but is
|
|
93
|
+
* a NON-final fragment whose same-sentence remainder survives, so the
|
|
94
|
+
* "match" is parallel rhetoric around a mid-sentence pause, not an
|
|
95
|
+
* abandoned take (the abandonment rule). Absent in the report-only
|
|
96
|
+
* (`kept: null`) posture, where silenceFrac is the story.
|
|
97
|
+
*/
|
|
98
|
+
reason?: "below-threshold" | "clause-boundary";
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* One chain of matching attempts at the same line.
|
|
103
|
+
*
|
|
104
|
+
* `kept` is `null` when the chain's LAST complete instance fails the
|
|
105
|
+
* RESTART_SPLIT_MIN_SIL survivor bar — including when no complete instance
|
|
106
|
+
* exists at all. Never cut, never keep, the same posture the hallucination
|
|
107
|
+
* guard already takes, and for the same reason: electing any OTHER survivor
|
|
108
|
+
* silently inverts keep-last (audit fix, §128 — a last complete instance at
|
|
109
|
+
* 0.375 silenceFrac was dropped in favor of an earlier cleaner attempt at a
|
|
110
|
+
* printed 100% match, with no hint the documented convention had flipped).
|
|
111
|
+
* `cuts` is empty in that case and every real instance in the chain is
|
|
112
|
+
* listed in `undecided` instead, so the report can say WHY nothing was
|
|
113
|
+
* decided rather than going silent. When a survivor IS kept, `undecided`
|
|
114
|
+
* holds any chain member that scored below RETAKE_SIM_THRESHOLD against it
|
|
115
|
+
* (see `buildGroup`) — reported, never cut.
|
|
116
|
+
*/
|
|
117
|
+
export interface RetakeGroup {
|
|
118
|
+
kept: RetakeInstance | null;
|
|
119
|
+
cuts: RetakeCut[];
|
|
120
|
+
hallucinated: RetakeHallucination[];
|
|
121
|
+
undecided: RetakeUndecided[];
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
// ---- token comparison -------------------------------------------------
|
|
125
|
+
|
|
126
|
+
function tokensEqual(a: string, b: string): boolean {
|
|
127
|
+
if (a === b) return true;
|
|
128
|
+
if (a.length >= TOKEN_FUZZ_MIN_LEN && b.length >= TOKEN_FUZZ_MIN_LEN) {
|
|
129
|
+
return levenshtein(a, b) <= TOKEN_FUZZ_MAX_DIST;
|
|
130
|
+
}
|
|
131
|
+
return false;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/** Levenshtein distance over TOKENS (word equality, not letters). */
|
|
135
|
+
function tokenEditDistance(a: readonly string[], b: readonly string[]): number {
|
|
136
|
+
if (a.length === 0) return b.length;
|
|
137
|
+
if (b.length === 0) return a.length;
|
|
138
|
+
let prev = Array.from({ length: b.length + 1 }, (_, i) => i);
|
|
139
|
+
for (let i = 1; i <= a.length; i++) {
|
|
140
|
+
const curr = [i];
|
|
141
|
+
for (let j = 1; j <= b.length; j++) {
|
|
142
|
+
curr[j] = Math.min(
|
|
143
|
+
prev[j]! + 1,
|
|
144
|
+
curr[j - 1]! + 1,
|
|
145
|
+
prev[j - 1]! + (tokensEqual(a[i - 1]!, b[j - 1]!) ? 0 : 1),
|
|
146
|
+
);
|
|
147
|
+
}
|
|
148
|
+
prev = curr;
|
|
149
|
+
}
|
|
150
|
+
return prev[b.length]!;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/** Full-sequence similarity: two attempts presumed roughly the same length. */
|
|
154
|
+
function fullSimilarity(a: readonly string[], b: readonly string[]): number {
|
|
155
|
+
const denom = Math.max(a.length, b.length);
|
|
156
|
+
if (denom === 0) return 1;
|
|
157
|
+
return 1 - tokenEditDistance(a, b) / denom;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
/**
|
|
161
|
+
* A partial's tokens against the same-length PREFIX of the other instance —
|
|
162
|
+
* comparing a partial to the other's full length would count everything past
|
|
163
|
+
* where the partial stopped as a mismatch, punishing an abandoned attempt for
|
|
164
|
+
* not having said the rest of the sentence yet.
|
|
165
|
+
*/
|
|
166
|
+
function prefixSimilarity(shorter: readonly string[], longer: readonly string[]): number {
|
|
167
|
+
const n = shorter.length;
|
|
168
|
+
const truncated = longer.slice(0, n);
|
|
169
|
+
const denom = Math.max(n, truncated.length);
|
|
170
|
+
if (denom === 0) return 1;
|
|
171
|
+
return 1 - tokenEditDistance(shorter, truncated) / denom;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// ---- segmentation -------------------------------------------------------
|
|
175
|
+
|
|
176
|
+
interface Instance {
|
|
177
|
+
startWord: number;
|
|
178
|
+
endWord: number;
|
|
179
|
+
startSec: number;
|
|
180
|
+
endSec: number;
|
|
181
|
+
/** Normalized tokens, fillers and a lone `transparentMarker` word dropped. */
|
|
182
|
+
tokens: string[];
|
|
183
|
+
/** Ends at a real sentence-end, vs. a speculative silence sub-split. */
|
|
184
|
+
complete: boolean;
|
|
185
|
+
/**
|
|
186
|
+
* The LAST fragment of its coarse sentence — nothing of that sentence
|
|
187
|
+
* follows it. A non-final fragment always has a same-sentence remainder
|
|
188
|
+
* after it, which is what the abandonment rule in `buildGroup` needs to
|
|
189
|
+
* know (§128): cutting a non-final fragment whose remainder lives on
|
|
190
|
+
* leaves that remainder grammatically orphaned mid-sentence.
|
|
191
|
+
*/
|
|
192
|
+
finalFragment: boolean;
|
|
193
|
+
silenceFrac: number;
|
|
194
|
+
hallucinated: boolean;
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
function silenceOverlap(silences: readonly Span[], start: number, end: number): number {
|
|
198
|
+
let covered = 0;
|
|
199
|
+
for (const s of silences) {
|
|
200
|
+
const lo = Math.max(s.start, start);
|
|
201
|
+
const hi = Math.min(s.end, end);
|
|
202
|
+
if (hi > lo) covered += hi - lo;
|
|
203
|
+
}
|
|
204
|
+
return covered;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
function silenceFraction(silences: readonly Span[], start: number, end: number): number {
|
|
208
|
+
const dur = end - start;
|
|
209
|
+
if (dur <= 0) return 0;
|
|
210
|
+
return Math.min(1, silenceOverlap(silences, start, end) / dur);
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
/**
|
|
214
|
+
* Punctuation sentences, sub-split at internal silence boundaries long enough
|
|
215
|
+
* to be a candidate restart (RESTART_SPLIT_MIN_SIL) — the unpunctuated-partial
|
|
216
|
+
* case: an abandoned attempt has no terminal punctuation of its own, so ASR
|
|
217
|
+
* glues it onto whatever comes next until the NEXT real sentence-end. Only the
|
|
218
|
+
* silence tells you where the restart actually happened. Over-splitting alone
|
|
219
|
+
* can never create a cut — but NOT because a spurious fragment can't match:
|
|
220
|
+
* probe C1 (§128) proved parallel rhetoric makes a clause-boundary fragment
|
|
221
|
+
* match an earlier sentence at a legitimate 1.0. The real backstop is
|
|
222
|
+
* `buildGroup`'s abandonment rule: a non-final fragment whose kept match
|
|
223
|
+
* sits earlier is never cut, so a spurious split ends at a report line,
|
|
224
|
+
* not a shear through live audio.
|
|
225
|
+
*/
|
|
226
|
+
function buildInstances(
|
|
227
|
+
transcript: Transcript,
|
|
228
|
+
analysis: Pick<Analysis, "silences" | "fillers">,
|
|
229
|
+
transparentMarker?: string,
|
|
230
|
+
): Instance[] {
|
|
231
|
+
const words = transcript.words;
|
|
232
|
+
if (words.length === 0) return [];
|
|
233
|
+
const fillerIndices = new Set(analysis.fillers.map((f) => f.wordIndex));
|
|
234
|
+
const marker = transparentMarker ? normalizeToken(transparentMarker) : undefined;
|
|
235
|
+
|
|
236
|
+
const coarse: Array<{ start: number; end: number }> = [];
|
|
237
|
+
let start = 0;
|
|
238
|
+
for (let i = 1; i < words.length; i++) {
|
|
239
|
+
if (isSentenceStart(transcript, i)) {
|
|
240
|
+
coarse.push({ start, end: i - 1 });
|
|
241
|
+
start = i;
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
coarse.push({ start, end: words.length - 1 });
|
|
245
|
+
|
|
246
|
+
const toInstance = (s: number, e: number, complete: boolean, finalFragment: boolean): Instance => {
|
|
247
|
+
const tokens: string[] = [];
|
|
248
|
+
for (let i = s; i <= e; i++) {
|
|
249
|
+
if (fillerIndices.has(i)) continue;
|
|
250
|
+
const norm = normalizeToken(words[i]!.text);
|
|
251
|
+
if (!norm) continue;
|
|
252
|
+
if (marker && norm === marker) continue;
|
|
253
|
+
tokens.push(norm);
|
|
254
|
+
}
|
|
255
|
+
const startSec = words[s]!.start;
|
|
256
|
+
const endSec = words[e]!.end;
|
|
257
|
+
const silenceFrac = silenceFraction(analysis.silences, startSec, endSec);
|
|
258
|
+
return {
|
|
259
|
+
startWord: s,
|
|
260
|
+
endWord: e,
|
|
261
|
+
startSec,
|
|
262
|
+
endSec,
|
|
263
|
+
tokens,
|
|
264
|
+
complete,
|
|
265
|
+
finalFragment,
|
|
266
|
+
silenceFrac,
|
|
267
|
+
hallucinated: silenceFrac >= HALLUCINATION_SILENCE_FRAC,
|
|
268
|
+
};
|
|
269
|
+
};
|
|
270
|
+
|
|
271
|
+
const instances: Instance[] = [];
|
|
272
|
+
for (const sent of coarse) {
|
|
273
|
+
// A sentence that is ALREADY silence-dominated end-to-end is the
|
|
274
|
+
// hallucination shape, not the restart shape: whisper sprinkles sparse
|
|
275
|
+
// word stamps across dead air, which makes EVERY inter-word gap clear
|
|
276
|
+
// RESTART_SPLIT_MIN_SIL. Sub-splitting on that basis would shred it into
|
|
277
|
+
// one-or-two-token fragments, each too short to ever clear
|
|
278
|
+
// RETAKE_MIN_TOKENS again — the hallucination becomes invisible to the
|
|
279
|
+
// very guard built to catch it. So the restart split only runs on a
|
|
280
|
+
// sentence whose OVERALL span isn't itself hallucination-shaped; a
|
|
281
|
+
// hallucinated stretch is instead emitted whole, as one instance, and
|
|
282
|
+
// caught by the ordinary per-instance hallucination check below.
|
|
283
|
+
const wholeFrac = silenceFraction(analysis.silences, words[sent.start]!.start, words[sent.end]!.end);
|
|
284
|
+
const splitAfterSet = new Set<number>();
|
|
285
|
+
if (wholeFrac < HALLUCINATION_SILENCE_FRAC) {
|
|
286
|
+
// Gap-based boundary: a real inter-word gap, where one exists, is still
|
|
287
|
+
// direct evidence of a pause. Kept even though it is nearly inert on
|
|
288
|
+
// field transcripts (below): it costs nothing and the test fixtures
|
|
289
|
+
// that predate the field probe still describe a legal input shape.
|
|
290
|
+
for (let i = sent.start; i < sent.end; i++) {
|
|
291
|
+
const gapDur = silenceOverlap(analysis.silences, words[i]!.end, words[i + 1]!.start);
|
|
292
|
+
if (gapDur >= RESTART_SPLIT_MIN_SIL) splitAfterSet.add(i);
|
|
293
|
+
}
|
|
294
|
+
// Stamp-based boundary (§128, audit fix): whisper `-ml 1` emits
|
|
295
|
+
// contiguous stamps — `parseWhisperJson` clamps `next.start = w.end` —
|
|
296
|
+
// so on a real transcript ~95% of inter-word gaps are exactly zero and
|
|
297
|
+
// the gap check above never sees a mid-sentence restart pause. The
|
|
298
|
+
// pause is still in the audio: stamps stretch over dead air (the same
|
|
299
|
+
// physics the hallucination guard exploits), so `analysis.silences` is
|
|
300
|
+
// read directly against the STAMPED word intervals instead. A silence
|
|
301
|
+
// span overlapping this sentence by at least RESTART_SPLIT_MIN_SIL
|
|
302
|
+
// marks a candidate split after the last word whose stamp begins
|
|
303
|
+
// before the silence does — that word's audio is the last thing said
|
|
304
|
+
// before the pause, whether the dead air was stamped into its own
|
|
305
|
+
// tail, across two contiguous stamps, or into the next word's head.
|
|
306
|
+
const sentStartSec = words[sent.start]!.start;
|
|
307
|
+
const sentEndSec = words[sent.end]!.end;
|
|
308
|
+
for (const s of analysis.silences) {
|
|
309
|
+
const lo = Math.max(s.start, sentStartSec);
|
|
310
|
+
const hi = Math.min(s.end, sentEndSec);
|
|
311
|
+
if (hi - lo < RESTART_SPLIT_MIN_SIL) continue;
|
|
312
|
+
// The scan INCLUDES the sentence-final word: a trailing inter-
|
|
313
|
+
// sentence pause is routinely stamped into that word's stretched
|
|
314
|
+
// tail (the field probe's "Linux."/"gate." shape), and excluding it
|
|
315
|
+
// would displace the split one word left — fragmenting a perfectly
|
|
316
|
+
// good sentence around a pause that is actually AFTER it. A split
|
|
317
|
+
// that lands after the final word is a no-op and is dropped.
|
|
318
|
+
let after = -1;
|
|
319
|
+
for (let i = sent.start; i <= sent.end; i++) {
|
|
320
|
+
if (words[i]!.start < s.start) after = i;
|
|
321
|
+
else break;
|
|
322
|
+
}
|
|
323
|
+
if (after >= sent.start && after < sent.end) splitAfterSet.add(after);
|
|
324
|
+
}
|
|
325
|
+
}
|
|
326
|
+
const splitAfter = [...splitAfterSet].sort((a, b) => a - b);
|
|
327
|
+
let fragStart = sent.start;
|
|
328
|
+
for (const at of splitAfter) {
|
|
329
|
+
instances.push(toInstance(fragStart, at, false, false));
|
|
330
|
+
fragStart = at + 1;
|
|
331
|
+
}
|
|
332
|
+
// The final fragment is only "complete" if the coarse block itself ended
|
|
333
|
+
// at REAL sentence punctuation — a transcript (or clip window) that just
|
|
334
|
+
// runs out of words mid-sentence is a trailing abandoned partial, not a
|
|
335
|
+
// finished take, whatever fragment boundary it happens to land on.
|
|
336
|
+
instances.push(toInstance(fragStart, sent.end, isSentenceEnd(transcript, sent.end), true));
|
|
337
|
+
}
|
|
338
|
+
return instances;
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
// ---- matching -------------------------------------------------------------
|
|
342
|
+
|
|
343
|
+
/**
|
|
344
|
+
* Raw similarity, no threshold — used for the report once a group exists,
|
|
345
|
+
* and for the match gate below.
|
|
346
|
+
*
|
|
347
|
+
* The prefix rule only models one shape: a restart/abandoned partial says
|
|
348
|
+
* FEWER words than the take it restarts. Picking "whichever instance has
|
|
349
|
+
* fewer tokens" as the prefix role — instead of "whichever instance is
|
|
350
|
+
* actually incomplete" — silently applies that same rule to the OPPOSITE
|
|
351
|
+
* shape: an incomplete instance that says MORE words than its counterpart (a
|
|
352
|
+
* continuation/elaboration, or a `--clip` slice that ends mid-sentence after
|
|
353
|
+
* accumulating more words than some earlier complete sentence). Truncating
|
|
354
|
+
* the shorter COMPLETE side's full text down to nothing extra and comparing
|
|
355
|
+
* it against only the incomplete side's matching opening reports a spurious
|
|
356
|
+
* near-1.0 score on two sentences that actually diverge in their second
|
|
357
|
+
* half — verified against a real repro: "Let me show you this." (complete)
|
|
358
|
+
* vs. the unpunctuated continuation "Let me show you this whole thing in
|
|
359
|
+
* detail" scored 1.0 and got the LONGER, more complete continuation cut
|
|
360
|
+
* instead of the short line. Full-sequence comparison scores that
|
|
361
|
+
* divergence honestly instead, whenever the incomplete side is the LONGER
|
|
362
|
+
* one.
|
|
363
|
+
*/
|
|
364
|
+
function rawSimilarity(a: Instance, b: Instance): number {
|
|
365
|
+
if (a.complete && b.complete) return fullSimilarity(a.tokens, b.tokens);
|
|
366
|
+
if (!a.complete && !b.complete) {
|
|
367
|
+
const [shorter, longer] = a.tokens.length <= b.tokens.length ? [a, b] : [b, a];
|
|
368
|
+
return prefixSimilarity(shorter.tokens, longer.tokens);
|
|
369
|
+
}
|
|
370
|
+
const incomplete = a.complete ? b : a;
|
|
371
|
+
const other = a.complete ? a : b;
|
|
372
|
+
if (incomplete.tokens.length <= other.tokens.length) {
|
|
373
|
+
return prefixSimilarity(incomplete.tokens, other.tokens);
|
|
374
|
+
}
|
|
375
|
+
return fullSimilarity(a.tokens, b.tokens);
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
/** Similarity if it clears both the threshold and RETAKE_MIN_TOKENS, else null. */
|
|
379
|
+
function matchScore(a: Instance, b: Instance): number | null {
|
|
380
|
+
const compareLen = Math.min(a.tokens.length, b.tokens.length);
|
|
381
|
+
if (compareLen < RETAKE_MIN_TOKENS) return null;
|
|
382
|
+
const sim = rawSimilarity(a, b);
|
|
383
|
+
return sim >= RETAKE_SIM_THRESHOLD ? sim : null;
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
function toPublic(i: Instance): RetakeInstance {
|
|
387
|
+
return { startWord: i.startWord, endWord: i.endWord, startSec: i.startSec, endSec: i.endSec };
|
|
388
|
+
}
|
|
389
|
+
|
|
390
|
+
/**
|
|
391
|
+
* Kept = the LAST live complete instance, and ONLY if its own silenceFrac
|
|
392
|
+
* clears the stricter RESTART_SPLIT_MIN_SIL survivor bar. When it fails —
|
|
393
|
+
* or the chain has no complete instance at all — the group goes report-only:
|
|
394
|
+
* `kept` is null, nothing cuts, and every real instance is returned in
|
|
395
|
+
* `undecided` with its own silenceFrac. Never fall back to an EARLIER
|
|
396
|
+
* complete instance (audit fix, §128): electing one silently inverts
|
|
397
|
+
* keep-last — a last complete take at silenceFrac 0.375 was dropped for an
|
|
398
|
+
* earlier cleaner attempt at a printed 100% match, with nothing in the
|
|
399
|
+
* report saying the documented convention had flipped. Keep-last is a
|
|
400
|
+
* convention, not a proof (§128's known limits), so when the bar rejects
|
|
401
|
+
* the one instance the convention names, the honest move is to decide
|
|
402
|
+
* nothing and say why.
|
|
403
|
+
*
|
|
404
|
+
* Cut-validation rule (audit fix, §128 — the wildcard-bridge failure):
|
|
405
|
+
* chain membership alone is NOT permission to cut. Matching is
|
|
406
|
+
* non-transitive — a 3-token abandoned fragment scores 1.0 against ANY
|
|
407
|
+
* sentence sharing its opening (the prefix rule), so a chain can drift or
|
|
408
|
+
* bridge across genuinely different sentences. Every member is re-scored
|
|
409
|
+
* against the actual KEPT instance, and only those clearing
|
|
410
|
+
* RETAKE_SIM_THRESHOLD are cut; the rest go to `undecided` (report-only) —
|
|
411
|
+
* executed proof: "Let me show you this." / "Let me show—" / "Let me show
|
|
412
|
+
* you how deploys work here." cut the first REAL, DISTINCT sentence at a
|
|
413
|
+
* printed 50% match before this rule existed.
|
|
414
|
+
*
|
|
415
|
+
* Abandonment rule (audit fix, §128 — probe C1, parallel-structure
|
|
416
|
+
* rhetoric): a similarity gate cannot catch a fragment whose match is
|
|
417
|
+
* GENUINELY 1.0. "If it fails, we retry. If it fails, [0.4s dramatic
|
|
418
|
+
* pause] we give up." — the sub-split shears the second sentence at the
|
|
419
|
+
* comma pause, and "If it fails," legitimately prefix-scores 1.0 against
|
|
420
|
+
* the kept first sentence, because parallel rhetoric repeats the opening
|
|
421
|
+
* on purpose. Cutting it hard-cuts live mid-sentence audio and leaves the
|
|
422
|
+
* grammatically orphaned remainder "we give up." behind. So a fragment is
|
|
423
|
+
* only ABANDONED — hence cuttable — when (a) the kept survivor starts
|
|
424
|
+
* AFTER it (a restart superseded by a later attempt), or (b) it is the
|
|
425
|
+
* FINAL fragment of its coarse sentence, i.e. nothing of its own sentence
|
|
426
|
+
* survives past it. A NON-final fragment whose kept match sits EARLIER is
|
|
427
|
+
* a clause boundary, not an abandoned take: its own sentence continues
|
|
428
|
+
* without it, so it goes to `undecided` (report-only), never `cuts`.
|
|
429
|
+
*/
|
|
430
|
+
function buildGroup(chain: readonly Instance[], hallucinated: readonly Instance[]): RetakeGroup {
|
|
431
|
+
const completes = chain.filter((i) => i.complete);
|
|
432
|
+
const lastComplete = completes[completes.length - 1];
|
|
433
|
+
const kept =
|
|
434
|
+
lastComplete !== undefined && lastComplete.silenceFrac <= RESTART_SPLIT_MIN_SIL
|
|
435
|
+
? lastComplete
|
|
436
|
+
: undefined;
|
|
437
|
+
const hallu: RetakeHallucination[] = hallucinated.map((i) => ({
|
|
438
|
+
...toPublic(i),
|
|
439
|
+
silenceFrac: i.silenceFrac,
|
|
440
|
+
}));
|
|
441
|
+
if (kept === undefined) {
|
|
442
|
+
const undecided: RetakeUndecided[] = chain.map((i) => ({
|
|
443
|
+
...toPublic(i),
|
|
444
|
+
silenceFrac: i.silenceFrac,
|
|
445
|
+
}));
|
|
446
|
+
return { kept: null, cuts: [], hallucinated: hallu, undecided };
|
|
447
|
+
}
|
|
448
|
+
const cuts: RetakeCut[] = [];
|
|
449
|
+
const undecided: RetakeUndecided[] = [];
|
|
450
|
+
for (const i of chain) {
|
|
451
|
+
if (i === kept) continue;
|
|
452
|
+
const sim = matchScore(i, kept);
|
|
453
|
+
if (sim === null) {
|
|
454
|
+
undecided.push({
|
|
455
|
+
...toPublic(i),
|
|
456
|
+
silenceFrac: i.silenceFrac,
|
|
457
|
+
similarity: rawSimilarity(i, kept),
|
|
458
|
+
reason: "below-threshold",
|
|
459
|
+
});
|
|
460
|
+
continue;
|
|
461
|
+
}
|
|
462
|
+
// The abandonment rule (see block comment): a complete instance is
|
|
463
|
+
// always its sentence's final fragment, so this only ever spares the
|
|
464
|
+
// non-final sub-split fragments C1 is about.
|
|
465
|
+
const abandoned = i.finalFragment || kept.startWord > i.endWord;
|
|
466
|
+
if (abandoned) cuts.push({ ...toPublic(i), similarity: sim });
|
|
467
|
+
else
|
|
468
|
+
undecided.push({
|
|
469
|
+
...toPublic(i),
|
|
470
|
+
silenceFrac: i.silenceFrac,
|
|
471
|
+
similarity: sim,
|
|
472
|
+
reason: "clause-boundary",
|
|
473
|
+
});
|
|
474
|
+
}
|
|
475
|
+
return { kept: toPublic(kept), cuts, hallucinated: hallu, undecided };
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
/**
|
|
479
|
+
* Finds retake chains in the transcript. `analysis` is `Pick<Analysis,
|
|
480
|
+
* "silences" | "fillers">` deliberately narrow — this runs on the RAW,
|
|
481
|
+
* pre-repair transcript at both `produce.ts` call sites, same ordering
|
|
482
|
+
* reason as `findBloopSpans` (§122): the repair pass reads a stray restart as
|
|
483
|
+
* an oddity and would rewrite the very pattern this is looking for.
|
|
484
|
+
*
|
|
485
|
+
* Chaining rule: comparison is always against the anchor — the last LIVE
|
|
486
|
+
* (non-hallucinated, non-empty) COMPLETE instance in the chain, or the
|
|
487
|
+
* instance that founded the chain when nothing complete has joined yet.
|
|
488
|
+
* Anything that MATCHES the anchor extends the same chain, and becomes the
|
|
489
|
+
* new anchor only if it is itself COMPLETE (three-take and beyond). An
|
|
490
|
+
* incomplete fragment is matchable and cuttable but NEVER becomes the anchor
|
|
491
|
+
* (audit fix, §128 — the wildcard-bridge failure): its 3-token opening
|
|
492
|
+
* scores 1.0 against ANY sentence starting the same way, so letting it
|
|
493
|
+
* anchor turned an abandoned "Let me show—" into a bridge that chained two
|
|
494
|
+
* genuinely different sentences together and cut one of them. The
|
|
495
|
+
* partial-then-complete ordering still works: a partial can FOUND a chain as
|
|
496
|
+
* its original anchor, and the complete take arriving after it matches (the
|
|
497
|
+
* prefix rule) and takes over as anchor. Anything that does NOT match the
|
|
498
|
+
* anchor starts a fresh one, which is what
|
|
499
|
+
* makes an unrelated sentence in between BLOCK a chain: the next candidate is
|
|
500
|
+
* compared against the un-matching sentence, not the earlier attempt behind
|
|
501
|
+
* it. Filler-only and marker-only instances (zero tokens after normalizing)
|
|
502
|
+
* are skipped entirely — never compared, never become the anchor — so they
|
|
503
|
+
* bridge a chain for free. A hallucinated instance is compared against the
|
|
504
|
+
* anchor (so it can be recognized and reported) but never becomes the anchor
|
|
505
|
+
* itself and never resets it: this is the field-case guard — the anchor
|
|
506
|
+
* stays pinned to the real take, so a later hallucinated repeat can never be
|
|
507
|
+
* elected "last" over it.
|
|
508
|
+
*/
|
|
509
|
+
export function findRetakeGroups(
|
|
510
|
+
transcript: Transcript,
|
|
511
|
+
analysis: Pick<Analysis, "silences" | "fillers">,
|
|
512
|
+
opts: { transparentMarker?: string } = {},
|
|
513
|
+
): RetakeGroup[] {
|
|
514
|
+
const instances = buildInstances(transcript, analysis, opts.transparentMarker);
|
|
515
|
+
const groups: RetakeGroup[] = [];
|
|
516
|
+
|
|
517
|
+
let anchor: Instance | null = null;
|
|
518
|
+
let chain: Instance[] = [];
|
|
519
|
+
let hallucinated: Instance[] = [];
|
|
520
|
+
|
|
521
|
+
const finalize = (): void => {
|
|
522
|
+
if (chain.length >= 2 || (chain.length >= 1 && hallucinated.length > 0)) {
|
|
523
|
+
groups.push(buildGroup(chain, hallucinated));
|
|
524
|
+
}
|
|
525
|
+
chain = [];
|
|
526
|
+
hallucinated = [];
|
|
527
|
+
};
|
|
528
|
+
|
|
529
|
+
for (const inst of instances) {
|
|
530
|
+
if (inst.tokens.length === 0) continue; // filler-only / marker-only: transparent
|
|
531
|
+
if (inst.hallucinated) {
|
|
532
|
+
if (anchor && matchScore(anchor, inst) !== null) {
|
|
533
|
+
if (chain.length === 0) chain.push(anchor);
|
|
534
|
+
hallucinated.push(inst);
|
|
535
|
+
}
|
|
536
|
+
continue;
|
|
537
|
+
}
|
|
538
|
+
if (anchor && matchScore(anchor, inst) !== null) {
|
|
539
|
+
if (chain.length === 0) chain.push(anchor);
|
|
540
|
+
chain.push(inst);
|
|
541
|
+
// §128 wildcard-bridge fix: only a COMPLETE instance may take over as
|
|
542
|
+
// anchor — an incomplete fragment's prefix-matched opening must not
|
|
543
|
+
// become the thing the NEXT sentence is compared against.
|
|
544
|
+
if (inst.complete) anchor = inst;
|
|
545
|
+
continue;
|
|
546
|
+
}
|
|
547
|
+
finalize();
|
|
548
|
+
anchor = inst;
|
|
549
|
+
}
|
|
550
|
+
finalize();
|
|
551
|
+
|
|
552
|
+
return groups;
|
|
553
|
+
}
|
|
554
|
+
|
|
555
|
+
/**
|
|
556
|
+
* One block per group for `report.txt`, beside the blooper lines (§122's
|
|
557
|
+
* design): kept / cut (with similarity) / ignored-as-hallucination (with its
|
|
558
|
+
* silence fraction), quoting the actual words — same audit-trail reasoning as
|
|
559
|
+
* `formatBloopSpan`, sharper here because nothing SAID this was a retake.
|
|
560
|
+
*
|
|
561
|
+
* `group.kept === null` is the report-only case: the LAST complete attempt
|
|
562
|
+
* (or the whole chain, when nothing complete exists) failed the survivor
|
|
563
|
+
* bar, so nothing was cut OR kept — every real instance is listed with its
|
|
564
|
+
* own silenceFrac instead, same shape as the hallucination lines, so the
|
|
565
|
+
* report says WHY nothing was decided rather than going silent. With a
|
|
566
|
+
* survivor, `undecided` members (below-threshold against the kept — §128's
|
|
567
|
+
* cut-validation rule) are listed with their similarity for the same reason:
|
|
568
|
+
* a spared member the user can see beats a wrong cut nobody can.
|
|
569
|
+
*/
|
|
570
|
+
export function formatRetakeGroup(transcript: Transcript, group: RetakeGroup): string {
|
|
571
|
+
const said = (i: RetakeInstance): string =>
|
|
572
|
+
transcript.words
|
|
573
|
+
.slice(i.startWord, i.endWord + 1)
|
|
574
|
+
.map((w) => w.text)
|
|
575
|
+
.join(" ");
|
|
576
|
+
const lines: string[] = [];
|
|
577
|
+
if (group.kept === null) {
|
|
578
|
+
lines.push(
|
|
579
|
+
"no cut: the last complete attempt's own dead-air fraction failed the survivor bar — reporting every attempt instead of guessing which one is real",
|
|
580
|
+
);
|
|
581
|
+
for (const u of group.undecided) {
|
|
582
|
+
lines.push(` attempt (${Math.round(u.silenceFrac * 100)}% silence): "${said(u)}"`);
|
|
583
|
+
}
|
|
584
|
+
} else {
|
|
585
|
+
lines.push(`kept: "${said(group.kept)}"`);
|
|
586
|
+
for (const c of group.cuts) {
|
|
587
|
+
lines.push(` cut (${Math.round(c.similarity * 100)}% match): "${said(c)}"`);
|
|
588
|
+
}
|
|
589
|
+
for (const u of group.undecided) {
|
|
590
|
+
const pct = Math.round((u.similarity ?? 0) * 100);
|
|
591
|
+
// The clause-boundary line must NOT read like a near-miss cut (§128,
|
|
592
|
+
// probe C1): the match there is often a legitimate 100%, and the
|
|
593
|
+
// reason it survived is grammatical, not numeric.
|
|
594
|
+
lines.push(
|
|
595
|
+
u.reason === "clause-boundary"
|
|
596
|
+
? ` not cut (${pct}% match, but its own sentence continues past it — a clause boundary, not an abandoned take): "${said(u)}"`
|
|
597
|
+
: ` not cut (${pct}% match to the kept take — below the cut floor): "${said(u)}"`,
|
|
598
|
+
);
|
|
599
|
+
}
|
|
600
|
+
}
|
|
601
|
+
for (const h of group.hallucinated) {
|
|
602
|
+
lines.push(
|
|
603
|
+
` ignored as hallucination (${Math.round(h.silenceFrac * 100)}% silence): "${said(h)}"`,
|
|
604
|
+
);
|
|
605
|
+
}
|
|
606
|
+
return lines.join("\n");
|
|
607
|
+
}
|
package/src/scene-schema.ts
CHANGED
|
@@ -98,11 +98,19 @@ export const SceneCueSchema = z
|
|
|
98
98
|
h: z.number(),
|
|
99
99
|
})
|
|
100
100
|
.optional(),
|
|
101
|
-
/**
|
|
101
|
+
/**
|
|
102
|
+
* Per-element nudges from the user's edit layer, by `data-edit-id`.
|
|
103
|
+
* Mirrors `ElementTransformSchema` (overrides.ts) field for field,
|
|
104
|
+
* duplicated because this is the RESOLVED cue shape, not the override
|
|
105
|
+
* doc — `hidden` (PLAN Task 2) travels the same path `dx`/`dy`/`scale`
|
|
106
|
+
* already do: `applyOverrides` copies the override's `elements` onto the
|
|
107
|
+
* cue verbatim, so a hand-set flag reaches here unchanged.
|
|
108
|
+
*/
|
|
102
109
|
elements: z.record(z.string(), z.object({
|
|
103
110
|
dx: z.number().optional(),
|
|
104
111
|
dy: z.number().optional(),
|
|
105
112
|
scale: z.number().positive().optional(),
|
|
113
|
+
hidden: z.boolean().optional(),
|
|
106
114
|
})).optional(),
|
|
107
115
|
/**
|
|
108
116
|
* How the video sits in this scene's slot, when the automatic face-aware
|
package/src/timemap.ts
CHANGED
|
@@ -113,3 +113,40 @@ export class TimeMap {
|
|
|
113
113
|
return { start, end };
|
|
114
114
|
}
|
|
115
115
|
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* Rebuild a `TimeMap` from another map's own `spans` — e.g. a PREVIOUS
|
|
119
|
+
* produce run's `render-props.json`, read back to learn which coordinate
|
|
120
|
+
* frame the user's stored splits/pins are CURRENTLY anchored to (PLAN
|
|
121
|
+
* 2026-08-04 Task 4b: `applyUserCuts`'s `priorMap`). The constructor only
|
|
122
|
+
* reasons about `keep`-kind segments when building `spans` — `remove`
|
|
123
|
+
* segments carry no information it uses — so handing back exactly the spans
|
|
124
|
+
* a map once produced, each re-labelled `keep`, reconstructs an identical
|
|
125
|
+
* map without needing the original cutlist's `remove` segments at all.
|
|
126
|
+
*/
|
|
127
|
+
export function mapFromKeptSpans(spans: readonly KeptSpan[]): TimeMap {
|
|
128
|
+
return new TimeMap(spans.map((s) => ({ srcIn: s.srcIn, srcOut: s.srcOut, kind: "keep" as const })));
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* Whether two maps describe the same kept-span structure, within `eps` —
|
|
133
|
+
* the re-anchoring gate (review fix wave, PLAN 2026-08-04 Task 4, finding
|
|
134
|
+
* 3): `applyUserCuts` re-anchors `splits`/pinned timing whenever `priorMap`
|
|
135
|
+
* and this run's final map disagree, span for span, REGARDLESS of whether
|
|
136
|
+
* `cuts` is empty — comparing the maps directly, not the doc, is what keeps
|
|
137
|
+
* "was there drift to correct" and "did correcting it change anything"
|
|
138
|
+
* separate questions.
|
|
139
|
+
*/
|
|
140
|
+
export function mapsClose(a: TimeMap, b: TimeMap, eps: number): boolean {
|
|
141
|
+
if (Math.abs(a.outputDuration - b.outputDuration) > eps) return false;
|
|
142
|
+
if (a.spans.length !== b.spans.length) return false;
|
|
143
|
+
return a.spans.every((sp, i) => {
|
|
144
|
+
const other = b.spans[i]!;
|
|
145
|
+
return (
|
|
146
|
+
Math.abs(sp.srcIn - other.srcIn) <= eps &&
|
|
147
|
+
Math.abs(sp.srcOut - other.srcOut) <= eps &&
|
|
148
|
+
Math.abs(sp.outIn - other.outIn) <= eps &&
|
|
149
|
+
Math.abs(sp.outOut - other.outOut) <= eps
|
|
150
|
+
);
|
|
151
|
+
});
|
|
152
|
+
}
|