@ossclip/core 0.1.31 → 0.1.34

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/restamp.ts ADDED
@@ -0,0 +1,383 @@
1
+ /**
2
+ * Re-stamping a re-transcribed source range onto the words already in
3
+ * `transcript.json` (Phase A, 2026-08-26).
4
+ *
5
+ * WHY THIS EXISTS: inside a kept retake whisper mis-POSITIONS words — "has
6
+ * its" displays while the audio says "could read 50 files", stamps off by
7
+ * 1.5-3s — because the original decode ran over material the cut later
8
+ * revived. Re-decoding just that span fixes the positions; the problem is
9
+ * that everything downstream indexes into the words ARRAY.
10
+ *
11
+ * THE ONE LOAD-BEARING CONSTRAINT: the splice changes STAMPS ONLY — never
12
+ * text, never word count. `ProductionSchema.transcript`'s doctrine is that
13
+ * analysis and the cutlist index into `transcript.words`, and produce's
14
+ * beat/clip caches (`beatSheetCacheKey`, `clipWindowCacheKey`) hash word
15
+ * TEXT. A count-preserving, text-preserving splice therefore keeps every word
16
+ * index valid, every LLM cache warm, `--transcript` replay unaffected and the
17
+ * repairs diff still applicable. Every function here is written to that rule:
18
+ * `alignRestamp` returns exactly as many words, with exactly the same text, as
19
+ * it was given, and `spliceTranscript` refuses a count that does not match.
20
+ *
21
+ * PURE, and browser-safe on purpose: `rekeyCaptionRecords` runs in the
22
+ * editor's `useEdits` reducer (the doc is client-owned — the server never
23
+ * touches `overrides.json`), so nothing in this module may reach a node
24
+ * built-in. That is also why `normalizeAlignToken` is restated here instead of
25
+ * imported from `analyze.ts`, which pulls in `./exec` → `child_process`.
26
+ */
27
+ import { captionKeyFor } from "./overrides";
28
+ import type { OverrideDoc } from "./overrides";
29
+ import type { Transcript, Word } from "./schema";
30
+
31
+ /**
32
+ * The comparison form for alignment: `analyze.ts`'s `normalizeToken`,
33
+ * character for character. Kept byte-identical (rather than "close enough")
34
+ * so a word the filler/retake passes consider the same word is the same word
35
+ * here too — two normalizers that drift are two different transcripts.
36
+ */
37
+ export function normalizeAlignToken(text: string): string {
38
+ return text
39
+ .toLowerCase()
40
+ .replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}-]+$/gu, "");
41
+ }
42
+
43
+ /**
44
+ * A caption key's millisecond, derived from `captionKeyFor` rather than
45
+ * restating its `Math.round(s * 1000)` (§137). The mapping this module emits
46
+ * is consumed by `rekeyCaptionRecords` against keys the EDITOR minted through
47
+ * `captionKeyFor`, so a second rounding rule here is a silent off-by-one-ms
48
+ * that parks every entry.
49
+ */
50
+ export function captionKeyMs(srcStart: number): number {
51
+ return Number(captionKeyFor(srcStart).slice(1));
52
+ }
53
+
54
+ /** One source anchor that MOVED, at `captionKeyFor`'s ms quantization. */
55
+ export interface StampMove {
56
+ fromMs: number;
57
+ toMs: number;
58
+ }
59
+
60
+ export interface RestampResult {
61
+ /** Same length, same text as `oldWords` — only `start`/`end` differ. */
62
+ words: Word[];
63
+ /** Every `srcStart` that moved; unmoved words are deliberately absent. */
64
+ mapping: StampMove[];
65
+ /** What the alignment could not do exactly, in the user's language. */
66
+ reports: string[];
67
+ }
68
+
69
+ /** Where an old word's stamps came from — see `alignRestamp`'s two cases. */
70
+ type Alignment = { kind: "matched"; newIndex: number } | { kind: "gap" };
71
+
72
+ /**
73
+ * Longest common subsequence over normalized tokens, as index pairs.
74
+ *
75
+ * MONOTONE BY CONSTRUCTION, which is the property the whole splice rests on:
76
+ * an alignment that could cross would let a later old word take an earlier new
77
+ * stamp and hand the transcript a non-monotone word list, which
78
+ * `captions.ts`'s line packing reads as gibberish. Classic O(n*m) DP —
79
+ * a re-transcribed range is a handful of seconds of speech (tens of words), so
80
+ * the table is tiny and the simple algorithm is the readable one.
81
+ */
82
+ function lcsPairs(a: readonly string[], b: readonly string[]): Array<[number, number]> {
83
+ const n = a.length;
84
+ const m = b.length;
85
+ const table: number[][] = Array.from({ length: n + 1 }, () => new Array<number>(m + 1).fill(0));
86
+ for (let i = n - 1; i >= 0; i--) {
87
+ for (let j = m - 1; j >= 0; j--) {
88
+ table[i]![j] = a[i] === b[j]
89
+ ? table[i + 1]![j + 1]! + 1
90
+ : Math.max(table[i + 1]![j]!, table[i]![j + 1]!);
91
+ }
92
+ }
93
+ const pairs: Array<[number, number]> = [];
94
+ let i = 0;
95
+ let j = 0;
96
+ while (i < n && j < m) {
97
+ if (a[i] === b[j]) {
98
+ pairs.push([i, j]);
99
+ i++;
100
+ j++;
101
+ } else if (table[i + 1]![j]! >= table[i]![j + 1]!) {
102
+ i++;
103
+ } else {
104
+ j++;
105
+ }
106
+ }
107
+ return pairs;
108
+ }
109
+
110
+ /**
111
+ * Re-stamp `oldWords` from a fresh decode of the same audio span.
112
+ *
113
+ * `newWords` carry CLIP-RELATIVE stamps (whisper decodes the sliced wav and
114
+ * knows nothing about where the slice came from), so `spanStart` — the slice's
115
+ * source second — is added here. Doing the offset inside the pure function
116
+ * rather than at the call site is what lets the whole "did the stamps land
117
+ * where the audio is" question be tested without a whisper binary
118
+ * (`openCommand`/`openInBrowser`, CLAUDE.md).
119
+ *
120
+ * Two cases, and only two:
121
+ * - MATCHED (the old word's normalized token is in the LCS): it takes the new
122
+ * word's stamps verbatim. This is the case the feature exists for.
123
+ * - GAP (the decode said something else here): the old TEXT is kept — the
124
+ * count/text constraint above is absolute — and its stamps are INTERPOLATED
125
+ * evenly across the interval its matched neighbours left free. Interpolating
126
+ * is a guess about position within a known interval; rewriting the text
127
+ * would be a guess about what was said, and the second one silently
128
+ * invalidates every word index in the production.
129
+ *
130
+ * NO ANCHORS AT ALL (empty LCS, or an empty decode) is refused rather than
131
+ * guessed at: the old stamps are returned untouched with a report. Simple and
132
+ * reported beats clever and silent — a range whose decode agrees with nothing
133
+ * is exactly where a stretched-to-fit interpolation would move every caption
134
+ * onto the wrong word.
135
+ */
136
+ export function alignRestamp(
137
+ oldWords: readonly Word[],
138
+ newWords: readonly Word[],
139
+ spanStart: number,
140
+ ): RestampResult {
141
+ const reports: string[] = [];
142
+ if (oldWords.length === 0) return { words: [], mapping: [], reports };
143
+
144
+ const pairs = lcsPairs(
145
+ oldWords.map((w) => normalizeAlignToken(w.text)),
146
+ newWords.map((w) => normalizeAlignToken(w.text)),
147
+ );
148
+ if (pairs.length === 0) {
149
+ reports.push(
150
+ `re-transcription of ${spanStart.toFixed(3)}s matched none of the ${oldWords.length} ` +
151
+ `word(s) already there — stamps left as they were`,
152
+ );
153
+ return { words: oldWords.map((w) => ({ ...w })), mapping: [], reports };
154
+ }
155
+
156
+ const at = new Map<number, number>(pairs.map(([o, n]) => [o, n]));
157
+ const plan: Alignment[] = oldWords.map((_, i) => {
158
+ const newIndex = at.get(i);
159
+ return newIndex === undefined ? { kind: "gap" as const } : { kind: "matched" as const, newIndex };
160
+ });
161
+ // Clip-relative → source seconds, once, here. Clamped at 0 because a slice
162
+ // that starts at 0 plus whisper's occasional tiny negative is still a
163
+ // `WordSchema.start` that must parse (`nonnegative`).
164
+ const newStart = (i: number): number => Math.max(0, newWords[i]!.start + spanStart);
165
+ const newEnd = (i: number): number => Math.max(0, newWords[i]!.end + spanStart);
166
+
167
+ const words: Word[] = oldWords.map((w) => ({ ...w }));
168
+ let squeezed = 0;
169
+ let i = 0;
170
+ while (i < words.length) {
171
+ const step = plan[i]!;
172
+ if (step.kind === "matched") {
173
+ words[i] = { ...words[i]!, start: newStart(step.newIndex), end: newEnd(step.newIndex) };
174
+ i++;
175
+ continue;
176
+ }
177
+ // The whole run of consecutive gap words shares one interval, so find its
178
+ // end before spending any of it.
179
+ let j = i;
180
+ while (j < words.length && plan[j]!.kind === "gap") j++;
181
+ const before = i > 0 ? plan[i - 1] : undefined;
182
+ const after = j < words.length ? plan[j] : undefined;
183
+ // Outside the matched region the interval is bounded by the SPAN, not by a
184
+ // neighbour: a leading gap can start no earlier than the slice does, and a
185
+ // trailing one can end no later than the decode heard anything.
186
+ const left = before?.kind === "matched" ? newEnd(before.newIndex) : Math.max(0, spanStart);
187
+ const right = after?.kind === "matched"
188
+ ? newStart(after.newIndex)
189
+ : newWords.length > 0
190
+ ? newEnd(newWords.length - 1)
191
+ : left;
192
+ const span = right - left;
193
+ if (span <= 0) squeezed += j - i;
194
+ // Equal slices, not old-duration-proportional: the old durations are the
195
+ // very thing this run has no evidence for (the decode disagreed about the
196
+ // words), so weighting by them dresses up a guess as a measurement. When
197
+ // the interval is empty — the decode dropped words the old transcript has
198
+ // — every word in the run collapses to a zero-length stamp at `left`,
199
+ // which keeps the list monotone and is REPORTED below rather than papered
200
+ // over by pushing past the next matched anchor.
201
+ const slice = span > 0 ? span / (j - i) : 0;
202
+ for (let k = i; k < j; k++) {
203
+ const s = left + slice * (k - i);
204
+ words[k] = { ...words[k]!, start: s, end: s + slice };
205
+ }
206
+ i = j;
207
+ }
208
+ if (squeezed > 0) {
209
+ reports.push(
210
+ `${squeezed} word(s) the re-transcription did not hear got zero-length stamps — ` +
211
+ `the decode has no room between the words it did hear`,
212
+ );
213
+ }
214
+
215
+ const mapping: StampMove[] = [];
216
+ for (let k = 0; k < words.length; k++) {
217
+ const fromMs = captionKeyMs(oldWords[k]!.start);
218
+ const toMs = captionKeyMs(words[k]!.start);
219
+ if (fromMs !== toMs) mapping.push({ fromMs, toMs });
220
+ }
221
+ return { words, mapping, reports };
222
+ }
223
+
224
+ export interface RekeyResult {
225
+ doc: OverrideDoc;
226
+ reports: string[];
227
+ }
228
+
229
+ /** `w123` → 123; anything else (a legacy positional key) → null. */
230
+ function keyMs(key: string): number | null {
231
+ const m = /^w(-?\d+)$/.exec(key);
232
+ return m ? Number(m[1]) : null;
233
+ }
234
+
235
+ /**
236
+ * Move every caption record that is anchored to a stamp the splice moved.
237
+ *
238
+ * The map is EXACT — it comes out of the splice itself, not out of a radius
239
+ * search like `migrateCaptionKeys` — so there is nothing to guess: a record
240
+ * keyed `w6000` whose word now starts at 6.42s belongs at `w6420` and nowhere
241
+ * else. What survives from §137 is its REFUSAL rule: two old stamps can round
242
+ * onto one new millisecond (the decode pulled two words together), and rather
243
+ * than let the second entry silently overwrite the first, the loser is PARKED
244
+ * at its original key and reported. A parked entry is stale, not lost — its
245
+ * own `was` guard will drop it with a report at apply time — and the user can
246
+ * see both keys named.
247
+ *
248
+ * Applies to the five source-keyed caption records: `captions`,
249
+ * `captionWordsHidden`, `captionLineTiming` and `captionLineWindows` (both
250
+ * line head keys) and `captionRangeEdits` (`fromKey`/`toKey` endpoints).
251
+ * `splits`/`cuts` are NOT re-keyed: they anchor to a moment of the FOOTAGE,
252
+ * which a re-decode does not move.
253
+ *
254
+ * A window's VALUE is left alone while its key moves, and the asymmetry is the
255
+ * point: the key names the WORD the caption belongs to (a stamp the re-decode
256
+ * just corrected), the value is where the user placed that caption against the
257
+ * AUDIO (which the re-decode did not touch).
258
+ */
259
+ export function rekeyCaptionRecords(doc: OverrideDoc, mapping: readonly StampMove[]): RekeyResult {
260
+ const reports: string[] = [];
261
+ if (mapping.length === 0) return { doc, reports };
262
+ const moves = new Map<number, number>();
263
+ for (const m of mapping) moves.set(m.fromMs, m.toMs);
264
+
265
+ /**
266
+ * Re-key one record. Entries are processed in ascending original ms so the
267
+ * outcome does not depend on JSON key order, and the first claimant of a
268
+ * target key wins.
269
+ */
270
+ const rekeyRecord = <T>(record: Record<string, T>, label: string): Record<string, T> => {
271
+ const out: Record<string, T> = {};
272
+ const entries = Object.entries(record).sort((a, b) => (keyMs(a[0]) ?? 0) - (keyMs(b[0]) ?? 0));
273
+ for (const [key, value] of entries) {
274
+ const ms = keyMs(key);
275
+ const to = ms === null ? undefined : moves.get(ms);
276
+ const want = to === undefined ? key : `w${to}`;
277
+ if (!(want in out)) {
278
+ out[want] = value;
279
+ continue;
280
+ }
281
+ // Target taken. Park at the original key when that is still free —
282
+ // never overwrite the entry that got there first (§137's never-misapply
283
+ // rule), and never drop the user's work silently.
284
+ if (!(key in out)) {
285
+ out[key] = value;
286
+ reports.push(
287
+ `${label} entry ${key} could not move to ${want} — another entry is already there; ` +
288
+ `left where it was`,
289
+ );
290
+ } else {
291
+ reports.push(`${label} entry ${key} collided on both ${want} and its own key — dropped`);
292
+ }
293
+ }
294
+ return out;
295
+ };
296
+
297
+ const moved = (key: string): string => {
298
+ const ms = keyMs(key);
299
+ const to = ms === null ? undefined : moves.get(ms);
300
+ return to === undefined ? key : `w${to}`;
301
+ };
302
+
303
+ return {
304
+ doc: {
305
+ ...doc,
306
+ captions: rekeyRecord(doc.captions, "caption retype"),
307
+ captionWordsHidden: rekeyRecord(doc.captionWordsHidden, "caption hide"),
308
+ captionLineTiming: rekeyRecord(doc.captionLineTiming, "caption timing"),
309
+ captionLineWindows: rekeyRecord(doc.captionLineWindows, "caption window"),
310
+ // An array, not a record, so there is no key to collide on: BOTH
311
+ // endpoints move independently and identity stays the `(fromKey, toKey)`
312
+ // pair the entry already had.
313
+ captionRangeEdits: doc.captionRangeEdits.map((e) => ({
314
+ ...e,
315
+ fromKey: moved(e.fromKey),
316
+ toKey: moved(e.toKey),
317
+ })),
318
+ },
319
+ reports,
320
+ };
321
+ }
322
+
323
+ /**
324
+ * The contiguous index range of `words` that lies wholly inside `[srcIn,
325
+ * srcOut]` — the ONE definition of "the words in this span", shared by the
326
+ * server picking what to re-align and `spliceTranscript` putting it back. Two
327
+ * copies of this predicate is how the two halves would splice different runs.
328
+ *
329
+ * WHOLLY inside, not overlapping: a word straddling the boundary has audio
330
+ * outside the slice, so the fresh decode never heard all of it and has no
331
+ * business re-stamping it. Monotone words make the answer contiguous.
332
+ */
333
+ export function wordsInSpan(
334
+ words: readonly Word[],
335
+ srcIn: number,
336
+ srcOut: number,
337
+ ): { from: number; to: number } {
338
+ let from = words.length;
339
+ let to = 0;
340
+ for (let i = 0; i < words.length; i++) {
341
+ const w = words[i]!;
342
+ if (w.start >= srcIn && w.end <= srcOut) {
343
+ if (i < from) from = i;
344
+ to = i + 1;
345
+ }
346
+ }
347
+ // Nothing in the span: an empty range AT ZERO, not `words.length..0`, so a
348
+ // caller that splices it back unconditionally is a no-op rather than a throw.
349
+ return from === words.length ? { from: 0, to: 0 } : { from, to };
350
+ }
351
+
352
+ /**
353
+ * Put a re-stamped run back into the transcript.
354
+ *
355
+ * THROWS on a count mismatch rather than accepting it: a splice that changes
356
+ * the word count invalidates every scene anchor and cutlist index in the
357
+ * production (see this module's header), so a caller that produced the wrong
358
+ * number of words is a programmer error and must not reach disk. `language`
359
+ * and every word outside the range are carried through untouched.
360
+ */
361
+ export function spliceTranscript(
362
+ transcript: Transcript,
363
+ range: { from: number; to: number },
364
+ restamped: readonly Word[],
365
+ ): Transcript {
366
+ const { from, to } = range;
367
+ if (from < 0 || to > transcript.words.length || from > to) {
368
+ throw new Error(`spliceTranscript: range ${from}..${to} is outside 0..${transcript.words.length}`);
369
+ }
370
+ if (restamped.length !== to - from) {
371
+ throw new Error(
372
+ `spliceTranscript: stamps-only splice needs ${to - from} word(s), got ${restamped.length}`,
373
+ );
374
+ }
375
+ return {
376
+ ...transcript,
377
+ words: [
378
+ ...transcript.words.slice(0, from),
379
+ ...restamped.map((w) => ({ ...w })),
380
+ ...transcript.words.slice(to),
381
+ ],
382
+ };
383
+ }