@ossclip/core 0.1.25 → 0.1.26
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/concat.ts +90 -3
- package/src/overrides.ts +728 -0
- package/src/phonetics.ts +101 -1
- package/src/producer/repair.ts +19 -10
- package/src/recut.ts +61 -0
- package/src/transcribe.ts +71 -5
package/package.json
CHANGED
package/src/concat.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { existsSync } from "node:fs";
|
|
2
2
|
import { readdir, readFile, rename, rm, stat, writeFile } from "node:fs/promises";
|
|
3
|
-
import { join } from "node:path";
|
|
3
|
+
import { isAbsolute, join, relative, resolve } from "node:path";
|
|
4
4
|
import { z } from "zod/v4";
|
|
5
5
|
import { probe, type IngestTools } from "./ingest";
|
|
6
6
|
import { run } from "./exec";
|
|
@@ -42,6 +42,78 @@ export interface ConcatEntry {
|
|
|
42
42
|
size: number;
|
|
43
43
|
}
|
|
44
44
|
|
|
45
|
+
/**
|
|
46
|
+
* Whether a produce output path lands INSIDE the folder being produced.
|
|
47
|
+
* 2026-08-18 field cascade: a folder input's clips are re-enumerated on
|
|
48
|
+
* EVERY run, so an `--out` written into that folder became a 7th "source
|
|
49
|
+
* clip" on the next run — the folder-content hash changed, produce minted a
|
|
50
|
+
* fresh workdir with EMPTY overrides, and the render silently dropped the
|
|
51
|
+
* user's saved edits while the output's duration doubled. It cascaded three
|
|
52
|
+
* times before the doubling duration was connected to the out path.
|
|
53
|
+
*
|
|
54
|
+
* Containment via `path.relative`, the same idiom as edit.ts's `isInside`:
|
|
55
|
+
* a `startsWith` string-prefix test has no separator boundary and is fooled
|
|
56
|
+
* by a sibling folder that merely shares a prefix (`/a/Clips` vs
|
|
57
|
+
* `/a/Clips-old/out.mp4`) — `child` is inside `parent` iff the relative path
|
|
58
|
+
* from one to the other never has to climb out with `..`. Both arguments are
|
|
59
|
+
* resolved here so a relative path is judged against cwd; `~` expansion is
|
|
60
|
+
* the CALLER's job (the 2026-08-16 rule in the CLI's paths.ts — expandHome
|
|
61
|
+
* at the call site — which also keeps core homedir-free).
|
|
62
|
+
*/
|
|
63
|
+
export function outPathInsideInput(outPath: string, inputDir: string): boolean {
|
|
64
|
+
const rel = relative(resolve(inputDir), resolve(outPath));
|
|
65
|
+
return rel === "" || (!rel.startsWith("..") && !isAbsolute(rel));
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* The refusal for `outPathInsideInput`, shared verbatim by produce's own
|
|
70
|
+
* gate and the edit server's `/api/render` 400 — one message, one place, so
|
|
71
|
+
* the two boundaries can never describe the same hazard differently. Names
|
|
72
|
+
* the actual risk (the field cascade above) rather than just "invalid path",
|
|
73
|
+
* and suggests the exact default a flag-less run would pick (produce's
|
|
74
|
+
* `defaultOutPath` shape: at most the LAST dot-segment replaced).
|
|
75
|
+
*/
|
|
76
|
+
/**
|
|
77
|
+
* The default output path for an input: beside it, `.ossclip.mp4` suffix.
|
|
78
|
+
* Trailing separators are stripped FIRST (2026-08-18 field case, second
|
|
79
|
+
* report): shells tab-complete folders with the slash, and the bare regex
|
|
80
|
+
* appended after it — the "default" landed INSIDE the folder as a hidden
|
|
81
|
+
* `.ossclip.mp4` dotfile, the exact self-ingesting shape
|
|
82
|
+
* `outPathInsideInput` exists to refuse. One definition shared by produce's
|
|
83
|
+
* `defaultOutPath` and the refusal message below, so the suggestion can
|
|
84
|
+
* never disagree with what omitting `--out` actually does.
|
|
85
|
+
*/
|
|
86
|
+
export function ossclipOutputPathFor(input: string): string {
|
|
87
|
+
return input.replace(/[/\\]+$/, "").replace(/(\.[^.]+)?$/, ".ossclip.mp4");
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
export function outInsideInputFolderMessage(folder: string): string {
|
|
91
|
+
const suggestion = ossclipOutputPathFor(folder);
|
|
92
|
+
return (
|
|
93
|
+
`refusing to write the output inside the input folder ${folder} — the next ` +
|
|
94
|
+
`run would ingest this output as a source clip and re-plan from scratch, ` +
|
|
95
|
+
`abandoning your edits (the folder's content hash changes, so produce mints ` +
|
|
96
|
+
`a fresh workdir with empty overrides). Write it beside the folder instead, ` +
|
|
97
|
+
`e.g. --out ${suggestion}`
|
|
98
|
+
);
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* Defense-in-depth BEHIND the out-path refusal above (same 2026-08-18 field
|
|
103
|
+
* cascade): a produce output already sitting in the folder — written by a
|
|
104
|
+
* pre-fix run, or moved there by hand — must never be ingested as a source
|
|
105
|
+
* clip, and must be filtered BEFORE `folderManifestKey` sees the entries so
|
|
106
|
+
* its presence can't re-key the workdir either. Matches any name carrying
|
|
107
|
+
* the `.ossclip` marker: `X.ossclip.mp4`, `X.ossclip_fixed.mp4`, and
|
|
108
|
+
* `X.ossclip.mp4.partial.mp4` leftovers all qualify (bare `.ossclip.mp4` is
|
|
109
|
+
* a dotfile and already skipped upstream). A custom-named output
|
|
110
|
+
* (`--out final.mp4`) is undetectable here — the out-path refusal is the
|
|
111
|
+
* real gate; this only keeps a pre-fix folder from compounding further.
|
|
112
|
+
*/
|
|
113
|
+
export function isOssclipOutputName(name: string): boolean {
|
|
114
|
+
return name.includes(".ossclip");
|
|
115
|
+
}
|
|
116
|
+
|
|
45
117
|
/**
|
|
46
118
|
* Order clips for concatenation. `name` (default, per the field request "sort
|
|
47
119
|
* them by name or date modified, name being default") is a PLAIN codepoint
|
|
@@ -321,6 +393,8 @@ export interface FolderListing {
|
|
|
321
393
|
entries: ConcatEntry[];
|
|
322
394
|
/** Files skipped for not matching a video extension (dotfiles excluded). */
|
|
323
395
|
nonVideoCount: number;
|
|
396
|
+
/** Video files skipped as produce's own outputs (`isOssclipOutputName`). */
|
|
397
|
+
ossclipOutputCount: number;
|
|
324
398
|
}
|
|
325
399
|
|
|
326
400
|
/**
|
|
@@ -339,6 +413,7 @@ export interface FolderListing {
|
|
|
339
413
|
export async function listFolderVideos(folder: string): Promise<FolderListing> {
|
|
340
414
|
const dirents = await readdir(folder, { withFileTypes: true });
|
|
341
415
|
let nonVideoCount = 0;
|
|
416
|
+
let ossclipOutputCount = 0;
|
|
342
417
|
const entries: ConcatEntry[] = [];
|
|
343
418
|
for (const d of dirents) {
|
|
344
419
|
if (d.name.startsWith(".")) continue;
|
|
@@ -357,11 +432,19 @@ export async function listFolderVideos(folder: string): Promise<FolderListing> {
|
|
|
357
432
|
nonVideoCount++;
|
|
358
433
|
continue;
|
|
359
434
|
}
|
|
435
|
+
// Skipped HERE, before the entries ever exist — never in concatFolder —
|
|
436
|
+
// so an ossclip output left in the folder can neither become a source
|
|
437
|
+
// clip nor perturb the workdir hash `folderManifestKey` derives from
|
|
438
|
+
// this listing (2026-08-18 field cascade, see isOssclipOutputName).
|
|
439
|
+
if (isOssclipOutputName(d.name)) {
|
|
440
|
+
ossclipOutputCount++;
|
|
441
|
+
continue;
|
|
442
|
+
}
|
|
360
443
|
const st = await stat(join(folder, d.name));
|
|
361
444
|
entries.push({ name: d.name, mtimeMs: st.mtimeMs, size: st.size });
|
|
362
445
|
}
|
|
363
446
|
if (entries.length === 0) throw noVideoFilesError(folder);
|
|
364
|
-
return { entries, nonVideoCount };
|
|
447
|
+
return { entries, nonVideoCount, ossclipOutputCount };
|
|
365
448
|
}
|
|
366
449
|
|
|
367
450
|
export interface FolderConcatResult {
|
|
@@ -371,6 +454,8 @@ export interface FolderConcatResult {
|
|
|
371
454
|
clips: Array<{ name: string; durationSec: number }>;
|
|
372
455
|
/** Files skipped for not matching a video extension. */
|
|
373
456
|
nonVideoCount: number;
|
|
457
|
+
/** Video files skipped as produce's own outputs (`isOssclipOutputName`). */
|
|
458
|
+
ossclipOutputCount: number;
|
|
374
459
|
/** True when the existing `source-concat.mp4` was reused, not rebuilt. */
|
|
375
460
|
cached: boolean;
|
|
376
461
|
/** The concat's own total duration (ffprobe'd from the output). */
|
|
@@ -393,7 +478,7 @@ export async function concatFolder(
|
|
|
393
478
|
sort: "name" | "mtime",
|
|
394
479
|
target: { w: number; h: number },
|
|
395
480
|
): Promise<FolderConcatResult> {
|
|
396
|
-
const { entries: current, nonVideoCount } = listing;
|
|
481
|
+
const { entries: current, nonVideoCount, ossclipOutputCount } = listing;
|
|
397
482
|
if (current.length === 0) throw noVideoFilesError(folder);
|
|
398
483
|
const order = planFolderConcat(current, sort);
|
|
399
484
|
|
|
@@ -408,6 +493,7 @@ export async function concatFolder(
|
|
|
408
493
|
path: outPath,
|
|
409
494
|
clips: order.map((name) => ({ name, durationSec: byName.get(name)!.durationSec })),
|
|
410
495
|
nonVideoCount,
|
|
496
|
+
ossclipOutputCount,
|
|
411
497
|
cached: true,
|
|
412
498
|
durationSec: outProbe.duration,
|
|
413
499
|
};
|
|
@@ -461,6 +547,7 @@ export async function concatFolder(
|
|
|
461
547
|
path: outPath,
|
|
462
548
|
clips: order.map((name, i) => ({ name, durationSec: probes[i]!.duration })),
|
|
463
549
|
nonVideoCount,
|
|
550
|
+
ossclipOutputCount,
|
|
464
551
|
cached: false,
|
|
465
552
|
durationSec: outProbe.duration,
|
|
466
553
|
};
|
package/src/overrides.ts
CHANGED
|
@@ -152,6 +152,37 @@ export const CaptionEditSchema = z.object({
|
|
|
152
152
|
});
|
|
153
153
|
export type CaptionEdit = z.infer<typeof CaptionEditSchema>;
|
|
154
154
|
|
|
155
|
+
/**
|
|
156
|
+
* A free-text rewrite of a contiguous caption word RUN (2026-08-18) — the one
|
|
157
|
+
* deliberate relaxation of the 1:1 retype contract, for range edits only.
|
|
158
|
+
* Single-word retype (`CaptionEditSchema` above) is untouched, and
|
|
159
|
+
* `transcript.words` is NEVER spliced — scene anchors are raw indices into it
|
|
160
|
+
* — so everything happens on the derived `CaptionLine[]`
|
|
161
|
+
* (`applyCaptionRangeEdits` below).
|
|
162
|
+
*
|
|
163
|
+
* Endpoints are anchored by §137 source-time keys (`captionKeyFor`), so a
|
|
164
|
+
* user cut elsewhere cannot shift the run. `was` is the NFC-normalized,
|
|
165
|
+
* space-joined BASE text of the run — the `captionEditWas` base-truth rule,
|
|
166
|
+
* run-wide: the reducer scrubs every per-word retype inside the interval in
|
|
167
|
+
* the same commit that stores the entry, so the run `applyCaptionRangeEdits`
|
|
168
|
+
* reads at apply time IS the base run, and a live (post-retype) join would
|
|
169
|
+
* fail the guard forever. A WHOLE-RUN stale guard: if any word in the run is
|
|
170
|
+
* re-worded or cut later, the entire edit is reported dropped, never
|
|
171
|
+
* partially guessed at. Identity is the `(fromKey, toKey)`
|
|
172
|
+
* pair — retyping the run back to its `was` DELETES the entry (the
|
|
173
|
+
* clearVideo/`patchCaption` rule). An array like `cuts`, `.default([])` so
|
|
174
|
+
* every pre-existing overrides.json parses byte-identically. NEVER
|
|
175
|
+
* legacy-keyed: the field postdates §137, so `migrateCaptionKeys` must not
|
|
176
|
+
* process it — there are no positional range edits to upgrade.
|
|
177
|
+
*/
|
|
178
|
+
export const CaptionRangeEditSchema = z.object({
|
|
179
|
+
fromKey: z.string().regex(/^w\d+$/),
|
|
180
|
+
toKey: z.string().regex(/^w\d+$/),
|
|
181
|
+
text: z.string().min(1).max(400),
|
|
182
|
+
was: z.string(),
|
|
183
|
+
});
|
|
184
|
+
export type CaptionRangeEdit = z.infer<typeof CaptionRangeEditSchema>;
|
|
185
|
+
|
|
155
186
|
/**
|
|
156
187
|
* The `was` a caption edit should store (R15 §59). The FIRST edit's `was` is
|
|
157
188
|
* the base truth (the word as transcribed); every later re-edit of the same
|
|
@@ -169,6 +200,25 @@ export function captionEditWas(
|
|
|
169
200
|
return captions[key]?.was ?? seen;
|
|
170
201
|
}
|
|
171
202
|
|
|
203
|
+
/**
|
|
204
|
+
* The `was` a RANGE edit should store — `captionEditWas` for the
|
|
205
|
+
* `(fromKey, toKey)` pair. The first edit's `was` is the base truth; a
|
|
206
|
+
* re-edit of the SAME run (its endpoints are re-minted verbatim, see
|
|
207
|
+
* `applyCaptionRangeEdits`' srcStart minting) sees the LIVE, already-rewritten
|
|
208
|
+
* text, and storing that as `was` would stale the guard against the base
|
|
209
|
+
* lines the next apply runs on. Preserving the existing pair's `was` keeps
|
|
210
|
+
* the guard anchored to the base — and makes "retyped back to the original"
|
|
211
|
+
* detectable, which is when the entry should clear entirely.
|
|
212
|
+
*/
|
|
213
|
+
export function captionRangeEditWas(
|
|
214
|
+
rangeEdits: readonly CaptionRangeEdit[],
|
|
215
|
+
fromKey: string,
|
|
216
|
+
toKey: string,
|
|
217
|
+
seen: string,
|
|
218
|
+
): string {
|
|
219
|
+
return rangeEdits.find((e) => e.fromKey === fromKey && e.toKey === toKey)?.was ?? seen;
|
|
220
|
+
}
|
|
221
|
+
|
|
172
222
|
/**
|
|
173
223
|
* The id a pre-§137 split gets when it is upgraded: the output milliseconds of
|
|
174
224
|
* whatever `at` the file holds NOW.
|
|
@@ -256,6 +306,76 @@ export const OverrideDocSchema = z.object({
|
|
|
256
306
|
scenes: z.record(z.string(), SceneOverrideSchema).default({}),
|
|
257
307
|
/** Retyped caption words, keyed by the word's source time (§137). */
|
|
258
308
|
captions: z.record(z.string(), CaptionEditSchema).default({}),
|
|
309
|
+
/**
|
|
310
|
+
* Per-word caption HIDES ("delete word from captions") — non-destructive:
|
|
311
|
+
* the word stays in the transcript and in the video's audio; only the
|
|
312
|
+
* rendered caption drops it. Keyed by the word's source time
|
|
313
|
+
* (`captionKeyFor`, §137) like `captions` above, so a user cut never
|
|
314
|
+
* shifts a hide onto a different word. `was` is the LIVE (post-retype)
|
|
315
|
+
* text at hide time — hides apply AFTER retypes (`applyCaptionLayers`
|
|
316
|
+
* below) — the same stale-guard contract as `CaptionEditSchema.was`:
|
|
317
|
+
* a re-derived stream under a surviving anchor drops the hide WITH A
|
|
318
|
+
* REPORT rather than deleting the wrong word. Restore DELETES the key
|
|
319
|
+
* (the restoreScene/captionsHidden rule — an entry with nothing to say is
|
|
320
|
+
* still an override), and `.default({})` keeps every pre-existing
|
|
321
|
+
* overrides.json parsing byte-identically. This field NEVER existed in
|
|
322
|
+
* the legacy positional-key era, so `migrateCaptionKeys` must NOT process
|
|
323
|
+
* it — there are no legacy hides to upgrade.
|
|
324
|
+
*/
|
|
325
|
+
captionWordsHidden: z.record(z.string(), z.object({ was: z.string() })).default({}),
|
|
326
|
+
/**
|
|
327
|
+
* Multi-word free-text rewrites — see `CaptionRangeEditSchema` for the
|
|
328
|
+
* whole contract (endpoint anchoring, the whole-run `was` guard, identity
|
|
329
|
+
* by pair, why it is never legacy-keyed). Applied between per-word retypes
|
|
330
|
+
* and hides (`applyCaptionLayers`).
|
|
331
|
+
*/
|
|
332
|
+
captionRangeEdits: z.array(CaptionRangeEditSchema).default([]),
|
|
333
|
+
/**
|
|
334
|
+
* Per-LINE caption TIMING nudges — "when does this caption appear, and when
|
|
335
|
+
* does it leave". Stored as DELTAS against the DERIVED window (`lead` moves
|
|
336
|
+
* the line's OPENING seam, `tail` its CLOSING seam), keyed by the LINE's
|
|
337
|
+
* FIRST WORD's SOURCE time (`captionKeyFor`, §137). Deltas over source keys
|
|
338
|
+
* make the record recut-immune for free: a recut rebuilds every derived
|
|
339
|
+
* `start`/`end` through the new TimeMap and the deltas simply re-apply on
|
|
340
|
+
* top — zero work in `remapOverridesThroughRecut`, the same property every
|
|
341
|
+
* other caption record leans on (captions.ts:14-20: `srcStart` is the one
|
|
342
|
+
* field a re-cut cannot move). Restore DELETES the key, and a patch whose
|
|
343
|
+
* deltas are both under 1ms in magnitude also deletes (the clearVideo/
|
|
344
|
+
* patchCaption clear-override rule — a nudge of nothing is still an
|
|
345
|
+
* override). `.default({})` keeps every pre-existing overrides.json parsing
|
|
346
|
+
* byte-identically, and the field NEVER existed in the legacy
|
|
347
|
+
* positional-key era, so `migrateCaptionKeys` must not process it.
|
|
348
|
+
*
|
|
349
|
+
* PER LINE, NOT PER WORD, and that is the whole point of the field. It
|
|
350
|
+
* replaces `captionWordTiming` (deleted 2026-08-18), which stored the same
|
|
351
|
+
* shape against individual WORDS and was measured to be MATHEMATICALLY
|
|
352
|
+
* INERT: on a live workdir (117 lines / 301 words) 116/116 inter-line gaps
|
|
353
|
+
* were exactly 0.0, 184/184 intra-line word boundaries exactly 0.0,
|
|
354
|
+
* `line.start === words[0].start` 117/117 and `line.end === lastWord.end`
|
|
355
|
+
* 117/117 — `transcribe.ts` chains words (`next.start = w.end`) and
|
|
356
|
+
* `captions.ts:203-213`'s hold pass clamps each line's end to the next
|
|
357
|
+
* line's start, so the caption stream is a GAP-FREE PARTITION. A per-word
|
|
358
|
+
* clamp of `[max(lineStart, prevEnd), min(lineEnd, nextStart)]` therefore
|
|
359
|
+
* collapsed to exactly `[w.start, w.end]` for EVERY word: the user dragged,
|
|
360
|
+
* every stored delta came back zero, and the reducer's sub-ms rule deleted
|
|
361
|
+
* them again. Do not reintroduce word-level clamping against a packed
|
|
362
|
+
* stream. Word stamps also only drive the karaoke highlight INSIDE a line's
|
|
363
|
+
* `<Sequence>` window (CaptionTrack.tsx:228-229, 387) — "when a caption
|
|
364
|
+
* appears" IS `line.start`/`line.end`, so timing has to move LINE windows.
|
|
365
|
+
* The ±30s range is per SEAM, which is why it is wider than the old
|
|
366
|
+
* per-word ±10s: a line may be dragged well clear of its neighbours, and
|
|
367
|
+
* `applyCaptionLineTiming`'s sweep — not the schema — is what keeps seams
|
|
368
|
+
* ordered and inside the track.
|
|
369
|
+
*/
|
|
370
|
+
captionLineTiming: z
|
|
371
|
+
.record(
|
|
372
|
+
z.string(),
|
|
373
|
+
z.object({
|
|
374
|
+
lead: z.number().min(-30).max(30),
|
|
375
|
+
tail: z.number().min(-30).max(30),
|
|
376
|
+
}),
|
|
377
|
+
)
|
|
378
|
+
.default({}),
|
|
259
379
|
/**
|
|
260
380
|
* Scene split points. `at` is ABSOLUTE output seconds (R16 §61 — Cmd/Ctrl+B
|
|
261
381
|
* at the playhead) and moves when a re-cut re-anchors the doc; `id` is
|
|
@@ -978,6 +1098,614 @@ export function applyCaptionEdits(
|
|
|
978
1098
|
return { lines: out, dropped };
|
|
979
1099
|
}
|
|
980
1100
|
|
|
1101
|
+
/**
|
|
1102
|
+
* Re-time replacement tokens over ONE line's stretch of a rewritten run —
|
|
1103
|
+
* `repair.ts`'s `retime` model (producer/repair.ts:137-154), restated here
|
|
1104
|
+
* for CaptionWords: stamps distributed across the window weighted by token
|
|
1105
|
+
* length + 1, strictly increasing, the last token's `end` pinned to the
|
|
1106
|
+
* window end so the run never leaks past the span it replaced. The measured
|
|
1107
|
+
* window edges (first run word's start, last run word's end) are kept;
|
|
1108
|
+
* only the interior boundaries are interpolated — interpolated boundaries
|
|
1109
|
+
* are a guess, and `retime`'s comment is explicit that a guess must never
|
|
1110
|
+
* displace a measurement, which is why the equal-count fast path in
|
|
1111
|
+
* `applyCaptionRangeEdits` below bypasses this entirely.
|
|
1112
|
+
*/
|
|
1113
|
+
function retimeCaptionTokens(
|
|
1114
|
+
tokens: readonly string[],
|
|
1115
|
+
windowStart: number,
|
|
1116
|
+
windowEnd: number,
|
|
1117
|
+
srcStarts: readonly number[],
|
|
1118
|
+
): CaptionWord[] {
|
|
1119
|
+
const weights = tokens.map((t) => t.length + 1);
|
|
1120
|
+
const total = weights.reduce((a, b) => a + b, 0);
|
|
1121
|
+
const out: CaptionWord[] = [];
|
|
1122
|
+
let cursor = windowStart;
|
|
1123
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
1124
|
+
const share = ((windowEnd - windowStart) * weights[i]!) / total;
|
|
1125
|
+
const end = i === tokens.length - 1 ? windowEnd : cursor + share;
|
|
1126
|
+
out.push({ text: tokens[i]!, start: cursor, end, srcStart: srcStarts[i]! });
|
|
1127
|
+
cursor = end;
|
|
1128
|
+
}
|
|
1129
|
+
return out;
|
|
1130
|
+
}
|
|
1131
|
+
|
|
1132
|
+
/**
|
|
1133
|
+
* Apply the free-text RANGE rewrites (`captionRangeEdits`) — the one layer
|
|
1134
|
+
* allowed to change word COUNT, which is why it exists at all: everything it
|
|
1135
|
+
* reshapes is the derived `CaptionLine[]`, never `transcript.words` (scene
|
|
1136
|
+
* anchors are raw indices into that array — splicing it is the forbidden
|
|
1137
|
+
* operation this whole edit family is built around).
|
|
1138
|
+
*
|
|
1139
|
+
* Same reporting shape as `applyCaptionEdits`; drop `key`s are the COMPOSITE
|
|
1140
|
+
* `${fromKey}..${toKey}` — the pair is the entry's identity, and either half
|
|
1141
|
+
* alone names only an endpoint. Each entry drops AT MOST ONCE (unlike the
|
|
1142
|
+
* per-word layers, where one key can be reported per extra claimant), which
|
|
1143
|
+
* is what lets `reconcileCaptionEdits` count applied entries by subtraction.
|
|
1144
|
+
*
|
|
1145
|
+
* Locating: `fromKey`'s first claimant across the flat word order (the
|
|
1146
|
+
* per-word first-claimant rule — ms-quantised keys CAN collide,
|
|
1147
|
+
* captions.ts:44-50), then a FORWARD walk to `toKey`; a missing endpoint, or
|
|
1148
|
+
* a `toKey` that only occurs before `fromKey`, is `found: null`. An entry
|
|
1149
|
+
* whose pair was already applied, or whose `fromKey` an earlier range edit's
|
|
1150
|
+
* run consumed, is `duplicate-anchor` — reachable only in a hand-edited doc,
|
|
1151
|
+
* since the reducer scrubs overlapping entries at creation, and reported
|
|
1152
|
+
* rather than guessed at like every other collision in this file.
|
|
1153
|
+
*
|
|
1154
|
+
* The whole-run stale guard: the run's live texts, NFC-normalized and
|
|
1155
|
+
* space-joined, must equal `was` byte for byte, or the WHOLE edit drops with
|
|
1156
|
+
* the joined text as `found` — never a partial rewrite of the words that
|
|
1157
|
+
* still match (a half-applied rewrite reads as garbage, and there is no
|
|
1158
|
+
* per-word truth to fall back on once the counts differ).
|
|
1159
|
+
*
|
|
1160
|
+
* Retiming across lines: the run may span several lines, and their `start`/
|
|
1161
|
+
* `end` WINDOWS are deliberately not re-packed — Sequence windows and
|
|
1162
|
+
* `buildCaptionLines`' breakpoint semantics stay exactly as produced.
|
|
1163
|
+
* Replacement tokens are distributed across the affected lines
|
|
1164
|
+
* proportionally to each line's share of the run's summed word duration,
|
|
1165
|
+
* rounded by largest remainder (deterministic — earlier line wins a tie) so
|
|
1166
|
+
* every token lands somewhere and the totals match. Within a line the stamps
|
|
1167
|
+
* follow `retimeCaptionTokens` above; a token count equal to the run's word
|
|
1168
|
+
* count skips all of it and keeps the measured per-word stamps AND srcStarts
|
|
1169
|
+
* verbatim (measured ASR boundaries beat interpolation — `retime`'s rule).
|
|
1170
|
+
* A line allotted zero tokens loses its run words, and if that empties it
|
|
1171
|
+
* the line is omitted (the `applyCaptionWordHides` rule — no zero-word
|
|
1172
|
+
* Sequence).
|
|
1173
|
+
*
|
|
1174
|
+
* srcStart minting for count-changed runs: linear across `[fromSrc, toSrc]`
|
|
1175
|
+
* (the endpoints' own source starts), endpoints re-minted verbatim — which
|
|
1176
|
+
* is what lets the user select a rewritten run again and edit it (its
|
|
1177
|
+
* endpoints still answer to the same pair). Strictly increasing whenever the
|
|
1178
|
+
* span is non-degenerate; when the span is too short for 1ms-distinct
|
|
1179
|
+
* quantised keys (`captionKeyFor` rounds to ms), later words SHARE quantised
|
|
1180
|
+
* keys — an accepted, documented duplicate-anchor case the existing
|
|
1181
|
+
* machinery reports if a per-word edit ever targets one.
|
|
1182
|
+
*/
|
|
1183
|
+
export function applyCaptionRangeEdits(
|
|
1184
|
+
lines: readonly CaptionLine[],
|
|
1185
|
+
rangeEdits: readonly CaptionRangeEdit[],
|
|
1186
|
+
): AppliedCaptionEdits {
|
|
1187
|
+
const dropped: AppliedCaptionEdits["dropped"] = [];
|
|
1188
|
+
if (rangeEdits.length === 0) return { lines: [...lines], dropped };
|
|
1189
|
+
|
|
1190
|
+
let out: CaptionLine[] = [...lines];
|
|
1191
|
+
const seenPairs = new Set<string>();
|
|
1192
|
+
const consumed = new Set<string>();
|
|
1193
|
+
|
|
1194
|
+
for (const entry of rangeEdits) {
|
|
1195
|
+
const key = `${entry.fromKey}..${entry.toKey}`;
|
|
1196
|
+
// Flatten the CURRENT lines — edits apply sequentially, so a later entry
|
|
1197
|
+
// addresses the stream as the earlier ones left it (that is how a
|
|
1198
|
+
// re-minted endpoint stays addressable at all).
|
|
1199
|
+
const flat: Array<{ line: number; word: number; w: CaptionWord }> = [];
|
|
1200
|
+
for (let li = 0; li < out.length; li++) {
|
|
1201
|
+
for (let wi = 0; wi < out[li]!.words.length; wi++) {
|
|
1202
|
+
flat.push({ line: li, word: wi, w: out[li]!.words[wi]! });
|
|
1203
|
+
}
|
|
1204
|
+
}
|
|
1205
|
+
const fromIdx = flat.findIndex((f) => captionAnchorOf(f.w) === entry.fromKey);
|
|
1206
|
+
if (seenPairs.has(key) || consumed.has(entry.fromKey)) {
|
|
1207
|
+
dropped.push({
|
|
1208
|
+
key,
|
|
1209
|
+
expected: entry.was,
|
|
1210
|
+
found: fromIdx === -1 ? null : flat[fromIdx]!.w.text,
|
|
1211
|
+
reason: "duplicate-anchor",
|
|
1212
|
+
});
|
|
1213
|
+
continue;
|
|
1214
|
+
}
|
|
1215
|
+
if (fromIdx === -1) {
|
|
1216
|
+
dropped.push({ key, expected: entry.was, found: null });
|
|
1217
|
+
continue;
|
|
1218
|
+
}
|
|
1219
|
+
// FORWARD only: a toKey sitting before fromKey is a run that crosses a
|
|
1220
|
+
// gap the stream no longer bridges — `found: null`, never a guess.
|
|
1221
|
+
let toIdx = -1;
|
|
1222
|
+
for (let i = fromIdx; i < flat.length; i++) {
|
|
1223
|
+
if (captionAnchorOf(flat[i]!.w) === entry.toKey) {
|
|
1224
|
+
toIdx = i;
|
|
1225
|
+
break;
|
|
1226
|
+
}
|
|
1227
|
+
}
|
|
1228
|
+
if (toIdx === -1) {
|
|
1229
|
+
dropped.push({ key, expected: entry.was, found: null });
|
|
1230
|
+
continue;
|
|
1231
|
+
}
|
|
1232
|
+
const run = flat.slice(fromIdx, toIdx + 1);
|
|
1233
|
+
const joined = run
|
|
1234
|
+
.map((f) => f.w.text)
|
|
1235
|
+
.join(" ")
|
|
1236
|
+
.normalize("NFC");
|
|
1237
|
+
if (joined !== entry.was.normalize("NFC")) {
|
|
1238
|
+
dropped.push({ key, expected: entry.was, found: joined });
|
|
1239
|
+
continue;
|
|
1240
|
+
}
|
|
1241
|
+
const tokens = entry.text.trim().split(/\s+/).filter(Boolean);
|
|
1242
|
+
if (tokens.length === 0) {
|
|
1243
|
+
// Defensive: zod's min(1) admits a whitespace-only string, and a run
|
|
1244
|
+
// rewritten to NOTHING is a delete, which is the hide layer's job —
|
|
1245
|
+
// treated as a stale-style drop rather than silently emptying the run.
|
|
1246
|
+
dropped.push({ key, expected: entry.was, found: joined });
|
|
1247
|
+
continue;
|
|
1248
|
+
}
|
|
1249
|
+
seenPairs.add(key);
|
|
1250
|
+
for (const f of run) {
|
|
1251
|
+
const a = captionAnchorOf(f.w);
|
|
1252
|
+
if (a !== null) consumed.add(a);
|
|
1253
|
+
}
|
|
1254
|
+
|
|
1255
|
+
if (tokens.length === run.length) {
|
|
1256
|
+
// Equal count: keep the measured stamps AND srcStarts verbatim —
|
|
1257
|
+
// `retime`'s fast path, for its reason (measured ASR onsets beat any
|
|
1258
|
+
// interpolation, and verbatim srcStarts keep every anchor addressable).
|
|
1259
|
+
const replaced = new Map(run.map((f, i) => [`${f.line}:${f.word}`, tokens[i]!]));
|
|
1260
|
+
out = out.map((line, li) => ({
|
|
1261
|
+
...line,
|
|
1262
|
+
words: line.words.map((w, wi) => {
|
|
1263
|
+
const text = replaced.get(`${li}:${wi}`);
|
|
1264
|
+
return text === undefined ? w : { ...w, text };
|
|
1265
|
+
}),
|
|
1266
|
+
}));
|
|
1267
|
+
continue;
|
|
1268
|
+
}
|
|
1269
|
+
|
|
1270
|
+
// Count changed: distribute tokens across the affected lines by each
|
|
1271
|
+
// line's share of the run's total duration, largest-remainder rounded.
|
|
1272
|
+
const lineOrder: number[] = [];
|
|
1273
|
+
const runByLine = new Map<number, { first: number; last: number; words: CaptionWord[] }>();
|
|
1274
|
+
for (const f of run) {
|
|
1275
|
+
const seg = runByLine.get(f.line);
|
|
1276
|
+
if (seg) {
|
|
1277
|
+
seg.last = f.word;
|
|
1278
|
+
seg.words.push(f.w);
|
|
1279
|
+
} else {
|
|
1280
|
+
lineOrder.push(f.line);
|
|
1281
|
+
runByLine.set(f.line, { first: f.word, last: f.word, words: [f.w] });
|
|
1282
|
+
}
|
|
1283
|
+
}
|
|
1284
|
+
const shares = lineOrder.map((li) =>
|
|
1285
|
+
runByLine.get(li)!.words.reduce((a, w) => a + (w.end - w.start), 0),
|
|
1286
|
+
);
|
|
1287
|
+
const totalShare = shares.reduce((a, b) => a + b, 0);
|
|
1288
|
+
// Zero total duration (every run word zero-width) has no proportion to
|
|
1289
|
+
// honor — fall back to equal weights so the rounding below still lands
|
|
1290
|
+
// every token somewhere deterministic.
|
|
1291
|
+
const weights = totalShare > 0 ? shares : shares.map(() => 1);
|
|
1292
|
+
const weightTotal = totalShare > 0 ? totalShare : shares.length;
|
|
1293
|
+
const quotas = weights.map((s) => (tokens.length * s) / weightTotal);
|
|
1294
|
+
const counts = quotas.map((q) => Math.floor(q));
|
|
1295
|
+
let leftover = tokens.length - counts.reduce((a, b) => a + b, 0);
|
|
1296
|
+
// Largest remainder first; ties break to the EARLIER line — stated so
|
|
1297
|
+
// the distribution is reproducible from the doc alone, like every other
|
|
1298
|
+
// persisted derivation in this file.
|
|
1299
|
+
const byRemainder = quotas
|
|
1300
|
+
.map((q, i) => ({ i, rem: q - Math.floor(q) }))
|
|
1301
|
+
.sort((a, b) => b.rem - a.rem || a.i - b.i);
|
|
1302
|
+
for (let k = 0; leftover > 0; k = (k + 1) % byRemainder.length) {
|
|
1303
|
+
counts[byRemainder[k]!.i]!++;
|
|
1304
|
+
leftover--;
|
|
1305
|
+
}
|
|
1306
|
+
|
|
1307
|
+
const fromSrc = run[0]!.w.srcStart;
|
|
1308
|
+
const toSrc = run[run.length - 1]!.w.srcStart;
|
|
1309
|
+
const srcStarts = tokens.map((_, j) =>
|
|
1310
|
+
tokens.length === 1 ? fromSrc : fromSrc + ((toSrc - fromSrc) * j) / (tokens.length - 1),
|
|
1311
|
+
);
|
|
1312
|
+
|
|
1313
|
+
let tokenCursor = 0;
|
|
1314
|
+
const next: CaptionLine[] = [];
|
|
1315
|
+
for (let li = 0; li < out.length; li++) {
|
|
1316
|
+
const line = out[li]!;
|
|
1317
|
+
const seg = runByLine.get(li);
|
|
1318
|
+
if (!seg) {
|
|
1319
|
+
next.push(line);
|
|
1320
|
+
continue;
|
|
1321
|
+
}
|
|
1322
|
+
const n = counts[lineOrder.indexOf(li)]!;
|
|
1323
|
+
const lineTokens = tokens.slice(tokenCursor, tokenCursor + n);
|
|
1324
|
+
const lineSrcs = srcStarts.slice(tokenCursor, tokenCursor + n);
|
|
1325
|
+
tokenCursor += n;
|
|
1326
|
+
const minted =
|
|
1327
|
+
n === 0
|
|
1328
|
+
? []
|
|
1329
|
+
: retimeCaptionTokens(
|
|
1330
|
+
lineTokens,
|
|
1331
|
+
seg.words[0]!.start,
|
|
1332
|
+
seg.words[seg.words.length - 1]!.end,
|
|
1333
|
+
lineSrcs,
|
|
1334
|
+
);
|
|
1335
|
+
const words = [...line.words.slice(0, seg.first), ...minted, ...line.words.slice(seg.last + 1)];
|
|
1336
|
+
// Window untouched (the no-re-pack rule above); an emptied line is
|
|
1337
|
+
// omitted, same as `applyCaptionWordHides`.
|
|
1338
|
+
if (words.length === 0) continue;
|
|
1339
|
+
next.push({ ...line, words });
|
|
1340
|
+
}
|
|
1341
|
+
out = next;
|
|
1342
|
+
}
|
|
1343
|
+
return { lines: out, dropped };
|
|
1344
|
+
}
|
|
1345
|
+
|
|
1346
|
+
/**
|
|
1347
|
+
* Drop hidden caption words (the `captionWordsHidden` layer). Same reporting
|
|
1348
|
+
* shape as `applyCaptionEdits` — callers must surface `dropped` for the same
|
|
1349
|
+
* reason: a hide that silently fails looks like the editor forgot it.
|
|
1350
|
+
*
|
|
1351
|
+
* Runs on the DERIVED `CaptionLine[]`, never on `transcript.words` — scene
|
|
1352
|
+
* anchors are raw word INDICES into the transcript, so splicing a word out of
|
|
1353
|
+
* it would shift every later anchor onto the wrong word: the forbidden
|
|
1354
|
+
* operation this whole layer exists to avoid. The transcript stays intact;
|
|
1355
|
+
* only the rendered caption stream loses the word.
|
|
1356
|
+
*
|
|
1357
|
+
* Line WINDOWS are recomputed here, deliberately: `buildCaptionLines` derives
|
|
1358
|
+
* `start` from the first word and `end` from the last word plus a hold
|
|
1359
|
+
* (captions.ts:203-213), so hiding a boundary word would otherwise leave the
|
|
1360
|
+
* line lingering on screen over silence — up for the hidden first word's
|
|
1361
|
+
* duration, or held past the hidden last word's end. A hidden FIRST word moves
|
|
1362
|
+
* `start` to the first survivor; a hidden LAST word re-bases the packer's hold
|
|
1363
|
+
* delta onto whichever word is now last (clamped so the line never ends before
|
|
1364
|
+
* its own last word); middle hides leave the window alone. A line whose words
|
|
1365
|
+
* are ALL hidden is omitted entirely, so the downstream CaptionTrack emits no
|
|
1366
|
+
* Sequence for it.
|
|
1367
|
+
*
|
|
1368
|
+
* `was` is the LIVE (post-retype) text at hide time — hides apply AFTER
|
|
1369
|
+
* retypes (`applyCaptionLayers` below) — so un-retyping a word under a hide
|
|
1370
|
+
* stales the hide, and it is REPORTED rather than guessed at. Same
|
|
1371
|
+
* first-claimant rule as `applyCaptionEdits`: ms-quantised keys CAN collide
|
|
1372
|
+
* (`captions.ts:44-50` manufactures duplicates by design), and one hide must
|
|
1373
|
+
* remove one word, not every word sharing its instant.
|
|
1374
|
+
*/
|
|
1375
|
+
export function applyCaptionWordHides(
|
|
1376
|
+
lines: readonly CaptionLine[],
|
|
1377
|
+
hides: Record<string, { was: string }>,
|
|
1378
|
+
): AppliedCaptionEdits {
|
|
1379
|
+
const dropped: AppliedCaptionEdits["dropped"] = [];
|
|
1380
|
+
if (Object.keys(hides).length === 0) return { lines: [...lines], dropped };
|
|
1381
|
+
|
|
1382
|
+
const seen = new Set<string>();
|
|
1383
|
+
const out: CaptionLine[] = [];
|
|
1384
|
+
for (const line of lines) {
|
|
1385
|
+
const kept: CaptionWord[] = [];
|
|
1386
|
+
for (const w of line.words) {
|
|
1387
|
+
// No anchor, no hide — same boundary rule as `applyCaptionEdits`: a
|
|
1388
|
+
// pre-§137 word cannot be addressed, and the stored hides then fall out
|
|
1389
|
+
// of the sweep below as `found: null`.
|
|
1390
|
+
const key = captionAnchorOf(w);
|
|
1391
|
+
const hide = key === null ? undefined : hides[key];
|
|
1392
|
+
if (key === null || !hide) {
|
|
1393
|
+
kept.push(w);
|
|
1394
|
+
continue;
|
|
1395
|
+
}
|
|
1396
|
+
// An earlier word already answered for this anchor — whichever way it
|
|
1397
|
+
// answered. Hiding here too would fan one delete onto a second word.
|
|
1398
|
+
if (seen.has(key)) {
|
|
1399
|
+
dropped.push({ key, expected: hide.was, found: w.text, reason: "duplicate-anchor" });
|
|
1400
|
+
kept.push(w);
|
|
1401
|
+
continue;
|
|
1402
|
+
}
|
|
1403
|
+
seen.add(key);
|
|
1404
|
+
if (w.text !== hide.was) {
|
|
1405
|
+
dropped.push({ key, expected: hide.was, found: w.text });
|
|
1406
|
+
kept.push(w);
|
|
1407
|
+
continue;
|
|
1408
|
+
}
|
|
1409
|
+
// Matched: the word is dropped from the line.
|
|
1410
|
+
}
|
|
1411
|
+
if (kept.length === line.words.length) {
|
|
1412
|
+
out.push(line);
|
|
1413
|
+
continue;
|
|
1414
|
+
}
|
|
1415
|
+
// Every word hidden — the line goes with them, rather than a zero-word
|
|
1416
|
+
// line the CaptionTrack would still mount a Sequence for.
|
|
1417
|
+
if (kept.length === 0) continue;
|
|
1418
|
+
const lastOriginal = line.words[line.words.length - 1]!;
|
|
1419
|
+
const firstKept = kept[0]!;
|
|
1420
|
+
const lastKept = kept[kept.length - 1]!;
|
|
1421
|
+
const start = firstKept === line.words[0] ? line.start : firstKept.start;
|
|
1422
|
+
// The packer's hold delta (captions.ts:203-213) rides on whichever word
|
|
1423
|
+
// is now last; clamped so the line never ends before its own last word
|
|
1424
|
+
// (the delta can be negative when the hold was clamped to outputDuration).
|
|
1425
|
+
const end =
|
|
1426
|
+
lastKept === lastOriginal
|
|
1427
|
+
? line.end
|
|
1428
|
+
: Math.max(lastKept.end, lastKept.end + (line.end - lastOriginal.end));
|
|
1429
|
+
out.push({ words: kept, start, end });
|
|
1430
|
+
}
|
|
1431
|
+
|
|
1432
|
+
// An anchor no word carries any more — a later cut removed the word the
|
|
1433
|
+
// user hid. Silence here is the field-case failure mode, so say it.
|
|
1434
|
+
for (const [key, hide] of Object.entries(hides)) {
|
|
1435
|
+
if (!seen.has(key)) dropped.push({ key, expected: hide.was, found: null });
|
|
1436
|
+
}
|
|
1437
|
+
return { lines: out, dropped };
|
|
1438
|
+
}
|
|
1439
|
+
|
|
1440
|
+
/**
|
|
1441
|
+
* The floor a caption's window may shrink to. A caption nobody can read is a
|
|
1442
|
+
* delete wearing a timing nudge's clothes — deletes are the hide layer's
|
|
1443
|
+
* gesture, with its own guard and report. Also the minimum WIDTH of every
|
|
1444
|
+
* line's window, which is what keeps §115 (`packages/scenes/src/frames.ts`)
|
|
1445
|
+
* true: 50ms is more than one frame at any fps this renders at, so two
|
|
1446
|
+
* adjacent windows can never round onto the same frame.
|
|
1447
|
+
*
|
|
1448
|
+
* (Was `MIN_TIMED_WORD_SEC`, the same 0.05 measured against a WORD, until the
|
|
1449
|
+
* per-word layer was found inert — see `captionLineTiming`'s docstring.)
|
|
1450
|
+
*
|
|
1451
|
+
* Exported for the EDITOR's drag bounds (`captionDragBounds`,
|
|
1452
|
+
* apps/editor/src/TranscriptPanel.tsx): the popover has to stop a drag exactly
|
|
1453
|
+
* where this sweep would, and a second copy of the floor in the browser is how
|
|
1454
|
+
* the two would drift apart.
|
|
1455
|
+
*/
|
|
1456
|
+
export const MIN_CAPTION_SEC = 0.05;
|
|
1457
|
+
|
|
1458
|
+
/**
|
|
1459
|
+
* Re-time a line's words from one window onto another, PROPORTIONALLY — the
|
|
1460
|
+
* arithmetic that keeps the karaoke highlight in sync when a line's
|
|
1461
|
+
* `<Sequence>` window moves under it (`CaptionTrack.tsx:228-229, 387` reads
|
|
1462
|
+
* the word stamps INSIDE the window; a window moved without them would light
|
|
1463
|
+
* the wrong words up, or none).
|
|
1464
|
+
*
|
|
1465
|
+
* The source is the WINDOW, not the words' own span: on the packed stream
|
|
1466
|
+
* both are the same interval (`line.start === words[0].start` and
|
|
1467
|
+
* `line.end === lastWord.end`, measured 117/117 — `captionLineTiming`'s
|
|
1468
|
+
* docstring), and on a line that DOES carry lead-in or hold (the hide layer
|
|
1469
|
+
* can re-base either edge) mapping the window preserves that slack instead of
|
|
1470
|
+
* stretching the words over it.
|
|
1471
|
+
*
|
|
1472
|
+
* Pure and exported so a caller previewing a drag and the apply pass below
|
|
1473
|
+
* share ONE piece of arithmetic (the openCommand/openInBrowser split).
|
|
1474
|
+
* Identity when the source window is degenerate — a zero-width or inverted
|
|
1475
|
+
* span has no ratio to scale by, and `0/0` would put NaN stamps in the render
|
|
1476
|
+
* props. The caller owns `toStart < toEnd`; a target handed backwards would
|
|
1477
|
+
* mirror the word order, which `applyCaptionLineTiming`'s edge sweep makes
|
|
1478
|
+
* unreachable.
|
|
1479
|
+
*/
|
|
1480
|
+
export function scaleWordsIntoWindow(
|
|
1481
|
+
words: readonly CaptionWord[],
|
|
1482
|
+
fromStart: number,
|
|
1483
|
+
fromEnd: number,
|
|
1484
|
+
toStart: number,
|
|
1485
|
+
toEnd: number,
|
|
1486
|
+
): CaptionWord[] {
|
|
1487
|
+
const span = fromEnd - fromStart;
|
|
1488
|
+
if (!(span > 0)) return words.map((w) => ({ ...w }));
|
|
1489
|
+
const ratio = (toEnd - toStart) / span;
|
|
1490
|
+
const at = (t: number): number => toStart + (t - fromStart) * ratio;
|
|
1491
|
+
return words.map((w) => ({ ...w, start: at(w.start), end: at(w.end) }));
|
|
1492
|
+
}
|
|
1493
|
+
|
|
1494
|
+
/**
|
|
1495
|
+
* Apply per-LINE caption TIMING nudges (`captionLineTiming`) — the LAST
|
|
1496
|
+
* layer, after hides, because it must operate on the SURVIVING lines: a hide
|
|
1497
|
+
* can move a line's window (or remove the line entirely), and a nudge stored
|
|
1498
|
+
* on a line the hides emptied has no window to move (it falls out of the
|
|
1499
|
+
* sweep as `found: null`, like every other orphaned caption record).
|
|
1500
|
+
*
|
|
1501
|
+
* EDGES, NOT ONE SHARED SEAM. Each line owns its `[start, end]` pair, and a
|
|
1502
|
+
* line's END and the next line's START are two separate numbers here — even
|
|
1503
|
+
* though on a real transcript they are always equal, because the packer chains
|
|
1504
|
+
* words (`transcribe.ts`: `next.start = w.end`) and clamps each line's end to
|
|
1505
|
+
* the next line's start (`captions.ts:203-213`), giving inter-line gaps of
|
|
1506
|
+
* exactly zero (measured 116/116, see `captionLineTiming`). A nudge CLOSES the
|
|
1507
|
+
* two onto one value only when they were already COINCIDENT: that is what
|
|
1508
|
+
* makes a lead on the packed stream move both sides of the boundary, one edit
|
|
1509
|
+
* and two windows, exactly as before.
|
|
1510
|
+
*
|
|
1511
|
+
* They are two numbers because GAPS ARE REAL: `applyCaptionWordHides` re-bases
|
|
1512
|
+
* a line's window onto its surviving words, `MAX_CAPTION_WORD_LEAD_SEC`
|
|
1513
|
+
* (captions.ts:147, 169) clamps a word's display start, and an overrides.json
|
|
1514
|
+
* can be hand-edited. This code
|
|
1515
|
+
* used to hold ONE `seams` array whose interior entry was read off the later
|
|
1516
|
+
* line's start, conflating the two: with lines `[0,2] [2,4] [5,6]`, a
|
|
1517
|
+
* lead-only drag of the middle line (`{lead: -0.05, tail: 0}`, exactly what
|
|
1518
|
+
* the editor writes) rebuilt the UNTOUCHED third caption as `[4,6]` — a full
|
|
1519
|
+
* second early, its words stretched 2x by `scaleWordsIntoWindow`, with no drop
|
|
1520
|
+
* reported (review 2026-08-19).
|
|
1521
|
+
*
|
|
1522
|
+
* The edge model still protects §115 (`packages/scenes/src/frames.ts:1-21` —
|
|
1523
|
+
* no two lines may share a frame) BY CONSTRUCTION, which is what the old "LINE
|
|
1524
|
+
* WINDOWS NEVER CHANGE" rule existed for: the sweep below leaves the edges
|
|
1525
|
+
* ORDERED (`start_0 <= end_0 <= start_1 <= ... <= end_n-1`) with every window
|
|
1526
|
+
* at least `MIN_CAPTION_SEC` wide, and ordered non-overlapping windows at
|
|
1527
|
+
* least 50ms wide cannot round onto a shared frame.
|
|
1528
|
+
*
|
|
1529
|
+
* THE SWEEP, forward: every edge is clamped into the track's ORIGINAL outer
|
|
1530
|
+
* bounds, no line may open before the previous line CLOSED, and no window may
|
|
1531
|
+
* be narrower than `MIN_CAPTION_SEC`. A backward pass then pulls lines left if
|
|
1532
|
+
* a track too short to hold every line at the floor made the forward pass run
|
|
1533
|
+
* into the end. Ordering is enforced against the NEIGHBOUR'S OWN edge, never a
|
|
1534
|
+
* derived seam: a nudge that runs past it is BLOCKED there rather than pushing
|
|
1535
|
+
* it, so a gap gets consumed but no untouched caption ever moves. (A nudge
|
|
1536
|
+
* takes time FROM a neighbour only through the coincidence rule above — the
|
|
1537
|
+
* packed case, where the two share the boundary being dragged.) The outer
|
|
1538
|
+
* bounds never GROW: a caption must not appear before the first caption of the
|
|
1539
|
+
* track or linger past the last, where there is no output left to show it
|
|
1540
|
+
* over.
|
|
1541
|
+
*
|
|
1542
|
+
* BOTH SIDES OF ONE BOUNDARY: line i's `tail` and line i+1's `lead` address
|
|
1543
|
+
* the same coincident boundary. The LATER line's `lead` wins,
|
|
1544
|
+
* deterministically — the UI writes both sides of a drag consistently, so this
|
|
1545
|
+
* only decides hand-edited docs, and a stated winner beats an
|
|
1546
|
+
* order-of-iteration accident. (A stored `lead: 0` still claims its edge; an
|
|
1547
|
+
* entry the user cleared is DELETED from the doc, not written as zeros.)
|
|
1548
|
+
*
|
|
1549
|
+
* Lines whose window the sweep did not move are returned VERBATIM — including
|
|
1550
|
+
* their word stamps — so a nudge on one caption cannot perturb the rest of
|
|
1551
|
+
* the track. The ones that did move (the nudged line AND, on a coincident
|
|
1552
|
+
* boundary, its neighbour) have their words scaled into the new window by
|
|
1553
|
+
* `scaleWordsIntoWindow`.
|
|
1554
|
+
*
|
|
1555
|
+
* DELIBERATELY NO `was` GUARD, unlike `captionWordsHidden`: timing is
|
|
1556
|
+
* text-orthogonal — a retype under a timing nudge changes what the caption
|
|
1557
|
+
* says, not when it is said, and staleness on text would drop nudges the user
|
|
1558
|
+
* never un-meant. `expected` in the drop reports is therefore always `""`
|
|
1559
|
+
* (the record stores no text to expect). Same first-claimant rule as every
|
|
1560
|
+
* per-word layer: ms-quantised anchors CAN collide (captions.ts:44-50), and
|
|
1561
|
+
* one nudge must move one line.
|
|
1562
|
+
*/
|
|
1563
|
+
export function applyCaptionLineTiming(
|
|
1564
|
+
lines: readonly CaptionLine[],
|
|
1565
|
+
timing: Record<string, { lead: number; tail: number }>,
|
|
1566
|
+
): AppliedCaptionEdits {
|
|
1567
|
+
const dropped: AppliedCaptionEdits["dropped"] = [];
|
|
1568
|
+
const n = lines.length;
|
|
1569
|
+
// NO LINES is not "no nudges to report": every stored key is an anchor that
|
|
1570
|
+
// no line starts on, which is exactly the `found: null` case the sweep at
|
|
1571
|
+
// the bottom exists to say out loud, and what this function's own docstring
|
|
1572
|
+
// promises. `applyCaptionEdits` and `applyCaptionWordHides` never took this
|
|
1573
|
+
// shortcut either. The editor's false-banner guard lives at the CALLER
|
|
1574
|
+
// (`App.tsx`: `if (!renderProps) return { lines: [], dropped: [] }`), where
|
|
1575
|
+
// "nothing loaded yet" is distinguishable from "this cut has no captions" —
|
|
1576
|
+
// silence here instead let produce report nudges as applied that never were.
|
|
1577
|
+
if (n === 0) {
|
|
1578
|
+
for (const key of Object.keys(timing)) dropped.push({ key, expected: "", found: null });
|
|
1579
|
+
return { lines: [], dropped };
|
|
1580
|
+
}
|
|
1581
|
+
if (Object.keys(timing).length === 0) return { lines: [...lines], dropped };
|
|
1582
|
+
|
|
1583
|
+
// One `[start, end]` pair PER LINE — never a shared seam array (see the
|
|
1584
|
+
// docstring: the conflation moved untouched captions on a gapped stream).
|
|
1585
|
+
const starts = lines.map((l) => l.start);
|
|
1586
|
+
const ends = lines.map((l) => l.end);
|
|
1587
|
+
|
|
1588
|
+
const seen = new Set<string>();
|
|
1589
|
+
for (let i = 0; i < n; i++) {
|
|
1590
|
+
const line = lines[i]!;
|
|
1591
|
+
// No anchor, no nudge — the same boundary rule as `applyCaptionEdits`: a
|
|
1592
|
+
// pre-§137 word cannot be addressed, and the stored nudges then fall out
|
|
1593
|
+
// of the sweep below as `found: null`.
|
|
1594
|
+
const key = captionAnchorOf(line.words[0]);
|
|
1595
|
+
const entry = key === null ? undefined : timing[key];
|
|
1596
|
+
if (key === null || !entry) continue;
|
|
1597
|
+
// An earlier line already answered for this anchor — nudging here too
|
|
1598
|
+
// would fan one nudge onto a second line.
|
|
1599
|
+
if (seen.has(key)) {
|
|
1600
|
+
dropped.push({ key, expected: "", found: line.words[0]!.text, reason: "duplicate-anchor" });
|
|
1601
|
+
continue;
|
|
1602
|
+
}
|
|
1603
|
+
seen.add(key);
|
|
1604
|
+
// Deltas ride on the line's OWN edges, so a gapped stream moves the edge
|
|
1605
|
+
// the user dragged rather than the neighbour's. Tail first, then lead:
|
|
1606
|
+
// lines are visited in order, so line i+1's lead lands on a shared
|
|
1607
|
+
// boundary AFTER line i's tail — the documented "later lead wins".
|
|
1608
|
+
ends[i] = line.end + entry.tail;
|
|
1609
|
+
// COINCIDENCE, tested against the ORIGINAL edges: only a boundary the two
|
|
1610
|
+
// lines already SHARED travels with the nudge (the packed stream, where
|
|
1611
|
+
// every one of them is shared). Across a gap the neighbour stays where it
|
|
1612
|
+
// is — the sweep below still stops the moved edge from crossing it.
|
|
1613
|
+
// Assigning the same number, not recomputing it, keeps the two exactly
|
|
1614
|
+
// equal: a float `+ delta` computed twice can differ in the last bit, and
|
|
1615
|
+
// an unequal pair is an overlap the sweep would then have to fix.
|
|
1616
|
+
if (i + 1 < n && lines[i + 1]!.start === line.end) starts[i + 1] = ends[i]!;
|
|
1617
|
+
starts[i] = line.start + entry.lead;
|
|
1618
|
+
if (i > 0 && lines[i - 1]!.end === line.start) ends[i - 1] = starts[i]!;
|
|
1619
|
+
}
|
|
1620
|
+
|
|
1621
|
+
const lo = lines[0]!.start;
|
|
1622
|
+
const hi = lines[n - 1]!.end;
|
|
1623
|
+
// Forward: into the track's bounds, never opening before the previous line
|
|
1624
|
+
// CLOSED (its own edge, not a derived seam), never narrower than the floor.
|
|
1625
|
+
for (let i = 0; i < n; i++) {
|
|
1626
|
+
const floor = i === 0 ? lo : Math.max(lo, ends[i - 1]!);
|
|
1627
|
+
starts[i] = Math.min(Math.max(starts[i]!, floor), hi);
|
|
1628
|
+
ends[i] = Math.min(Math.max(ends[i]!, starts[i]! + MIN_CAPTION_SEC), hi);
|
|
1629
|
+
}
|
|
1630
|
+
// The forward pass caps at `hi`, so a track with less room than
|
|
1631
|
+
// `n * MIN_CAPTION_SEC` can leave the last lines piled on the end. Pull them
|
|
1632
|
+
// back (never before `lo`) so the edges stay ordered.
|
|
1633
|
+
for (let i = n - 1; i >= 0; i--) {
|
|
1634
|
+
const ceil = i === n - 1 ? hi : Math.min(hi, starts[i + 1]!);
|
|
1635
|
+
ends[i] = Math.max(Math.min(ends[i]!, ceil), lo);
|
|
1636
|
+
starts[i] = Math.max(Math.min(starts[i]!, ends[i]! - MIN_CAPTION_SEC), lo);
|
|
1637
|
+
}
|
|
1638
|
+
|
|
1639
|
+
const out = lines.map((line, i) => {
|
|
1640
|
+
const start = starts[i]!;
|
|
1641
|
+
const end = ends[i]!;
|
|
1642
|
+
// Neither edge moved: VERBATIM, same reference and same word stamps.
|
|
1643
|
+
if (start === line.start && end === line.end) return line;
|
|
1644
|
+
return {
|
|
1645
|
+
...line,
|
|
1646
|
+
start,
|
|
1647
|
+
end,
|
|
1648
|
+
words: scaleWordsIntoWindow(line.words, line.start, line.end, start, end),
|
|
1649
|
+
};
|
|
1650
|
+
});
|
|
1651
|
+
|
|
1652
|
+
// An anchor no line starts on any more — a later cut removed the word the
|
|
1653
|
+
// line was keyed to, or a hide emptied the line. Silence here is the
|
|
1654
|
+
// field-case failure mode, so say it.
|
|
1655
|
+
for (const key of Object.keys(timing)) {
|
|
1656
|
+
if (!seen.has(key)) dropped.push({ key, expected: "", found: null });
|
|
1657
|
+
}
|
|
1658
|
+
return { lines: out, dropped };
|
|
1659
|
+
}
|
|
1660
|
+
|
|
1661
|
+
export interface AppliedCaptionLayers {
|
|
1662
|
+
lines: CaptionLine[];
|
|
1663
|
+
/** Every layer's drop reports, tagged with which layer refused them. */
|
|
1664
|
+
dropped: Array<
|
|
1665
|
+
AppliedCaptionEdits["dropped"][number] & { layer: "edit" | "range" | "hide" | "timing" }
|
|
1666
|
+
>;
|
|
1667
|
+
}
|
|
1668
|
+
|
|
1669
|
+
/**
|
|
1670
|
+
* The caption edit layers, composed in their ONE authoritative order — the
|
|
1671
|
+
* single chokepoint both the editor preview and produce consume, so the two
|
|
1672
|
+
* can never disagree about caption content.
|
|
1673
|
+
*
|
|
1674
|
+
* Per-word edits → RANGE edits → hides → LINE TIMING. Edits BEFORE hides is the
|
|
1675
|
+
* `was` contract: a hide's `was` records the LIVE text the user saw when they
|
|
1676
|
+
* deleted the word, which is the post-retype text — running hides first
|
|
1677
|
+
* would stale every hide sitting on a retyped word. Range edits sit between
|
|
1678
|
+
* the two, but the order barely earns the word: the reducer's creation-time
|
|
1679
|
+
* scrubbing (`useEdits`' `patchCaptionRange`) removes every per-word edit
|
|
1680
|
+
* and hide inside a new range's interval, so a LIVE range edit never
|
|
1681
|
+
* coexists with either inside its own words — the order only matters for
|
|
1682
|
+
* hand-edited docs, where the layers' own guards report rather than guess.
|
|
1683
|
+
* Timing runs LAST because it must see the surviving LINES: the hide layer
|
|
1684
|
+
* re-bases a line's window onto its surviving words and drops a line whose
|
|
1685
|
+
* words are all hidden, and a nudge on a line that no longer exists has no
|
|
1686
|
+
* window to move (`applyCaptionLineTiming`). Drop reports carry which layer
|
|
1687
|
+
* refused them, since "the retype missed", "the rewrite missed" and "the
|
|
1688
|
+
* delete missed" send the user to different gestures.
|
|
1689
|
+
*/
|
|
1690
|
+
export function applyCaptionLayers(
|
|
1691
|
+
lines: readonly CaptionLine[],
|
|
1692
|
+
doc: OverrideDoc,
|
|
1693
|
+
): AppliedCaptionLayers {
|
|
1694
|
+
const edited = applyCaptionEdits(lines, doc.captions);
|
|
1695
|
+
const ranged = applyCaptionRangeEdits(edited.lines, doc.captionRangeEdits);
|
|
1696
|
+
const hidden = applyCaptionWordHides(ranged.lines, doc.captionWordsHidden);
|
|
1697
|
+
const timed = applyCaptionLineTiming(hidden.lines, doc.captionLineTiming);
|
|
1698
|
+
return {
|
|
1699
|
+
lines: timed.lines,
|
|
1700
|
+
dropped: [
|
|
1701
|
+
...edited.dropped.map((d) => ({ ...d, layer: "edit" as const })),
|
|
1702
|
+
...ranged.dropped.map((d) => ({ ...d, layer: "range" as const })),
|
|
1703
|
+
...hidden.dropped.map((d) => ({ ...d, layer: "hide" as const })),
|
|
1704
|
+
...timed.dropped.map((d) => ({ ...d, layer: "timing" as const })),
|
|
1705
|
+
],
|
|
1706
|
+
};
|
|
1707
|
+
}
|
|
1708
|
+
|
|
981
1709
|
/** Theme tokens the user set, over whatever the production already had. */
|
|
982
1710
|
export function resolveTheme(base: Theme, doc: OverrideDoc): Theme {
|
|
983
1711
|
return ThemeSchema.parse({ ...base, ...doc.theme });
|
package/src/phonetics.ts
CHANGED
|
@@ -12,6 +12,49 @@
|
|
|
12
12
|
* "unrelated", and it must stay dependency-free and deterministic.
|
|
13
13
|
*/
|
|
14
14
|
|
|
15
|
+
/**
|
|
16
|
+
* Marks Arabic-script text carries that two transcribers disagree about
|
|
17
|
+
* without disagreeing about the WORD: harakat/vowel diacritics
|
|
18
|
+
* (U+064B–U+065F, U+0670), tatweel (U+0640, a pure typographic stretch), and
|
|
19
|
+
* the zero-width joiners (U+200C/U+200D). Whisper emits them inconsistently
|
|
20
|
+
* and an LLM writing a correction rarely reproduces them, so leaving them in
|
|
21
|
+
* makes a correct repair either miss `locate()` outright or read as
|
|
22
|
+
* "different from what was heard" purely on invisible marks. Only tatweel
|
|
23
|
+
* survives the `\p{L}\p{N}` filter below (it is Lm, a letter); the rest are
|
|
24
|
+
* stripped here so the intent is legible rather than an accident of Unicode
|
|
25
|
+
* categories.
|
|
26
|
+
*/
|
|
27
|
+
// Escaped, not literal: three of these code points are invisible in an editor.
|
|
28
|
+
const ARABIC_NOISE = /[\u064B-\u065F\u0670\u0640\u200C\u200D]/g;
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Comparable form for two pieces of text: case-, punctuation- and
|
|
32
|
+
* whitespace-insensitive, in ANY script.
|
|
33
|
+
*
|
|
34
|
+
* Shared by `phonetics.ts` and `producer/repair.ts` on purpose — they used to
|
|
35
|
+
* hold two copies and the copy in `repair.ts` was `[^a-z0-9\s]`, i.e.
|
|
36
|
+
* Latin-only. Field case (2026-08-18): every one of 11 recorded Urdu repairs
|
|
37
|
+
* normalized to the empty string, so `norm(heard) === norm(correction)` was
|
|
38
|
+
* `"" === ""` and ALL 11 were refused as "identical to what was heard" —
|
|
39
|
+
* including `پرسٹ` → `فرسٹ`, which shares no letters with what it replaced.
|
|
40
|
+
*
|
|
41
|
+
* Keeping letters and digits of every script also KEEPS accented Latin
|
|
42
|
+
* ("café", "über") where the old expression deleted it. That is the same bug
|
|
43
|
+
* in miniature — a French word normalized to "caf" — so it is a fix, not a
|
|
44
|
+
* regression. Pure-ASCII input is byte-identical to the old behaviour, which
|
|
45
|
+
* is pinned by a test.
|
|
46
|
+
*/
|
|
47
|
+
export function normalizeForCompare(s: string): string {
|
|
48
|
+
return s
|
|
49
|
+
.normalize("NFC")
|
|
50
|
+
.toLowerCase()
|
|
51
|
+
.replace(ARABIC_NOISE, "")
|
|
52
|
+
.replace(/[^\p{L}\p{N}\s]/gu, "")
|
|
53
|
+
.split(/\s+/)
|
|
54
|
+
.filter(Boolean)
|
|
55
|
+
.join(" ");
|
|
56
|
+
}
|
|
57
|
+
|
|
15
58
|
/**
|
|
16
59
|
* Digraphs collapsed before single letters, longest first. The ch/sh and th
|
|
17
60
|
* sounds get DIGIT placeholders on purpose: a letter placeholder would be
|
|
@@ -106,6 +149,47 @@ export function soundsLike(a: string, b: string): number {
|
|
|
106
149
|
return Math.max(0, 1 - dist / Math.max(ka.length, kb.length));
|
|
107
150
|
}
|
|
108
151
|
|
|
152
|
+
/**
|
|
153
|
+
* 0..1 similarity of the TEXT itself, for scripts the phonetic key cannot
|
|
154
|
+
* represent (`phoneticKey` is defined over a-z, so anything non-Latin keys to
|
|
155
|
+
* ""). Same shape as `soundsLike` — normalized edit distance over the longer
|
|
156
|
+
* string — but run on the normalized text rather than a consonant skeleton.
|
|
157
|
+
*
|
|
158
|
+
* This is a weaker signal than a phonetic key and it is meant to be: an
|
|
159
|
+
* Urdu-script mishearing differs from the truth by a letter or two of the same
|
|
160
|
+
* script, so edit distance still separates it from an unrelated phrase. What
|
|
161
|
+
* it cannot do is fold vowels, which is why the floor below is calibrated
|
|
162
|
+
* against real data instead of borrowing SOUNDS_LIKE_FLOOR.
|
|
163
|
+
*/
|
|
164
|
+
export function textSimilarity(a: string, b: string): number {
|
|
165
|
+
const na = normalizeForCompare(a);
|
|
166
|
+
const nb = normalizeForCompare(b);
|
|
167
|
+
if (na.length === 0 && nb.length === 0) return 1;
|
|
168
|
+
if (na.length === 0 || nb.length === 0) return 0;
|
|
169
|
+
return Math.max(0, 1 - levenshtein(na, nb) / Math.max(na.length, nb.length));
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Floor for the non-Latin fallback, MEASURED rather than guessed.
|
|
174
|
+
*
|
|
175
|
+
* The 11 Urdu repairs recorded in a real production.json (2026-08-18) score,
|
|
176
|
+
* sorted: 0.333, 0.400, 0.500, 0.500, 0.545, 0.583, 0.636, 0.750, 0.750,
|
|
177
|
+
* 0.800, 0.800. The 0.333 is `حقیقہ ٹون` → `ہیکاتھون` ("hackathon"), a genuine
|
|
178
|
+
* repair and the worst of the set because the recognizer both re-segmented the
|
|
179
|
+
* word and changed its opening letter. Admitting it sets the ceiling on the
|
|
180
|
+
* floor; 0.33 is the largest value that does.
|
|
181
|
+
*
|
|
182
|
+
* Against that, unrelated four-word spans lifted from the same transcript
|
|
183
|
+
* score 0.167–0.250 and are refused. The band is narrow, and it is narrow for
|
|
184
|
+
* the same reason the Latin one is (see SOUNDS_LIKE_FLOOR): two SHORT
|
|
185
|
+
* unrelated Urdu spans can still land above it — measured, `پرسٹ ہیک` vs
|
|
186
|
+
* `ٹرس می` scores 0.500. There is no onset test here to catch that, because
|
|
187
|
+
* two of the 11 genuine repairs change their first letter. So this gate is
|
|
188
|
+
* real but shallow; the span, token-count and length guards in
|
|
189
|
+
* `applyRepairs` are what keep it from being a rewrite licence.
|
|
190
|
+
*/
|
|
191
|
+
export const TEXT_SIMILARITY_FLOOR = 0.33;
|
|
192
|
+
|
|
109
193
|
/**
|
|
110
194
|
* Default floor for "this is a repair, not a rewrite". Deliberately low,
|
|
111
195
|
* because a real mishearing can move word boundaries ("code churn" → "coach
|
|
@@ -124,11 +208,27 @@ export const SOUNDS_LIKE_FLOOR = 0.34;
|
|
|
124
208
|
* a phrase, essentially never its onset. This is what rejects a rewrite:
|
|
125
209
|
* "revenue" for "churn" and "monetization" for "agents" both score in the
|
|
126
210
|
* same range as a true repair, and both fail the onset test.
|
|
211
|
+
*
|
|
212
|
+
* When either side has no Latin letters there is no key to compare, and this
|
|
213
|
+
* used to answer `ka === kb` — `"" === ""`, i.e. YES for any two non-Latin
|
|
214
|
+
* strings however unrelated. That is no gate at all for an Urdu transcript, so
|
|
215
|
+
* those pairs route to `textSimilarity` instead (2026-08-18 field case).
|
|
216
|
+
* Latin-to-Latin comparisons never reach that branch and are unchanged.
|
|
127
217
|
*/
|
|
128
218
|
export function soundsSimilar(a: string, b: string, floor = SOUNDS_LIKE_FLOOR): boolean {
|
|
129
219
|
const ka = phraseKey(a);
|
|
130
220
|
const kb = phraseKey(b);
|
|
131
|
-
if (ka.length === 0 || kb.length === 0)
|
|
221
|
+
if (ka.length === 0 || kb.length === 0) {
|
|
222
|
+
// `floor` is deliberately NOT reused here: it is calibrated against
|
|
223
|
+
// consonant skeletons, which are shorter and coarser than the text this
|
|
224
|
+
// branch compares, so the same number means something else. Taking the
|
|
225
|
+
// larger of the two was tried and is wrong — the default 0.34 alone
|
|
226
|
+
// rejects a measured genuine repair scoring 0.333. The only caller that
|
|
227
|
+
// raises the floor (reconcileCopy, 0.6) cannot reach this branch anyway:
|
|
228
|
+
// its candidate tokens are stripped to `[A-Za-z]`, so its key is never
|
|
229
|
+
// empty and a non-Latin spoken word scores ~0 against it regardless.
|
|
230
|
+
return textSimilarity(a, b) >= TEXT_SIMILARITY_FLOOR;
|
|
231
|
+
}
|
|
132
232
|
if (ka[0] !== kb[0]) return false;
|
|
133
233
|
return soundsLike(a, b) >= floor;
|
|
134
234
|
}
|
package/src/producer/repair.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { z } from "zod/v4";
|
|
2
2
|
import type { Transcript, Word } from "../schema";
|
|
3
3
|
import type { Scene } from "../scene-schema";
|
|
4
|
-
import { soundsSimilar } from "../phonetics";
|
|
4
|
+
import { normalizeForCompare, soundsSimilar } from "../phonetics";
|
|
5
5
|
import type { LlmProvider } from "./provider";
|
|
6
6
|
|
|
7
7
|
/**
|
|
@@ -105,15 +105,18 @@ export function buildRepairUserPrompt(
|
|
|
105
105
|
);
|
|
106
106
|
}
|
|
107
107
|
|
|
108
|
-
/**
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
108
|
+
/**
|
|
109
|
+
* Comparable form: case- and punctuation-insensitive, in any script.
|
|
110
|
+
*
|
|
111
|
+
* This was a second, Latin-only copy (`[^a-z0-9\s]`) of what is now
|
|
112
|
+
* `normalizeForCompare`. The two drifted in the worst possible way: on an Urdu
|
|
113
|
+
* transcript every string normalized to "", so `norm(actual) ===
|
|
114
|
+
* norm(r.correction)` was true for every proposal and all 11 repairs in a real
|
|
115
|
+
* run were refused as "identical to what was heard" (2026-08-18) — while
|
|
116
|
+
* `locate()`, comparing "" to "", "matched" the first span it tried without
|
|
117
|
+
* verifying anything. One shared helper so they cannot drift again.
|
|
118
|
+
*/
|
|
119
|
+
const norm = normalizeForCompare;
|
|
117
120
|
|
|
118
121
|
function spanText(transcript: Transcript, startWord: number, endWord: number): string {
|
|
119
122
|
return transcript.words
|
|
@@ -228,6 +231,12 @@ export function applyRepairs(
|
|
|
228
231
|
*/
|
|
229
232
|
const locate = (r: TranscriptRepair): { startWord: number; endWord: number } | null => {
|
|
230
233
|
const want = norm(r.heard);
|
|
234
|
+
// A quote that normalizes to nothing is not an anchor: "" compares equal
|
|
235
|
+
// to the first span whose own normalisation is empty, so the search would
|
|
236
|
+
// "find" a span it never verified. That was live for every non-Latin
|
|
237
|
+
// transcript until norm() was fixed above; refuse it explicitly so it
|
|
238
|
+
// cannot come back through some other all-punctuation quote.
|
|
239
|
+
if (want.length === 0) return null;
|
|
231
240
|
// Widths to try, in order of trust. The QUOTED TEXT is the reliable part
|
|
232
241
|
// of a proposal, so its own token count leads; the claimed span is a
|
|
233
242
|
// fallback for a quote whose normalisation splits differently. Trusting
|
package/src/recut.ts
CHANGED
|
@@ -391,3 +391,64 @@ export function applyUserCuts(
|
|
|
391
391
|
removedSec,
|
|
392
392
|
};
|
|
393
393
|
}
|
|
394
|
+
|
|
395
|
+
/** What `pruneHidesInsideCuts` hands back: the doc (same reference when
|
|
396
|
+
* nothing was pruned — the caller's changed-gate reads `pruned.length`), and
|
|
397
|
+
* the retired keys so produce can SAY what it retired. */
|
|
398
|
+
export interface PrunedHides {
|
|
399
|
+
doc: OverrideDoc;
|
|
400
|
+
pruned: string[];
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
/**
|
|
404
|
+
* Retire `captionWordsHidden` entries whose word the final cutlist REMOVES
|
|
405
|
+
* (§59b revisited 2026-08-18 — the "captions + video" delete gesture writes
|
|
406
|
+
* both a hide and a cut in one commit).
|
|
407
|
+
*
|
|
408
|
+
* Once the cut lands, `buildCaptionLines` drops the word before the hide
|
|
409
|
+
* layer ever sees it, so the hide key would report `found: null` ("the cut
|
|
410
|
+
* removed it", `captionHideDropLine`) on every subsequent run forever — the
|
|
411
|
+
* cut SUPERSEDES the hide, the same superseded philosophy `overrides.ts`'s
|
|
412
|
+
* caption-key migration applies. Hides whose source instant is OUTSIDE every
|
|
413
|
+
* removed segment are kept verbatim — as are keys that are not §137 `w<ms>`
|
|
414
|
+
* anchors at all, which name no instant this can test (see the guard below).
|
|
415
|
+
*
|
|
416
|
+
* HALF-OPEN interval (`srcIn <= src < srcOut`), on purpose — the two edges
|
|
417
|
+
* are NOT symmetric. A word starting exactly at `srcIn` IS cut: that is
|
|
418
|
+
* precisely where the FIRST word of a captions+video delete lands (its
|
|
419
|
+
* srcStart round-trips through the TimeMap to the resolved cut's own srcIn),
|
|
420
|
+
* and `mapWord` clamps that instant into the removal and drops the word — a
|
|
421
|
+
* strictly-inside test never retired the gesture's own first hide, leaving
|
|
422
|
+
* it a permanent `found: null` drop report. A word starting exactly at
|
|
423
|
+
* `srcOut` belongs to the NEXT kept span (`buildCaptionLines` still emits it
|
|
424
|
+
* — a seam instant has a kept-side preimage, `timemap.ts`), so its hide is
|
|
425
|
+
* still doing work and must survive.
|
|
426
|
+
*/
|
|
427
|
+
export function pruneHidesInsideCuts(doc: OverrideDoc, cutlist: readonly Segment[]): PrunedHides {
|
|
428
|
+
const pruned: string[] = [];
|
|
429
|
+
const kept: OverrideDoc["captionWordsHidden"] = {};
|
|
430
|
+
for (const [key, entry] of Object.entries(doc.captionWordsHidden)) {
|
|
431
|
+
// SOURCE-KEYED ONLY, parsed and not coerced. `captionWordsHidden` is an
|
|
432
|
+
// unpinned `z.record` (unlike `CaptionRangeEditSchema`'s `/^w\d+$/`), so a
|
|
433
|
+
// hand-edited or legacy-keyed doc reaches here: a POSITIONAL key like "17"
|
|
434
|
+
// would slice to "7", parse as 7ms, land inside any early cut and be
|
|
435
|
+
// deleted with nothing said. The editor guards the identical case and
|
|
436
|
+
// states the rule (`apps/editor/src/useEdits.ts:587-600`): only §137
|
|
437
|
+
// `w<ms>` keys carry an interval-testable instant, and an entry this
|
|
438
|
+
// function cannot honestly locate is KEPT.
|
|
439
|
+
if (!/^w\d+$/.test(key)) {
|
|
440
|
+
kept[key] = entry;
|
|
441
|
+
continue;
|
|
442
|
+
}
|
|
443
|
+
// `captionKeyFor`'s quantization inverted (`w${Math.round(sec * 1000)}`,
|
|
444
|
+
// overrides.ts): the key IS the word's source instant, ms-quantized.
|
|
445
|
+
const srcSec = parseInt(key.slice(1), 10) / 1000;
|
|
446
|
+
const removed = cutlist.some(
|
|
447
|
+
(seg) => seg.kind === "remove" && srcSec >= seg.srcIn && srcSec < seg.srcOut,
|
|
448
|
+
);
|
|
449
|
+
if (removed) pruned.push(key);
|
|
450
|
+
else kept[key] = entry;
|
|
451
|
+
}
|
|
452
|
+
if (pruned.length === 0) return { doc, pruned };
|
|
453
|
+
return { doc: { ...doc, captionWordsHidden: kept }, pruned };
|
|
454
|
+
}
|
package/src/transcribe.ts
CHANGED
|
@@ -78,6 +78,51 @@ function repairSplitSegments(json: WhisperJson): WhisperJson {
|
|
|
78
78
|
};
|
|
79
79
|
}
|
|
80
80
|
|
|
81
|
+
/**
|
|
82
|
+
* Run length at which a stack of zero-length words at ONE instant stops being
|
|
83
|
+
* a rounding artifact and becomes a repetition-loop hallucination. Real speech
|
|
84
|
+
* never emits 8 tokens at a single instant; the field case emitted 118.
|
|
85
|
+
*/
|
|
86
|
+
export const REPETITION_BURST_MIN = 8;
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* Drop whisper repetition-loop bursts (field case 2026-08-18): an Urdu take
|
|
90
|
+
* re-decoded a whole phrase as 118 CONSECUTIVE tokens all stamped
|
|
91
|
+
* `from === to === 31040` — zero length, at one instant. The stamp repair
|
|
92
|
+
* below then fans such a burst out into 118 fabricated 50ms words marching
|
|
93
|
+
* forward from 31.04s, so the phrase ships TWICE in the captions (31.04s and
|
|
94
|
+
* 33.54s) and a fifth of the transcript carries the tell-tale exactly-0.05s
|
|
95
|
+
* duration. `-mc 0` in whisperArgs is the decoder-side mitigation for the same
|
|
96
|
+
* failure; it did not prevent this occurrence, and it can never repair an
|
|
97
|
+
* already-cached transcript.json — hence a parse-side guard too.
|
|
98
|
+
*
|
|
99
|
+
* A burst is a MAXIMAL run of consecutive zero-length/inverted words sharing
|
|
100
|
+
* one `start`. Equality is exact, not epsilon: these stamps are integer
|
|
101
|
+
* milliseconds divided by 1000, so members of one burst are the same double
|
|
102
|
+
* bit-for-bit, and a tolerance would only start swallowing real neighbors.
|
|
103
|
+
* Runs shorter than REPETITION_BURST_MIN fall through untouched — a lone
|
|
104
|
+
* zero-length stamp is a rounding artifact, not a hallucination. The drop is
|
|
105
|
+
* silent by design: this function is pure and total, and there is no logging
|
|
106
|
+
* channel in the parse path to warn on.
|
|
107
|
+
*/
|
|
108
|
+
export function dropRepetitionBursts(words: readonly Word[]): Word[] {
|
|
109
|
+
const out: Word[] = [];
|
|
110
|
+
let i = 0;
|
|
111
|
+
while (i < words.length) {
|
|
112
|
+
const w = words[i]!;
|
|
113
|
+
if (w.end > w.start) {
|
|
114
|
+
out.push(w);
|
|
115
|
+
i++;
|
|
116
|
+
continue;
|
|
117
|
+
}
|
|
118
|
+
let j = i + 1;
|
|
119
|
+
while (j < words.length && words[j]!.end <= words[j]!.start && words[j]!.start === w.start) j++;
|
|
120
|
+
if (j - i < REPETITION_BURST_MIN) for (let k = i; k < j; k++) out.push(words[k]!);
|
|
121
|
+
i = j;
|
|
122
|
+
}
|
|
123
|
+
return out;
|
|
124
|
+
}
|
|
125
|
+
|
|
81
126
|
const STRICT_UTF8 = new TextDecoder("utf-8", { fatal: true });
|
|
82
127
|
|
|
83
128
|
/**
|
|
@@ -115,11 +160,19 @@ export function parseWhisperJson(json: WhisperJson): Transcript {
|
|
|
115
160
|
if (!raw || !raw.trim()) continue;
|
|
116
161
|
const text = raw.trim();
|
|
117
162
|
if (NOISE_TOKEN.test(text)) continue;
|
|
163
|
+
// A word already CLOSED by Arabic-script sentence punctuation refuses
|
|
164
|
+
// continuations (field case 2026-08-18): whisper emits `۔` and the next
|
|
165
|
+
// sentence's first token with no leading whitespace, and the plain
|
|
166
|
+
// whitespace rule fused them into one unsplittable word ("ہوں۔اس").
|
|
167
|
+
// Deliberately NOT the Latin `.`/`!`/`?` — whisper tokenizes decimals
|
|
168
|
+
// ("3", ".", "5") and abbreviations as bare continuations too, and
|
|
169
|
+
// splitting those would shred "3.5" into two words. ۔ (U+06D4) and
|
|
170
|
+
// ؟ (U+061F) have no such second job.
|
|
118
171
|
const startsWord = /^\s/.test(raw) || words.length === 0;
|
|
119
172
|
const start = seg.offsets.from / 1000;
|
|
120
173
|
const end = seg.offsets.to / 1000;
|
|
121
174
|
const last = words[words.length - 1];
|
|
122
|
-
if (!startsWord && last) {
|
|
175
|
+
if (!startsWord && last && !/[۔؟]$/.test(last.text)) {
|
|
123
176
|
last.text += text;
|
|
124
177
|
last.end = Math.max(last.end, end);
|
|
125
178
|
} else {
|
|
@@ -146,14 +199,18 @@ export function parseWhisperJson(json: WhisperJson): Transcript {
|
|
|
146
199
|
else if (next) next.start = Math.min(next.start, w.start);
|
|
147
200
|
words.splice(i, 1);
|
|
148
201
|
}
|
|
202
|
+
// BEFORE the repair loop, never after: the repair rewrites every burst
|
|
203
|
+
// member into a distinct monotone stamp, so once it has run the shared
|
|
204
|
+
// timestamp — the only evidence a burst existed — is gone.
|
|
205
|
+
const kept = dropRepetitionBursts(words);
|
|
149
206
|
// Whisper occasionally emits zero-length or inverted stamps; repair minimally.
|
|
150
|
-
for (let i = 0; i <
|
|
151
|
-
const w =
|
|
207
|
+
for (let i = 0; i < kept.length; i++) {
|
|
208
|
+
const w = kept[i]!;
|
|
152
209
|
if (w.end <= w.start) w.end = w.start + 0.05;
|
|
153
|
-
const next =
|
|
210
|
+
const next = kept[i + 1];
|
|
154
211
|
if (next && next.start < w.end) next.start = w.end;
|
|
155
212
|
}
|
|
156
|
-
return { language: json.result?.language ?? "en", words };
|
|
213
|
+
return { language: json.result?.language ?? "en", words: kept };
|
|
157
214
|
}
|
|
158
215
|
|
|
159
216
|
export interface WhisperOptions {
|
|
@@ -192,6 +249,15 @@ export function whisperArgs(opts: WhisperOptions, wavPath: string): string[] {
|
|
|
192
249
|
"-oj",
|
|
193
250
|
"-of", opts.outBase,
|
|
194
251
|
"-ml", "1",
|
|
252
|
+
// No text context across 30s decode windows (field case 2026-08-18): an
|
|
253
|
+
// Urdu take hit whisper's repetition loop — a whole sentence re-decoded
|
|
254
|
+
// as 261 zero-duration tokens — and carrying the previous window's text
|
|
255
|
+
// into the decoder is the known trigger. `-mc 0` is the standard
|
|
256
|
+
// mitigation and leaves `--prompt` (the dictionary bias) untouched.
|
|
257
|
+
// Cached transcript.json files decoded without it are knowingly still
|
|
258
|
+
// reused (transcriptCacheReusable's no-spurious-retranscribe rule);
|
|
259
|
+
// delete a workdir's transcript.json to re-decode with it.
|
|
260
|
+
"-mc", "0",
|
|
195
261
|
"--no-prints",
|
|
196
262
|
];
|
|
197
263
|
if (opts.language !== undefined) args.push("-l", opts.language);
|