@ossclip/core 0.1.25 → 0.1.26

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ossclip/core",
3
- "version": "0.1.25",
3
+ "version": "0.1.26",
4
4
  "description": "ossclip's framework-free pipeline: schema, transcription, analysis, cutlist, captions, framing, and the LLM producer",
5
5
  "type": "module",
6
6
  "license": "MIT",
package/src/concat.ts CHANGED
@@ -1,6 +1,6 @@
1
1
  import { existsSync } from "node:fs";
2
2
  import { readdir, readFile, rename, rm, stat, writeFile } from "node:fs/promises";
3
- import { join } from "node:path";
3
+ import { isAbsolute, join, relative, resolve } from "node:path";
4
4
  import { z } from "zod/v4";
5
5
  import { probe, type IngestTools } from "./ingest";
6
6
  import { run } from "./exec";
@@ -42,6 +42,78 @@ export interface ConcatEntry {
42
42
  size: number;
43
43
  }
44
44
 
45
+ /**
46
+ * Whether a produce output path lands INSIDE the folder being produced.
47
+ * 2026-08-18 field cascade: a folder input's clips are re-enumerated on
48
+ * EVERY run, so an `--out` written into that folder became a 7th "source
49
+ * clip" on the next run — the folder-content hash changed, produce minted a
50
+ * fresh workdir with EMPTY overrides, and the render silently dropped the
51
+ * user's saved edits while the output's duration doubled. It cascaded three
52
+ * times before the doubling duration was connected to the out path.
53
+ *
54
+ * Containment via `path.relative`, the same idiom as edit.ts's `isInside`:
55
+ * a `startsWith` string-prefix test has no separator boundary and is fooled
56
+ * by a sibling folder that merely shares a prefix (`/a/Clips` vs
57
+ * `/a/Clips-old/out.mp4`) — `child` is inside `parent` iff the relative path
58
+ * from one to the other never has to climb out with `..`. Both arguments are
59
+ * resolved here so a relative path is judged against cwd; `~` expansion is
60
+ * the CALLER's job (the 2026-08-16 rule in the CLI's paths.ts — expandHome
61
+ * at the call site — which also keeps core homedir-free).
62
+ */
63
+ export function outPathInsideInput(outPath: string, inputDir: string): boolean {
64
+ const rel = relative(resolve(inputDir), resolve(outPath));
65
+ return rel === "" || (!rel.startsWith("..") && !isAbsolute(rel));
66
+ }
67
+
68
+ /**
69
+ * The refusal for `outPathInsideInput`, shared verbatim by produce's own
70
+ * gate and the edit server's `/api/render` 400 — one message, one place, so
71
+ * the two boundaries can never describe the same hazard differently. Names
72
+ * the actual risk (the field cascade above) rather than just "invalid path",
73
+ * and suggests the exact default a flag-less run would pick (produce's
74
+ * `defaultOutPath` shape: at most the LAST dot-segment replaced).
75
+ */
76
+ /**
77
+ * The default output path for an input: beside it, `.ossclip.mp4` suffix.
78
+ * Trailing separators are stripped FIRST (2026-08-18 field case, second
79
+ * report): shells tab-complete folders with the slash, and the bare regex
80
+ * appended after it — the "default" landed INSIDE the folder as a hidden
81
+ * `.ossclip.mp4` dotfile, the exact self-ingesting shape
82
+ * `outPathInsideInput` exists to refuse. One definition shared by produce's
83
+ * `defaultOutPath` and the refusal message below, so the suggestion can
84
+ * never disagree with what omitting `--out` actually does.
85
+ */
86
+ export function ossclipOutputPathFor(input: string): string {
87
+ return input.replace(/[/\\]+$/, "").replace(/(\.[^.]+)?$/, ".ossclip.mp4");
88
+ }
89
+
90
+ export function outInsideInputFolderMessage(folder: string): string {
91
+ const suggestion = ossclipOutputPathFor(folder);
92
+ return (
93
+ `refusing to write the output inside the input folder ${folder} — the next ` +
94
+ `run would ingest this output as a source clip and re-plan from scratch, ` +
95
+ `abandoning your edits (the folder's content hash changes, so produce mints ` +
96
+ `a fresh workdir with empty overrides). Write it beside the folder instead, ` +
97
+ `e.g. --out ${suggestion}`
98
+ );
99
+ }
100
+
101
+ /**
102
+ * Defense-in-depth BEHIND the out-path refusal above (same 2026-08-18 field
103
+ * cascade): a produce output already sitting in the folder — written by a
104
+ * pre-fix run, or moved there by hand — must never be ingested as a source
105
+ * clip, and must be filtered BEFORE `folderManifestKey` sees the entries so
106
+ * its presence can't re-key the workdir either. Matches any name carrying
107
+ * the `.ossclip` marker: `X.ossclip.mp4`, `X.ossclip_fixed.mp4`, and
108
+ * `X.ossclip.mp4.partial.mp4` leftovers all qualify (bare `.ossclip.mp4` is
109
+ * a dotfile and already skipped upstream). A custom-named output
110
+ * (`--out final.mp4`) is undetectable here — the out-path refusal is the
111
+ * real gate; this only keeps a pre-fix folder from compounding further.
112
+ */
113
+ export function isOssclipOutputName(name: string): boolean {
114
+ return name.includes(".ossclip");
115
+ }
116
+
45
117
  /**
46
118
  * Order clips for concatenation. `name` (default, per the field request "sort
47
119
  * them by name or date modified, name being default") is a PLAIN codepoint
@@ -321,6 +393,8 @@ export interface FolderListing {
321
393
  entries: ConcatEntry[];
322
394
  /** Files skipped for not matching a video extension (dotfiles excluded). */
323
395
  nonVideoCount: number;
396
+ /** Video files skipped as produce's own outputs (`isOssclipOutputName`). */
397
+ ossclipOutputCount: number;
324
398
  }
325
399
 
326
400
  /**
@@ -339,6 +413,7 @@ export interface FolderListing {
339
413
  export async function listFolderVideos(folder: string): Promise<FolderListing> {
340
414
  const dirents = await readdir(folder, { withFileTypes: true });
341
415
  let nonVideoCount = 0;
416
+ let ossclipOutputCount = 0;
342
417
  const entries: ConcatEntry[] = [];
343
418
  for (const d of dirents) {
344
419
  if (d.name.startsWith(".")) continue;
@@ -357,11 +432,19 @@ export async function listFolderVideos(folder: string): Promise<FolderListing> {
357
432
  nonVideoCount++;
358
433
  continue;
359
434
  }
435
+ // Skipped HERE, before the entries ever exist — never in concatFolder —
436
+ // so an ossclip output left in the folder can neither become a source
437
+ // clip nor perturb the workdir hash `folderManifestKey` derives from
438
+ // this listing (2026-08-18 field cascade, see isOssclipOutputName).
439
+ if (isOssclipOutputName(d.name)) {
440
+ ossclipOutputCount++;
441
+ continue;
442
+ }
360
443
  const st = await stat(join(folder, d.name));
361
444
  entries.push({ name: d.name, mtimeMs: st.mtimeMs, size: st.size });
362
445
  }
363
446
  if (entries.length === 0) throw noVideoFilesError(folder);
364
- return { entries, nonVideoCount };
447
+ return { entries, nonVideoCount, ossclipOutputCount };
365
448
  }
366
449
 
367
450
  export interface FolderConcatResult {
@@ -371,6 +454,8 @@ export interface FolderConcatResult {
371
454
  clips: Array<{ name: string; durationSec: number }>;
372
455
  /** Files skipped for not matching a video extension. */
373
456
  nonVideoCount: number;
457
+ /** Video files skipped as produce's own outputs (`isOssclipOutputName`). */
458
+ ossclipOutputCount: number;
374
459
  /** True when the existing `source-concat.mp4` was reused, not rebuilt. */
375
460
  cached: boolean;
376
461
  /** The concat's own total duration (ffprobe'd from the output). */
@@ -393,7 +478,7 @@ export async function concatFolder(
393
478
  sort: "name" | "mtime",
394
479
  target: { w: number; h: number },
395
480
  ): Promise<FolderConcatResult> {
396
- const { entries: current, nonVideoCount } = listing;
481
+ const { entries: current, nonVideoCount, ossclipOutputCount } = listing;
397
482
  if (current.length === 0) throw noVideoFilesError(folder);
398
483
  const order = planFolderConcat(current, sort);
399
484
 
@@ -408,6 +493,7 @@ export async function concatFolder(
408
493
  path: outPath,
409
494
  clips: order.map((name) => ({ name, durationSec: byName.get(name)!.durationSec })),
410
495
  nonVideoCount,
496
+ ossclipOutputCount,
411
497
  cached: true,
412
498
  durationSec: outProbe.duration,
413
499
  };
@@ -461,6 +547,7 @@ export async function concatFolder(
461
547
  path: outPath,
462
548
  clips: order.map((name, i) => ({ name, durationSec: probes[i]!.duration })),
463
549
  nonVideoCount,
550
+ ossclipOutputCount,
464
551
  cached: false,
465
552
  durationSec: outProbe.duration,
466
553
  };
package/src/overrides.ts CHANGED
@@ -152,6 +152,37 @@ export const CaptionEditSchema = z.object({
152
152
  });
153
153
  export type CaptionEdit = z.infer<typeof CaptionEditSchema>;
154
154
 
155
+ /**
156
+ * A free-text rewrite of a contiguous caption word RUN (2026-08-18) — the one
157
+ * deliberate relaxation of the 1:1 retype contract, for range edits only.
158
+ * Single-word retype (`CaptionEditSchema` above) is untouched, and
159
+ * `transcript.words` is NEVER spliced — scene anchors are raw indices into it
160
+ * — so everything happens on the derived `CaptionLine[]`
161
+ * (`applyCaptionRangeEdits` below).
162
+ *
163
+ * Endpoints are anchored by §137 source-time keys (`captionKeyFor`), so a
164
+ * user cut elsewhere cannot shift the run. `was` is the NFC-normalized,
165
+ * space-joined BASE text of the run — the `captionEditWas` base-truth rule,
166
+ * run-wide: the reducer scrubs every per-word retype inside the interval in
167
+ * the same commit that stores the entry, so the run `applyCaptionRangeEdits`
168
+ * reads at apply time IS the base run, and a live (post-retype) join would
169
+ * fail the guard forever. A WHOLE-RUN stale guard: if any word in the run is
170
+ * re-worded or cut later, the entire edit is reported dropped, never
171
+ * partially guessed at. Identity is the `(fromKey, toKey)`
172
+ * pair — retyping the run back to its `was` DELETES the entry (the
173
+ * clearVideo/`patchCaption` rule). An array like `cuts`, `.default([])` so
174
+ * every pre-existing overrides.json parses byte-identically. NEVER
175
+ * legacy-keyed: the field postdates §137, so `migrateCaptionKeys` must not
176
+ * process it — there are no positional range edits to upgrade.
177
+ */
178
+ export const CaptionRangeEditSchema = z.object({
179
+ fromKey: z.string().regex(/^w\d+$/),
180
+ toKey: z.string().regex(/^w\d+$/),
181
+ text: z.string().min(1).max(400),
182
+ was: z.string(),
183
+ });
184
+ export type CaptionRangeEdit = z.infer<typeof CaptionRangeEditSchema>;
185
+
155
186
  /**
156
187
  * The `was` a caption edit should store (R15 §59). The FIRST edit's `was` is
157
188
  * the base truth (the word as transcribed); every later re-edit of the same
@@ -169,6 +200,25 @@ export function captionEditWas(
169
200
  return captions[key]?.was ?? seen;
170
201
  }
171
202
 
203
+ /**
204
+ * The `was` a RANGE edit should store — `captionEditWas` for the
205
+ * `(fromKey, toKey)` pair. The first edit's `was` is the base truth; a
206
+ * re-edit of the SAME run (its endpoints are re-minted verbatim, see
207
+ * `applyCaptionRangeEdits`' srcStart minting) sees the LIVE, already-rewritten
208
+ * text, and storing that as `was` would stale the guard against the base
209
+ * lines the next apply runs on. Preserving the existing pair's `was` keeps
210
+ * the guard anchored to the base — and makes "retyped back to the original"
211
+ * detectable, which is when the entry should clear entirely.
212
+ */
213
+ export function captionRangeEditWas(
214
+ rangeEdits: readonly CaptionRangeEdit[],
215
+ fromKey: string,
216
+ toKey: string,
217
+ seen: string,
218
+ ): string {
219
+ return rangeEdits.find((e) => e.fromKey === fromKey && e.toKey === toKey)?.was ?? seen;
220
+ }
221
+
172
222
  /**
173
223
  * The id a pre-§137 split gets when it is upgraded: the output milliseconds of
174
224
  * whatever `at` the file holds NOW.
@@ -256,6 +306,76 @@ export const OverrideDocSchema = z.object({
256
306
  scenes: z.record(z.string(), SceneOverrideSchema).default({}),
257
307
  /** Retyped caption words, keyed by the word's source time (§137). */
258
308
  captions: z.record(z.string(), CaptionEditSchema).default({}),
309
+ /**
310
+ * Per-word caption HIDES ("delete word from captions") — non-destructive:
311
+ * the word stays in the transcript and in the video's audio; only the
312
+ * rendered caption drops it. Keyed by the word's source time
313
+ * (`captionKeyFor`, §137) like `captions` above, so a user cut never
314
+ * shifts a hide onto a different word. `was` is the LIVE (post-retype)
315
+ * text at hide time — hides apply AFTER retypes (`applyCaptionLayers`
316
+ * below) — the same stale-guard contract as `CaptionEditSchema.was`:
317
+ * a re-derived stream under a surviving anchor drops the hide WITH A
318
+ * REPORT rather than deleting the wrong word. Restore DELETES the key
319
+ * (the restoreScene/captionsHidden rule — an entry with nothing to say is
320
+ * still an override), and `.default({})` keeps every pre-existing
321
+ * overrides.json parsing byte-identically. This field NEVER existed in
322
+ * the legacy positional-key era, so `migrateCaptionKeys` must NOT process
323
+ * it — there are no legacy hides to upgrade.
324
+ */
325
+ captionWordsHidden: z.record(z.string(), z.object({ was: z.string() })).default({}),
326
+ /**
327
+ * Multi-word free-text rewrites — see `CaptionRangeEditSchema` for the
328
+ * whole contract (endpoint anchoring, the whole-run `was` guard, identity
329
+ * by pair, why it is never legacy-keyed). Applied between per-word retypes
330
+ * and hides (`applyCaptionLayers`).
331
+ */
332
+ captionRangeEdits: z.array(CaptionRangeEditSchema).default([]),
333
+ /**
334
+ * Per-LINE caption TIMING nudges — "when does this caption appear, and when
335
+ * does it leave". Stored as DELTAS against the DERIVED window (`lead` moves
336
+ * the line's OPENING seam, `tail` its CLOSING seam), keyed by the LINE's
337
+ * FIRST WORD's SOURCE time (`captionKeyFor`, §137). Deltas over source keys
338
+ * make the record recut-immune for free: a recut rebuilds every derived
339
+ * `start`/`end` through the new TimeMap and the deltas simply re-apply on
340
+ * top — zero work in `remapOverridesThroughRecut`, the same property every
341
+ * other caption record leans on (captions.ts:14-20: `srcStart` is the one
342
+ * field a re-cut cannot move). Restore DELETES the key, and a patch whose
343
+ * deltas are both under 1ms in magnitude also deletes (the clearVideo/
344
+ * patchCaption clear-override rule — a nudge of nothing is still an
345
+ * override). `.default({})` keeps every pre-existing overrides.json parsing
346
+ * byte-identically, and the field NEVER existed in the legacy
347
+ * positional-key era, so `migrateCaptionKeys` must not process it.
348
+ *
349
+ * PER LINE, NOT PER WORD, and that is the whole point of the field. It
350
+ * replaces `captionWordTiming` (deleted 2026-08-18), which stored the same
351
+ * shape against individual WORDS and was measured to be MATHEMATICALLY
352
+ * INERT: on a live workdir (117 lines / 301 words) 116/116 inter-line gaps
353
+ * were exactly 0.0, 184/184 intra-line word boundaries exactly 0.0,
354
+ * `line.start === words[0].start` 117/117 and `line.end === lastWord.end`
355
+ * 117/117 — `transcribe.ts` chains words (`next.start = w.end`) and
356
+ * `captions.ts:203-213`'s hold pass clamps each line's end to the next
357
+ * line's start, so the caption stream is a GAP-FREE PARTITION. A per-word
358
+ * clamp of `[max(lineStart, prevEnd), min(lineEnd, nextStart)]` therefore
359
+ * collapsed to exactly `[w.start, w.end]` for EVERY word: the user dragged,
360
+ * every stored delta came back zero, and the reducer's sub-ms rule deleted
361
+ * them again. Do not reintroduce word-level clamping against a packed
362
+ * stream. Word stamps also only drive the karaoke highlight INSIDE a line's
363
+ * `<Sequence>` window (CaptionTrack.tsx:228-229, 387) — "when a caption
364
+ * appears" IS `line.start`/`line.end`, so timing has to move LINE windows.
365
+ * The ±30s range is per SEAM, which is why it is wider than the old
366
+ * per-word ±10s: a line may be dragged well clear of its neighbours, and
367
+ * `applyCaptionLineTiming`'s sweep — not the schema — is what keeps seams
368
+ * ordered and inside the track.
369
+ */
370
+ captionLineTiming: z
371
+ .record(
372
+ z.string(),
373
+ z.object({
374
+ lead: z.number().min(-30).max(30),
375
+ tail: z.number().min(-30).max(30),
376
+ }),
377
+ )
378
+ .default({}),
259
379
  /**
260
380
  * Scene split points. `at` is ABSOLUTE output seconds (R16 §61 — Cmd/Ctrl+B
261
381
  * at the playhead) and moves when a re-cut re-anchors the doc; `id` is
@@ -978,6 +1098,614 @@ export function applyCaptionEdits(
978
1098
  return { lines: out, dropped };
979
1099
  }
980
1100
 
1101
+ /**
1102
+ * Re-time replacement tokens over ONE line's stretch of a rewritten run —
1103
+ * `repair.ts`'s `retime` model (producer/repair.ts:137-154), restated here
1104
+ * for CaptionWords: stamps distributed across the window weighted by token
1105
+ * length + 1, strictly increasing, the last token's `end` pinned to the
1106
+ * window end so the run never leaks past the span it replaced. The measured
1107
+ * window edges (first run word's start, last run word's end) are kept;
1108
+ * only the interior boundaries are interpolated — interpolated boundaries
1109
+ * are a guess, and `retime`'s comment is explicit that a guess must never
1110
+ * displace a measurement, which is why the equal-count fast path in
1111
+ * `applyCaptionRangeEdits` below bypasses this entirely.
1112
+ */
1113
+ function retimeCaptionTokens(
1114
+ tokens: readonly string[],
1115
+ windowStart: number,
1116
+ windowEnd: number,
1117
+ srcStarts: readonly number[],
1118
+ ): CaptionWord[] {
1119
+ const weights = tokens.map((t) => t.length + 1);
1120
+ const total = weights.reduce((a, b) => a + b, 0);
1121
+ const out: CaptionWord[] = [];
1122
+ let cursor = windowStart;
1123
+ for (let i = 0; i < tokens.length; i++) {
1124
+ const share = ((windowEnd - windowStart) * weights[i]!) / total;
1125
+ const end = i === tokens.length - 1 ? windowEnd : cursor + share;
1126
+ out.push({ text: tokens[i]!, start: cursor, end, srcStart: srcStarts[i]! });
1127
+ cursor = end;
1128
+ }
1129
+ return out;
1130
+ }
1131
+
1132
+ /**
1133
+ * Apply the free-text RANGE rewrites (`captionRangeEdits`) — the one layer
1134
+ * allowed to change word COUNT, which is why it exists at all: everything it
1135
+ * reshapes is the derived `CaptionLine[]`, never `transcript.words` (scene
1136
+ * anchors are raw indices into that array — splicing it is the forbidden
1137
+ * operation this whole edit family is built around).
1138
+ *
1139
+ * Same reporting shape as `applyCaptionEdits`; drop `key`s are the COMPOSITE
1140
+ * `${fromKey}..${toKey}` — the pair is the entry's identity, and either half
1141
+ * alone names only an endpoint. Each entry drops AT MOST ONCE (unlike the
1142
+ * per-word layers, where one key can be reported per extra claimant), which
1143
+ * is what lets `reconcileCaptionEdits` count applied entries by subtraction.
1144
+ *
1145
+ * Locating: `fromKey`'s first claimant across the flat word order (the
1146
+ * per-word first-claimant rule — ms-quantised keys CAN collide,
1147
+ * captions.ts:44-50), then a FORWARD walk to `toKey`; a missing endpoint, or
1148
+ * a `toKey` that only occurs before `fromKey`, is `found: null`. An entry
1149
+ * whose pair was already applied, or whose `fromKey` an earlier range edit's
1150
+ * run consumed, is `duplicate-anchor` — reachable only in a hand-edited doc,
1151
+ * since the reducer scrubs overlapping entries at creation, and reported
1152
+ * rather than guessed at like every other collision in this file.
1153
+ *
1154
+ * The whole-run stale guard: the run's live texts, NFC-normalized and
1155
+ * space-joined, must equal `was` byte for byte, or the WHOLE edit drops with
1156
+ * the joined text as `found` — never a partial rewrite of the words that
1157
+ * still match (a half-applied rewrite reads as garbage, and there is no
1158
+ * per-word truth to fall back on once the counts differ).
1159
+ *
1160
+ * Retiming across lines: the run may span several lines, and their `start`/
1161
+ * `end` WINDOWS are deliberately not re-packed — Sequence windows and
1162
+ * `buildCaptionLines`' breakpoint semantics stay exactly as produced.
1163
+ * Replacement tokens are distributed across the affected lines
1164
+ * proportionally to each line's share of the run's summed word duration,
1165
+ * rounded by largest remainder (deterministic — earlier line wins a tie) so
1166
+ * every token lands somewhere and the totals match. Within a line the stamps
1167
+ * follow `retimeCaptionTokens` above; a token count equal to the run's word
1168
+ * count skips all of it and keeps the measured per-word stamps AND srcStarts
1169
+ * verbatim (measured ASR boundaries beat interpolation — `retime`'s rule).
1170
+ * A line allotted zero tokens loses its run words, and if that empties it
1171
+ * the line is omitted (the `applyCaptionWordHides` rule — no zero-word
1172
+ * Sequence).
1173
+ *
1174
+ * srcStart minting for count-changed runs: linear across `[fromSrc, toSrc]`
1175
+ * (the endpoints' own source starts), endpoints re-minted verbatim — which
1176
+ * is what lets the user select a rewritten run again and edit it (its
1177
+ * endpoints still answer to the same pair). Strictly increasing whenever the
1178
+ * span is non-degenerate; when the span is too short for 1ms-distinct
1179
+ * quantised keys (`captionKeyFor` rounds to ms), later words SHARE quantised
1180
+ * keys — an accepted, documented duplicate-anchor case the existing
1181
+ * machinery reports if a per-word edit ever targets one.
1182
+ */
1183
+ export function applyCaptionRangeEdits(
1184
+ lines: readonly CaptionLine[],
1185
+ rangeEdits: readonly CaptionRangeEdit[],
1186
+ ): AppliedCaptionEdits {
1187
+ const dropped: AppliedCaptionEdits["dropped"] = [];
1188
+ if (rangeEdits.length === 0) return { lines: [...lines], dropped };
1189
+
1190
+ let out: CaptionLine[] = [...lines];
1191
+ const seenPairs = new Set<string>();
1192
+ const consumed = new Set<string>();
1193
+
1194
+ for (const entry of rangeEdits) {
1195
+ const key = `${entry.fromKey}..${entry.toKey}`;
1196
+ // Flatten the CURRENT lines — edits apply sequentially, so a later entry
1197
+ // addresses the stream as the earlier ones left it (that is how a
1198
+ // re-minted endpoint stays addressable at all).
1199
+ const flat: Array<{ line: number; word: number; w: CaptionWord }> = [];
1200
+ for (let li = 0; li < out.length; li++) {
1201
+ for (let wi = 0; wi < out[li]!.words.length; wi++) {
1202
+ flat.push({ line: li, word: wi, w: out[li]!.words[wi]! });
1203
+ }
1204
+ }
1205
+ const fromIdx = flat.findIndex((f) => captionAnchorOf(f.w) === entry.fromKey);
1206
+ if (seenPairs.has(key) || consumed.has(entry.fromKey)) {
1207
+ dropped.push({
1208
+ key,
1209
+ expected: entry.was,
1210
+ found: fromIdx === -1 ? null : flat[fromIdx]!.w.text,
1211
+ reason: "duplicate-anchor",
1212
+ });
1213
+ continue;
1214
+ }
1215
+ if (fromIdx === -1) {
1216
+ dropped.push({ key, expected: entry.was, found: null });
1217
+ continue;
1218
+ }
1219
+ // FORWARD only: a toKey sitting before fromKey is a run that crosses a
1220
+ // gap the stream no longer bridges — `found: null`, never a guess.
1221
+ let toIdx = -1;
1222
+ for (let i = fromIdx; i < flat.length; i++) {
1223
+ if (captionAnchorOf(flat[i]!.w) === entry.toKey) {
1224
+ toIdx = i;
1225
+ break;
1226
+ }
1227
+ }
1228
+ if (toIdx === -1) {
1229
+ dropped.push({ key, expected: entry.was, found: null });
1230
+ continue;
1231
+ }
1232
+ const run = flat.slice(fromIdx, toIdx + 1);
1233
+ const joined = run
1234
+ .map((f) => f.w.text)
1235
+ .join(" ")
1236
+ .normalize("NFC");
1237
+ if (joined !== entry.was.normalize("NFC")) {
1238
+ dropped.push({ key, expected: entry.was, found: joined });
1239
+ continue;
1240
+ }
1241
+ const tokens = entry.text.trim().split(/\s+/).filter(Boolean);
1242
+ if (tokens.length === 0) {
1243
+ // Defensive: zod's min(1) admits a whitespace-only string, and a run
1244
+ // rewritten to NOTHING is a delete, which is the hide layer's job —
1245
+ // treated as a stale-style drop rather than silently emptying the run.
1246
+ dropped.push({ key, expected: entry.was, found: joined });
1247
+ continue;
1248
+ }
1249
+ seenPairs.add(key);
1250
+ for (const f of run) {
1251
+ const a = captionAnchorOf(f.w);
1252
+ if (a !== null) consumed.add(a);
1253
+ }
1254
+
1255
+ if (tokens.length === run.length) {
1256
+ // Equal count: keep the measured stamps AND srcStarts verbatim —
1257
+ // `retime`'s fast path, for its reason (measured ASR onsets beat any
1258
+ // interpolation, and verbatim srcStarts keep every anchor addressable).
1259
+ const replaced = new Map(run.map((f, i) => [`${f.line}:${f.word}`, tokens[i]!]));
1260
+ out = out.map((line, li) => ({
1261
+ ...line,
1262
+ words: line.words.map((w, wi) => {
1263
+ const text = replaced.get(`${li}:${wi}`);
1264
+ return text === undefined ? w : { ...w, text };
1265
+ }),
1266
+ }));
1267
+ continue;
1268
+ }
1269
+
1270
+ // Count changed: distribute tokens across the affected lines by each
1271
+ // line's share of the run's total duration, largest-remainder rounded.
1272
+ const lineOrder: number[] = [];
1273
+ const runByLine = new Map<number, { first: number; last: number; words: CaptionWord[] }>();
1274
+ for (const f of run) {
1275
+ const seg = runByLine.get(f.line);
1276
+ if (seg) {
1277
+ seg.last = f.word;
1278
+ seg.words.push(f.w);
1279
+ } else {
1280
+ lineOrder.push(f.line);
1281
+ runByLine.set(f.line, { first: f.word, last: f.word, words: [f.w] });
1282
+ }
1283
+ }
1284
+ const shares = lineOrder.map((li) =>
1285
+ runByLine.get(li)!.words.reduce((a, w) => a + (w.end - w.start), 0),
1286
+ );
1287
+ const totalShare = shares.reduce((a, b) => a + b, 0);
1288
+ // Zero total duration (every run word zero-width) has no proportion to
1289
+ // honor — fall back to equal weights so the rounding below still lands
1290
+ // every token somewhere deterministic.
1291
+ const weights = totalShare > 0 ? shares : shares.map(() => 1);
1292
+ const weightTotal = totalShare > 0 ? totalShare : shares.length;
1293
+ const quotas = weights.map((s) => (tokens.length * s) / weightTotal);
1294
+ const counts = quotas.map((q) => Math.floor(q));
1295
+ let leftover = tokens.length - counts.reduce((a, b) => a + b, 0);
1296
+ // Largest remainder first; ties break to the EARLIER line — stated so
1297
+ // the distribution is reproducible from the doc alone, like every other
1298
+ // persisted derivation in this file.
1299
+ const byRemainder = quotas
1300
+ .map((q, i) => ({ i, rem: q - Math.floor(q) }))
1301
+ .sort((a, b) => b.rem - a.rem || a.i - b.i);
1302
+ for (let k = 0; leftover > 0; k = (k + 1) % byRemainder.length) {
1303
+ counts[byRemainder[k]!.i]!++;
1304
+ leftover--;
1305
+ }
1306
+
1307
+ const fromSrc = run[0]!.w.srcStart;
1308
+ const toSrc = run[run.length - 1]!.w.srcStart;
1309
+ const srcStarts = tokens.map((_, j) =>
1310
+ tokens.length === 1 ? fromSrc : fromSrc + ((toSrc - fromSrc) * j) / (tokens.length - 1),
1311
+ );
1312
+
1313
+ let tokenCursor = 0;
1314
+ const next: CaptionLine[] = [];
1315
+ for (let li = 0; li < out.length; li++) {
1316
+ const line = out[li]!;
1317
+ const seg = runByLine.get(li);
1318
+ if (!seg) {
1319
+ next.push(line);
1320
+ continue;
1321
+ }
1322
+ const n = counts[lineOrder.indexOf(li)]!;
1323
+ const lineTokens = tokens.slice(tokenCursor, tokenCursor + n);
1324
+ const lineSrcs = srcStarts.slice(tokenCursor, tokenCursor + n);
1325
+ tokenCursor += n;
1326
+ const minted =
1327
+ n === 0
1328
+ ? []
1329
+ : retimeCaptionTokens(
1330
+ lineTokens,
1331
+ seg.words[0]!.start,
1332
+ seg.words[seg.words.length - 1]!.end,
1333
+ lineSrcs,
1334
+ );
1335
+ const words = [...line.words.slice(0, seg.first), ...minted, ...line.words.slice(seg.last + 1)];
1336
+ // Window untouched (the no-re-pack rule above); an emptied line is
1337
+ // omitted, same as `applyCaptionWordHides`.
1338
+ if (words.length === 0) continue;
1339
+ next.push({ ...line, words });
1340
+ }
1341
+ out = next;
1342
+ }
1343
+ return { lines: out, dropped };
1344
+ }
1345
+
1346
+ /**
1347
+ * Drop hidden caption words (the `captionWordsHidden` layer). Same reporting
1348
+ * shape as `applyCaptionEdits` — callers must surface `dropped` for the same
1349
+ * reason: a hide that silently fails looks like the editor forgot it.
1350
+ *
1351
+ * Runs on the DERIVED `CaptionLine[]`, never on `transcript.words` — scene
1352
+ * anchors are raw word INDICES into the transcript, so splicing a word out of
1353
+ * it would shift every later anchor onto the wrong word: the forbidden
1354
+ * operation this whole layer exists to avoid. The transcript stays intact;
1355
+ * only the rendered caption stream loses the word.
1356
+ *
1357
+ * Line WINDOWS are recomputed here, deliberately: `buildCaptionLines` derives
1358
+ * `start` from the first word and `end` from the last word plus a hold
1359
+ * (captions.ts:203-213), so hiding a boundary word would otherwise leave the
1360
+ * line lingering on screen over silence — up for the hidden first word's
1361
+ * duration, or held past the hidden last word's end. A hidden FIRST word moves
1362
+ * `start` to the first survivor; a hidden LAST word re-bases the packer's hold
1363
+ * delta onto whichever word is now last (clamped so the line never ends before
1364
+ * its own last word); middle hides leave the window alone. A line whose words
1365
+ * are ALL hidden is omitted entirely, so the downstream CaptionTrack emits no
1366
+ * Sequence for it.
1367
+ *
1368
+ * `was` is the LIVE (post-retype) text at hide time — hides apply AFTER
1369
+ * retypes (`applyCaptionLayers` below) — so un-retyping a word under a hide
1370
+ * stales the hide, and it is REPORTED rather than guessed at. Same
1371
+ * first-claimant rule as `applyCaptionEdits`: ms-quantised keys CAN collide
1372
+ * (`captions.ts:44-50` manufactures duplicates by design), and one hide must
1373
+ * remove one word, not every word sharing its instant.
1374
+ */
1375
+ export function applyCaptionWordHides(
1376
+ lines: readonly CaptionLine[],
1377
+ hides: Record<string, { was: string }>,
1378
+ ): AppliedCaptionEdits {
1379
+ const dropped: AppliedCaptionEdits["dropped"] = [];
1380
+ if (Object.keys(hides).length === 0) return { lines: [...lines], dropped };
1381
+
1382
+ const seen = new Set<string>();
1383
+ const out: CaptionLine[] = [];
1384
+ for (const line of lines) {
1385
+ const kept: CaptionWord[] = [];
1386
+ for (const w of line.words) {
1387
+ // No anchor, no hide — same boundary rule as `applyCaptionEdits`: a
1388
+ // pre-§137 word cannot be addressed, and the stored hides then fall out
1389
+ // of the sweep below as `found: null`.
1390
+ const key = captionAnchorOf(w);
1391
+ const hide = key === null ? undefined : hides[key];
1392
+ if (key === null || !hide) {
1393
+ kept.push(w);
1394
+ continue;
1395
+ }
1396
+ // An earlier word already answered for this anchor — whichever way it
1397
+ // answered. Hiding here too would fan one delete onto a second word.
1398
+ if (seen.has(key)) {
1399
+ dropped.push({ key, expected: hide.was, found: w.text, reason: "duplicate-anchor" });
1400
+ kept.push(w);
1401
+ continue;
1402
+ }
1403
+ seen.add(key);
1404
+ if (w.text !== hide.was) {
1405
+ dropped.push({ key, expected: hide.was, found: w.text });
1406
+ kept.push(w);
1407
+ continue;
1408
+ }
1409
+ // Matched: the word is dropped from the line.
1410
+ }
1411
+ if (kept.length === line.words.length) {
1412
+ out.push(line);
1413
+ continue;
1414
+ }
1415
+ // Every word hidden — the line goes with them, rather than a zero-word
1416
+ // line the CaptionTrack would still mount a Sequence for.
1417
+ if (kept.length === 0) continue;
1418
+ const lastOriginal = line.words[line.words.length - 1]!;
1419
+ const firstKept = kept[0]!;
1420
+ const lastKept = kept[kept.length - 1]!;
1421
+ const start = firstKept === line.words[0] ? line.start : firstKept.start;
1422
+ // The packer's hold delta (captions.ts:203-213) rides on whichever word
1423
+ // is now last; clamped so the line never ends before its own last word
1424
+ // (the delta can be negative when the hold was clamped to outputDuration).
1425
+ const end =
1426
+ lastKept === lastOriginal
1427
+ ? line.end
1428
+ : Math.max(lastKept.end, lastKept.end + (line.end - lastOriginal.end));
1429
+ out.push({ words: kept, start, end });
1430
+ }
1431
+
1432
+ // An anchor no word carries any more — a later cut removed the word the
1433
+ // user hid. Silence here is the field-case failure mode, so say it.
1434
+ for (const [key, hide] of Object.entries(hides)) {
1435
+ if (!seen.has(key)) dropped.push({ key, expected: hide.was, found: null });
1436
+ }
1437
+ return { lines: out, dropped };
1438
+ }
1439
+
1440
+ /**
1441
+ * The floor a caption's window may shrink to. A caption nobody can read is a
1442
+ * delete wearing a timing nudge's clothes — deletes are the hide layer's
1443
+ * gesture, with its own guard and report. Also the minimum WIDTH of every
1444
+ * line's window, which is what keeps §115 (`packages/scenes/src/frames.ts`)
1445
+ * true: 50ms is more than one frame at any fps this renders at, so two
1446
+ * adjacent windows can never round onto the same frame.
1447
+ *
1448
+ * (Was `MIN_TIMED_WORD_SEC`, the same 0.05 measured against a WORD, until the
1449
+ * per-word layer was found inert — see `captionLineTiming`'s docstring.)
1450
+ *
1451
+ * Exported for the EDITOR's drag bounds (`captionDragBounds`,
1452
+ * apps/editor/src/TranscriptPanel.tsx): the popover has to stop a drag exactly
1453
+ * where this sweep would, and a second copy of the floor in the browser is how
1454
+ * the two would drift apart.
1455
+ */
1456
+ export const MIN_CAPTION_SEC = 0.05;
1457
+
1458
+ /**
1459
+ * Re-time a line's words from one window onto another, PROPORTIONALLY — the
1460
+ * arithmetic that keeps the karaoke highlight in sync when a line's
1461
+ * `<Sequence>` window moves under it (`CaptionTrack.tsx:228-229, 387` reads
1462
+ * the word stamps INSIDE the window; a window moved without them would light
1463
+ * the wrong words up, or none).
1464
+ *
1465
+ * The source is the WINDOW, not the words' own span: on the packed stream
1466
+ * both are the same interval (`line.start === words[0].start` and
1467
+ * `line.end === lastWord.end`, measured 117/117 — `captionLineTiming`'s
1468
+ * docstring), and on a line that DOES carry lead-in or hold (the hide layer
1469
+ * can re-base either edge) mapping the window preserves that slack instead of
1470
+ * stretching the words over it.
1471
+ *
1472
+ * Pure and exported so a caller previewing a drag and the apply pass below
1473
+ * share ONE piece of arithmetic (the openCommand/openInBrowser split).
1474
+ * Identity when the source window is degenerate — a zero-width or inverted
1475
+ * span has no ratio to scale by, and `0/0` would put NaN stamps in the render
1476
+ * props. The caller owns `toStart < toEnd`; a target handed backwards would
1477
+ * mirror the word order, which `applyCaptionLineTiming`'s edge sweep makes
1478
+ * unreachable.
1479
+ */
1480
+ export function scaleWordsIntoWindow(
1481
+ words: readonly CaptionWord[],
1482
+ fromStart: number,
1483
+ fromEnd: number,
1484
+ toStart: number,
1485
+ toEnd: number,
1486
+ ): CaptionWord[] {
1487
+ const span = fromEnd - fromStart;
1488
+ if (!(span > 0)) return words.map((w) => ({ ...w }));
1489
+ const ratio = (toEnd - toStart) / span;
1490
+ const at = (t: number): number => toStart + (t - fromStart) * ratio;
1491
+ return words.map((w) => ({ ...w, start: at(w.start), end: at(w.end) }));
1492
+ }
1493
+
1494
+ /**
1495
+ * Apply per-LINE caption TIMING nudges (`captionLineTiming`) — the LAST
1496
+ * layer, after hides, because it must operate on the SURVIVING lines: a hide
1497
+ * can move a line's window (or remove the line entirely), and a nudge stored
1498
+ * on a line the hides emptied has no window to move (it falls out of the
1499
+ * sweep as `found: null`, like every other orphaned caption record).
1500
+ *
1501
+ * EDGES, NOT ONE SHARED SEAM. Each line owns its `[start, end]` pair, and a
1502
+ * line's END and the next line's START are two separate numbers here — even
1503
+ * though on a real transcript they are always equal, because the packer chains
1504
+ * words (`transcribe.ts`: `next.start = w.end`) and clamps each line's end to
1505
+ * the next line's start (`captions.ts:203-213`), giving inter-line gaps of
1506
+ * exactly zero (measured 116/116, see `captionLineTiming`). A nudge CLOSES the
1507
+ * two onto one value only when they were already COINCIDENT: that is what
1508
+ * makes a lead on the packed stream move both sides of the boundary, one edit
1509
+ * and two windows, exactly as before.
1510
+ *
1511
+ * They are two numbers because GAPS ARE REAL: `applyCaptionWordHides` re-bases
1512
+ * a line's window onto its surviving words, `MAX_CAPTION_WORD_LEAD_SEC`
1513
+ * (captions.ts:147, 169) clamps a word's display start, and an overrides.json
1514
+ * can be hand-edited. This code
1515
+ * used to hold ONE `seams` array whose interior entry was read off the later
1516
+ * line's start, conflating the two: with lines `[0,2] [2,4] [5,6]`, a
1517
+ * lead-only drag of the middle line (`{lead: -0.05, tail: 0}`, exactly what
1518
+ * the editor writes) rebuilt the UNTOUCHED third caption as `[4,6]` — a full
1519
+ * second early, its words stretched 2x by `scaleWordsIntoWindow`, with no drop
1520
+ * reported (review 2026-08-19).
1521
+ *
1522
+ * The edge model still protects §115 (`packages/scenes/src/frames.ts:1-21` —
1523
+ * no two lines may share a frame) BY CONSTRUCTION, which is what the old "LINE
1524
+ * WINDOWS NEVER CHANGE" rule existed for: the sweep below leaves the edges
1525
+ * ORDERED (`start_0 <= end_0 <= start_1 <= ... <= end_n-1`) with every window
1526
+ * at least `MIN_CAPTION_SEC` wide, and ordered non-overlapping windows at
1527
+ * least 50ms wide cannot round onto a shared frame.
1528
+ *
1529
+ * THE SWEEP, forward: every edge is clamped into the track's ORIGINAL outer
1530
+ * bounds, no line may open before the previous line CLOSED, and no window may
1531
+ * be narrower than `MIN_CAPTION_SEC`. A backward pass then pulls lines left if
1532
+ * a track too short to hold every line at the floor made the forward pass run
1533
+ * into the end. Ordering is enforced against the NEIGHBOUR'S OWN edge, never a
1534
+ * derived seam: a nudge that runs past it is BLOCKED there rather than pushing
1535
+ * it, so a gap gets consumed but no untouched caption ever moves. (A nudge
1536
+ * takes time FROM a neighbour only through the coincidence rule above — the
1537
+ * packed case, where the two share the boundary being dragged.) The outer
1538
+ * bounds never GROW: a caption must not appear before the first caption of the
1539
+ * track or linger past the last, where there is no output left to show it
1540
+ * over.
1541
+ *
1542
+ * BOTH SIDES OF ONE BOUNDARY: line i's `tail` and line i+1's `lead` address
1543
+ * the same coincident boundary. The LATER line's `lead` wins,
1544
+ * deterministically — the UI writes both sides of a drag consistently, so this
1545
+ * only decides hand-edited docs, and a stated winner beats an
1546
+ * order-of-iteration accident. (A stored `lead: 0` still claims its edge; an
1547
+ * entry the user cleared is DELETED from the doc, not written as zeros.)
1548
+ *
1549
+ * Lines whose window the sweep did not move are returned VERBATIM — including
1550
+ * their word stamps — so a nudge on one caption cannot perturb the rest of
1551
+ * the track. The ones that did move (the nudged line AND, on a coincident
1552
+ * boundary, its neighbour) have their words scaled into the new window by
1553
+ * `scaleWordsIntoWindow`.
1554
+ *
1555
+ * DELIBERATELY NO `was` GUARD, unlike `captionWordsHidden`: timing is
1556
+ * text-orthogonal — a retype under a timing nudge changes what the caption
1557
+ * says, not when it is said, and staleness on text would drop nudges the user
1558
+ * never un-meant. `expected` in the drop reports is therefore always `""`
1559
+ * (the record stores no text to expect). Same first-claimant rule as every
1560
+ * per-word layer: ms-quantised anchors CAN collide (captions.ts:44-50), and
1561
+ * one nudge must move one line.
1562
+ */
1563
+ export function applyCaptionLineTiming(
1564
+ lines: readonly CaptionLine[],
1565
+ timing: Record<string, { lead: number; tail: number }>,
1566
+ ): AppliedCaptionEdits {
1567
+ const dropped: AppliedCaptionEdits["dropped"] = [];
1568
+ const n = lines.length;
1569
+ // NO LINES is not "no nudges to report": every stored key is an anchor that
1570
+ // no line starts on, which is exactly the `found: null` case the sweep at
1571
+ // the bottom exists to say out loud, and what this function's own docstring
1572
+ // promises. `applyCaptionEdits` and `applyCaptionWordHides` never took this
1573
+ // shortcut either. The editor's false-banner guard lives at the CALLER
1574
+ // (`App.tsx`: `if (!renderProps) return { lines: [], dropped: [] }`), where
1575
+ // "nothing loaded yet" is distinguishable from "this cut has no captions" —
1576
+ // silence here instead let produce report nudges as applied that never were.
1577
+ if (n === 0) {
1578
+ for (const key of Object.keys(timing)) dropped.push({ key, expected: "", found: null });
1579
+ return { lines: [], dropped };
1580
+ }
1581
+ if (Object.keys(timing).length === 0) return { lines: [...lines], dropped };
1582
+
1583
+ // One `[start, end]` pair PER LINE — never a shared seam array (see the
1584
+ // docstring: the conflation moved untouched captions on a gapped stream).
1585
+ const starts = lines.map((l) => l.start);
1586
+ const ends = lines.map((l) => l.end);
1587
+
1588
+ const seen = new Set<string>();
1589
+ for (let i = 0; i < n; i++) {
1590
+ const line = lines[i]!;
1591
+ // No anchor, no nudge — the same boundary rule as `applyCaptionEdits`: a
1592
+ // pre-§137 word cannot be addressed, and the stored nudges then fall out
1593
+ // of the sweep below as `found: null`.
1594
+ const key = captionAnchorOf(line.words[0]);
1595
+ const entry = key === null ? undefined : timing[key];
1596
+ if (key === null || !entry) continue;
1597
+ // An earlier line already answered for this anchor — nudging here too
1598
+ // would fan one nudge onto a second line.
1599
+ if (seen.has(key)) {
1600
+ dropped.push({ key, expected: "", found: line.words[0]!.text, reason: "duplicate-anchor" });
1601
+ continue;
1602
+ }
1603
+ seen.add(key);
1604
+ // Deltas ride on the line's OWN edges, so a gapped stream moves the edge
1605
+ // the user dragged rather than the neighbour's. Tail first, then lead:
1606
+ // lines are visited in order, so line i+1's lead lands on a shared
1607
+ // boundary AFTER line i's tail — the documented "later lead wins".
1608
+ ends[i] = line.end + entry.tail;
1609
+ // COINCIDENCE, tested against the ORIGINAL edges: only a boundary the two
1610
+ // lines already SHARED travels with the nudge (the packed stream, where
1611
+ // every one of them is shared). Across a gap the neighbour stays where it
1612
+ // is — the sweep below still stops the moved edge from crossing it.
1613
+ // Assigning the same number, not recomputing it, keeps the two exactly
1614
+ // equal: a float `+ delta` computed twice can differ in the last bit, and
1615
+ // an unequal pair is an overlap the sweep would then have to fix.
1616
+ if (i + 1 < n && lines[i + 1]!.start === line.end) starts[i + 1] = ends[i]!;
1617
+ starts[i] = line.start + entry.lead;
1618
+ if (i > 0 && lines[i - 1]!.end === line.start) ends[i - 1] = starts[i]!;
1619
+ }
1620
+
1621
+ const lo = lines[0]!.start;
1622
+ const hi = lines[n - 1]!.end;
1623
+ // Forward: into the track's bounds, never opening before the previous line
1624
+ // CLOSED (its own edge, not a derived seam), never narrower than the floor.
1625
+ for (let i = 0; i < n; i++) {
1626
+ const floor = i === 0 ? lo : Math.max(lo, ends[i - 1]!);
1627
+ starts[i] = Math.min(Math.max(starts[i]!, floor), hi);
1628
+ ends[i] = Math.min(Math.max(ends[i]!, starts[i]! + MIN_CAPTION_SEC), hi);
1629
+ }
1630
+ // The forward pass caps at `hi`, so a track with less room than
1631
+ // `n * MIN_CAPTION_SEC` can leave the last lines piled on the end. Pull them
1632
+ // back (never before `lo`) so the edges stay ordered.
1633
+ for (let i = n - 1; i >= 0; i--) {
1634
+ const ceil = i === n - 1 ? hi : Math.min(hi, starts[i + 1]!);
1635
+ ends[i] = Math.max(Math.min(ends[i]!, ceil), lo);
1636
+ starts[i] = Math.max(Math.min(starts[i]!, ends[i]! - MIN_CAPTION_SEC), lo);
1637
+ }
1638
+
1639
+ const out = lines.map((line, i) => {
1640
+ const start = starts[i]!;
1641
+ const end = ends[i]!;
1642
+ // Neither edge moved: VERBATIM, same reference and same word stamps.
1643
+ if (start === line.start && end === line.end) return line;
1644
+ return {
1645
+ ...line,
1646
+ start,
1647
+ end,
1648
+ words: scaleWordsIntoWindow(line.words, line.start, line.end, start, end),
1649
+ };
1650
+ });
1651
+
1652
+ // An anchor no line starts on any more — a later cut removed the word the
1653
+ // line was keyed to, or a hide emptied the line. Silence here is the
1654
+ // field-case failure mode, so say it.
1655
+ for (const key of Object.keys(timing)) {
1656
+ if (!seen.has(key)) dropped.push({ key, expected: "", found: null });
1657
+ }
1658
+ return { lines: out, dropped };
1659
+ }
1660
+
1661
+ export interface AppliedCaptionLayers {
1662
+ lines: CaptionLine[];
1663
+ /** Every layer's drop reports, tagged with which layer refused them. */
1664
+ dropped: Array<
1665
+ AppliedCaptionEdits["dropped"][number] & { layer: "edit" | "range" | "hide" | "timing" }
1666
+ >;
1667
+ }
1668
+
1669
+ /**
1670
+ * The caption edit layers, composed in their ONE authoritative order — the
1671
+ * single chokepoint both the editor preview and produce consume, so the two
1672
+ * can never disagree about caption content.
1673
+ *
1674
+ * Per-word edits → RANGE edits → hides → LINE TIMING. Edits BEFORE hides is the
1675
+ * `was` contract: a hide's `was` records the LIVE text the user saw when they
1676
+ * deleted the word, which is the post-retype text — running hides first
1677
+ * would stale every hide sitting on a retyped word. Range edits sit between
1678
+ * the two, but the order barely earns the word: the reducer's creation-time
1679
+ * scrubbing (`useEdits`' `patchCaptionRange`) removes every per-word edit
1680
+ * and hide inside a new range's interval, so a LIVE range edit never
1681
+ * coexists with either inside its own words — the order only matters for
1682
+ * hand-edited docs, where the layers' own guards report rather than guess.
1683
+ * Timing runs LAST because it must see the surviving LINES: the hide layer
1684
+ * re-bases a line's window onto its surviving words and drops a line whose
1685
+ * words are all hidden, and a nudge on a line that no longer exists has no
1686
+ * window to move (`applyCaptionLineTiming`). Drop reports carry which layer
1687
+ * refused them, since "the retype missed", "the rewrite missed" and "the
1688
+ * delete missed" send the user to different gestures.
1689
+ */
1690
+ export function applyCaptionLayers(
1691
+ lines: readonly CaptionLine[],
1692
+ doc: OverrideDoc,
1693
+ ): AppliedCaptionLayers {
1694
+ const edited = applyCaptionEdits(lines, doc.captions);
1695
+ const ranged = applyCaptionRangeEdits(edited.lines, doc.captionRangeEdits);
1696
+ const hidden = applyCaptionWordHides(ranged.lines, doc.captionWordsHidden);
1697
+ const timed = applyCaptionLineTiming(hidden.lines, doc.captionLineTiming);
1698
+ return {
1699
+ lines: timed.lines,
1700
+ dropped: [
1701
+ ...edited.dropped.map((d) => ({ ...d, layer: "edit" as const })),
1702
+ ...ranged.dropped.map((d) => ({ ...d, layer: "range" as const })),
1703
+ ...hidden.dropped.map((d) => ({ ...d, layer: "hide" as const })),
1704
+ ...timed.dropped.map((d) => ({ ...d, layer: "timing" as const })),
1705
+ ],
1706
+ };
1707
+ }
1708
+
981
1709
  /** Theme tokens the user set, over whatever the production already had. */
982
1710
  export function resolveTheme(base: Theme, doc: OverrideDoc): Theme {
983
1711
  return ThemeSchema.parse({ ...base, ...doc.theme });
package/src/phonetics.ts CHANGED
@@ -12,6 +12,49 @@
12
12
  * "unrelated", and it must stay dependency-free and deterministic.
13
13
  */
14
14
 
15
+ /**
16
+ * Marks Arabic-script text carries that two transcribers disagree about
17
+ * without disagreeing about the WORD: harakat/vowel diacritics
18
+ * (U+064B–U+065F, U+0670), tatweel (U+0640, a pure typographic stretch), and
19
+ * the zero-width joiners (U+200C/U+200D). Whisper emits them inconsistently
20
+ * and an LLM writing a correction rarely reproduces them, so leaving them in
21
+ * makes a correct repair either miss `locate()` outright or read as
22
+ * "different from what was heard" purely on invisible marks. Only tatweel
23
+ * survives the `\p{L}\p{N}` filter below (it is Lm, a letter); the rest are
24
+ * stripped here so the intent is legible rather than an accident of Unicode
25
+ * categories.
26
+ */
27
+ // Escaped, not literal: three of these code points are invisible in an editor.
28
+ const ARABIC_NOISE = /[\u064B-\u065F\u0670\u0640\u200C\u200D]/g;
29
+
30
+ /**
31
+ * Comparable form for two pieces of text: case-, punctuation- and
32
+ * whitespace-insensitive, in ANY script.
33
+ *
34
+ * Shared by `phonetics.ts` and `producer/repair.ts` on purpose — they used to
35
+ * hold two copies and the copy in `repair.ts` was `[^a-z0-9\s]`, i.e.
36
+ * Latin-only. Field case (2026-08-18): every one of 11 recorded Urdu repairs
37
+ * normalized to the empty string, so `norm(heard) === norm(correction)` was
38
+ * `"" === ""` and ALL 11 were refused as "identical to what was heard" —
39
+ * including `پرسٹ` → `فرسٹ`, which shares no letters with what it replaced.
40
+ *
41
+ * Keeping letters and digits of every script also KEEPS accented Latin
42
+ * ("café", "über") where the old expression deleted it. That is the same bug
43
+ * in miniature — a French word normalized to "caf" — so it is a fix, not a
44
+ * regression. Pure-ASCII input is byte-identical to the old behaviour, which
45
+ * is pinned by a test.
46
+ */
47
+ export function normalizeForCompare(s: string): string {
48
+ return s
49
+ .normalize("NFC")
50
+ .toLowerCase()
51
+ .replace(ARABIC_NOISE, "")
52
+ .replace(/[^\p{L}\p{N}\s]/gu, "")
53
+ .split(/\s+/)
54
+ .filter(Boolean)
55
+ .join(" ");
56
+ }
57
+
15
58
  /**
16
59
  * Digraphs collapsed before single letters, longest first. The ch/sh and th
17
60
  * sounds get DIGIT placeholders on purpose: a letter placeholder would be
@@ -106,6 +149,47 @@ export function soundsLike(a: string, b: string): number {
106
149
  return Math.max(0, 1 - dist / Math.max(ka.length, kb.length));
107
150
  }
108
151
 
152
+ /**
153
+ * 0..1 similarity of the TEXT itself, for scripts the phonetic key cannot
154
+ * represent (`phoneticKey` is defined over a-z, so anything non-Latin keys to
155
+ * ""). Same shape as `soundsLike` — normalized edit distance over the longer
156
+ * string — but run on the normalized text rather than a consonant skeleton.
157
+ *
158
+ * This is a weaker signal than a phonetic key and it is meant to be: an
159
+ * Urdu-script mishearing differs from the truth by a letter or two of the same
160
+ * script, so edit distance still separates it from an unrelated phrase. What
161
+ * it cannot do is fold vowels, which is why the floor below is calibrated
162
+ * against real data instead of borrowing SOUNDS_LIKE_FLOOR.
163
+ */
164
+ export function textSimilarity(a: string, b: string): number {
165
+ const na = normalizeForCompare(a);
166
+ const nb = normalizeForCompare(b);
167
+ if (na.length === 0 && nb.length === 0) return 1;
168
+ if (na.length === 0 || nb.length === 0) return 0;
169
+ return Math.max(0, 1 - levenshtein(na, nb) / Math.max(na.length, nb.length));
170
+ }
171
+
172
+ /**
173
+ * Floor for the non-Latin fallback, MEASURED rather than guessed.
174
+ *
175
+ * The 11 Urdu repairs recorded in a real production.json (2026-08-18) score,
176
+ * sorted: 0.333, 0.400, 0.500, 0.500, 0.545, 0.583, 0.636, 0.750, 0.750,
177
+ * 0.800, 0.800. The 0.333 is `حقیقہ ٹون` → `ہیکاتھون` ("hackathon"), a genuine
178
+ * repair and the worst of the set because the recognizer both re-segmented the
179
+ * word and changed its opening letter. Admitting it sets the ceiling on the
180
+ * floor; 0.33 is the largest value that does.
181
+ *
182
+ * Against that, unrelated four-word spans lifted from the same transcript
183
+ * score 0.167–0.250 and are refused. The band is narrow, and it is narrow for
184
+ * the same reason the Latin one is (see SOUNDS_LIKE_FLOOR): two SHORT
185
+ * unrelated Urdu spans can still land above it — measured, `پرسٹ ہیک` vs
186
+ * `ٹرس می` scores 0.500. There is no onset test here to catch that, because
187
+ * two of the 11 genuine repairs change their first letter. So this gate is
188
+ * real but shallow; the span, token-count and length guards in
189
+ * `applyRepairs` are what keep it from being a rewrite licence.
190
+ */
191
+ export const TEXT_SIMILARITY_FLOOR = 0.33;
192
+
109
193
  /**
110
194
  * Default floor for "this is a repair, not a rewrite". Deliberately low,
111
195
  * because a real mishearing can move word boundaries ("code churn" → "coach
@@ -124,11 +208,27 @@ export const SOUNDS_LIKE_FLOOR = 0.34;
124
208
  * a phrase, essentially never its onset. This is what rejects a rewrite:
125
209
  * "revenue" for "churn" and "monetization" for "agents" both score in the
126
210
  * same range as a true repair, and both fail the onset test.
211
+ *
212
+ * When either side has no Latin letters there is no key to compare, and this
213
+ * used to answer `ka === kb` — `"" === ""`, i.e. YES for any two non-Latin
214
+ * strings however unrelated. That is no gate at all for an Urdu transcript, so
215
+ * those pairs route to `textSimilarity` instead (2026-08-18 field case).
216
+ * Latin-to-Latin comparisons never reach that branch and are unchanged.
127
217
  */
128
218
  export function soundsSimilar(a: string, b: string, floor = SOUNDS_LIKE_FLOOR): boolean {
129
219
  const ka = phraseKey(a);
130
220
  const kb = phraseKey(b);
131
- if (ka.length === 0 || kb.length === 0) return ka === kb;
221
+ if (ka.length === 0 || kb.length === 0) {
222
+ // `floor` is deliberately NOT reused here: it is calibrated against
223
+ // consonant skeletons, which are shorter and coarser than the text this
224
+ // branch compares, so the same number means something else. Taking the
225
+ // larger of the two was tried and is wrong — the default 0.34 alone
226
+ // rejects a measured genuine repair scoring 0.333. The only caller that
227
+ // raises the floor (reconcileCopy, 0.6) cannot reach this branch anyway:
228
+ // its candidate tokens are stripped to `[A-Za-z]`, so its key is never
229
+ // empty and a non-Latin spoken word scores ~0 against it regardless.
230
+ return textSimilarity(a, b) >= TEXT_SIMILARITY_FLOOR;
231
+ }
132
232
  if (ka[0] !== kb[0]) return false;
133
233
  return soundsLike(a, b) >= floor;
134
234
  }
@@ -1,7 +1,7 @@
1
1
  import { z } from "zod/v4";
2
2
  import type { Transcript, Word } from "../schema";
3
3
  import type { Scene } from "../scene-schema";
4
- import { soundsSimilar } from "../phonetics";
4
+ import { normalizeForCompare, soundsSimilar } from "../phonetics";
5
5
  import type { LlmProvider } from "./provider";
6
6
 
7
7
  /**
@@ -105,15 +105,18 @@ export function buildRepairUserPrompt(
105
105
  );
106
106
  }
107
107
 
108
- /** Comparable form: case- and punctuation-insensitive. */
109
- function norm(s: string): string {
110
- return s
111
- .toLowerCase()
112
- .replace(/[^a-z0-9\s]/g, "")
113
- .split(/\s+/)
114
- .filter(Boolean)
115
- .join(" ");
116
- }
108
+ /**
109
+ * Comparable form: case- and punctuation-insensitive, in any script.
110
+ *
111
+ * This was a second, Latin-only copy (`[^a-z0-9\s]`) of what is now
112
+ * `normalizeForCompare`. The two drifted in the worst possible way: on an Urdu
113
+ * transcript every string normalized to "", so `norm(actual) ===
114
+ * norm(r.correction)` was true for every proposal and all 11 repairs in a real
115
+ * run were refused as "identical to what was heard" (2026-08-18) — while
116
+ * `locate()`, comparing "" to "", "matched" the first span it tried without
117
+ * verifying anything. One shared helper so they cannot drift again.
118
+ */
119
+ const norm = normalizeForCompare;
117
120
 
118
121
  function spanText(transcript: Transcript, startWord: number, endWord: number): string {
119
122
  return transcript.words
@@ -228,6 +231,12 @@ export function applyRepairs(
228
231
  */
229
232
  const locate = (r: TranscriptRepair): { startWord: number; endWord: number } | null => {
230
233
  const want = norm(r.heard);
234
+ // A quote that normalizes to nothing is not an anchor: "" compares equal
235
+ // to the first span whose own normalisation is empty, so the search would
236
+ // "find" a span it never verified. That was live for every non-Latin
237
+ // transcript until norm() was fixed above; refuse it explicitly so it
238
+ // cannot come back through some other all-punctuation quote.
239
+ if (want.length === 0) return null;
231
240
  // Widths to try, in order of trust. The QUOTED TEXT is the reliable part
232
241
  // of a proposal, so its own token count leads; the claimed span is a
233
242
  // fallback for a quote whose normalisation splits differently. Trusting
package/src/recut.ts CHANGED
@@ -391,3 +391,64 @@ export function applyUserCuts(
391
391
  removedSec,
392
392
  };
393
393
  }
394
+
395
+ /** What `pruneHidesInsideCuts` hands back: the doc (same reference when
396
+ * nothing was pruned — the caller's changed-gate reads `pruned.length`), and
397
+ * the retired keys so produce can SAY what it retired. */
398
+ export interface PrunedHides {
399
+ doc: OverrideDoc;
400
+ pruned: string[];
401
+ }
402
+
403
+ /**
404
+ * Retire `captionWordsHidden` entries whose word the final cutlist REMOVES
405
+ * (§59b revisited 2026-08-18 — the "captions + video" delete gesture writes
406
+ * both a hide and a cut in one commit).
407
+ *
408
+ * Once the cut lands, `buildCaptionLines` drops the word before the hide
409
+ * layer ever sees it, so the hide key would report `found: null` ("the cut
410
+ * removed it", `captionHideDropLine`) on every subsequent run forever — the
411
+ * cut SUPERSEDES the hide, the same superseded philosophy `overrides.ts`'s
412
+ * caption-key migration applies. Hides whose source instant is OUTSIDE every
413
+ * removed segment are kept verbatim — as are keys that are not §137 `w<ms>`
414
+ * anchors at all, which name no instant this can test (see the guard below).
415
+ *
416
+ * HALF-OPEN interval (`srcIn <= src < srcOut`), on purpose — the two edges
417
+ * are NOT symmetric. A word starting exactly at `srcIn` IS cut: that is
418
+ * precisely where the FIRST word of a captions+video delete lands (its
419
+ * srcStart round-trips through the TimeMap to the resolved cut's own srcIn),
420
+ * and `mapWord` clamps that instant into the removal and drops the word — a
421
+ * strictly-inside test never retired the gesture's own first hide, leaving
422
+ * it a permanent `found: null` drop report. A word starting exactly at
423
+ * `srcOut` belongs to the NEXT kept span (`buildCaptionLines` still emits it
424
+ * — a seam instant has a kept-side preimage, `timemap.ts`), so its hide is
425
+ * still doing work and must survive.
426
+ */
427
+ export function pruneHidesInsideCuts(doc: OverrideDoc, cutlist: readonly Segment[]): PrunedHides {
428
+ const pruned: string[] = [];
429
+ const kept: OverrideDoc["captionWordsHidden"] = {};
430
+ for (const [key, entry] of Object.entries(doc.captionWordsHidden)) {
431
+ // SOURCE-KEYED ONLY, parsed and not coerced. `captionWordsHidden` is an
432
+ // unpinned `z.record` (unlike `CaptionRangeEditSchema`'s `/^w\d+$/`), so a
433
+ // hand-edited or legacy-keyed doc reaches here: a POSITIONAL key like "17"
434
+ // would slice to "7", parse as 7ms, land inside any early cut and be
435
+ // deleted with nothing said. The editor guards the identical case and
436
+ // states the rule (`apps/editor/src/useEdits.ts:587-600`): only §137
437
+ // `w<ms>` keys carry an interval-testable instant, and an entry this
438
+ // function cannot honestly locate is KEPT.
439
+ if (!/^w\d+$/.test(key)) {
440
+ kept[key] = entry;
441
+ continue;
442
+ }
443
+ // `captionKeyFor`'s quantization inverted (`w${Math.round(sec * 1000)}`,
444
+ // overrides.ts): the key IS the word's source instant, ms-quantized.
445
+ const srcSec = parseInt(key.slice(1), 10) / 1000;
446
+ const removed = cutlist.some(
447
+ (seg) => seg.kind === "remove" && srcSec >= seg.srcIn && srcSec < seg.srcOut,
448
+ );
449
+ if (removed) pruned.push(key);
450
+ else kept[key] = entry;
451
+ }
452
+ if (pruned.length === 0) return { doc, pruned };
453
+ return { doc: { ...doc, captionWordsHidden: kept }, pruned };
454
+ }
package/src/transcribe.ts CHANGED
@@ -78,6 +78,51 @@ function repairSplitSegments(json: WhisperJson): WhisperJson {
78
78
  };
79
79
  }
80
80
 
81
+ /**
82
+ * Run length at which a stack of zero-length words at ONE instant stops being
83
+ * a rounding artifact and becomes a repetition-loop hallucination. Real speech
84
+ * never emits 8 tokens at a single instant; the field case emitted 118.
85
+ */
86
+ export const REPETITION_BURST_MIN = 8;
87
+
88
+ /**
89
+ * Drop whisper repetition-loop bursts (field case 2026-08-18): an Urdu take
90
+ * re-decoded a whole phrase as 118 CONSECUTIVE tokens all stamped
91
+ * `from === to === 31040` — zero length, at one instant. The stamp repair
92
+ * below then fans such a burst out into 118 fabricated 50ms words marching
93
+ * forward from 31.04s, so the phrase ships TWICE in the captions (31.04s and
94
+ * 33.54s) and a fifth of the transcript carries the tell-tale exactly-0.05s
95
+ * duration. `-mc 0` in whisperArgs is the decoder-side mitigation for the same
96
+ * failure; it did not prevent this occurrence, and it can never repair an
97
+ * already-cached transcript.json — hence a parse-side guard too.
98
+ *
99
+ * A burst is a MAXIMAL run of consecutive zero-length/inverted words sharing
100
+ * one `start`. Equality is exact, not epsilon: these stamps are integer
101
+ * milliseconds divided by 1000, so members of one burst are the same double
102
+ * bit-for-bit, and a tolerance would only start swallowing real neighbors.
103
+ * Runs shorter than REPETITION_BURST_MIN fall through untouched — a lone
104
+ * zero-length stamp is a rounding artifact, not a hallucination. The drop is
105
+ * silent by design: this function is pure and total, and there is no logging
106
+ * channel in the parse path to warn on.
107
+ */
108
+ export function dropRepetitionBursts(words: readonly Word[]): Word[] {
109
+ const out: Word[] = [];
110
+ let i = 0;
111
+ while (i < words.length) {
112
+ const w = words[i]!;
113
+ if (w.end > w.start) {
114
+ out.push(w);
115
+ i++;
116
+ continue;
117
+ }
118
+ let j = i + 1;
119
+ while (j < words.length && words[j]!.end <= words[j]!.start && words[j]!.start === w.start) j++;
120
+ if (j - i < REPETITION_BURST_MIN) for (let k = i; k < j; k++) out.push(words[k]!);
121
+ i = j;
122
+ }
123
+ return out;
124
+ }
125
+
81
126
  const STRICT_UTF8 = new TextDecoder("utf-8", { fatal: true });
82
127
 
83
128
  /**
@@ -115,11 +160,19 @@ export function parseWhisperJson(json: WhisperJson): Transcript {
115
160
  if (!raw || !raw.trim()) continue;
116
161
  const text = raw.trim();
117
162
  if (NOISE_TOKEN.test(text)) continue;
163
+ // A word already CLOSED by Arabic-script sentence punctuation refuses
164
+ // continuations (field case 2026-08-18): whisper emits `۔` and the next
165
+ // sentence's first token with no leading whitespace, and the plain
166
+ // whitespace rule fused them into one unsplittable word ("ہوں۔اس").
167
+ // Deliberately NOT the Latin `.`/`!`/`?` — whisper tokenizes decimals
168
+ // ("3", ".", "5") and abbreviations as bare continuations too, and
169
+ // splitting those would shred "3.5" into two words. ۔ (U+06D4) and
170
+ // ؟ (U+061F) have no such second job.
118
171
  const startsWord = /^\s/.test(raw) || words.length === 0;
119
172
  const start = seg.offsets.from / 1000;
120
173
  const end = seg.offsets.to / 1000;
121
174
  const last = words[words.length - 1];
122
- if (!startsWord && last) {
175
+ if (!startsWord && last && !/[۔؟]$/.test(last.text)) {
123
176
  last.text += text;
124
177
  last.end = Math.max(last.end, end);
125
178
  } else {
@@ -146,14 +199,18 @@ export function parseWhisperJson(json: WhisperJson): Transcript {
146
199
  else if (next) next.start = Math.min(next.start, w.start);
147
200
  words.splice(i, 1);
148
201
  }
202
+ // BEFORE the repair loop, never after: the repair rewrites every burst
203
+ // member into a distinct monotone stamp, so once it has run the shared
204
+ // timestamp — the only evidence a burst existed — is gone.
205
+ const kept = dropRepetitionBursts(words);
149
206
  // Whisper occasionally emits zero-length or inverted stamps; repair minimally.
150
- for (let i = 0; i < words.length; i++) {
151
- const w = words[i]!;
207
+ for (let i = 0; i < kept.length; i++) {
208
+ const w = kept[i]!;
152
209
  if (w.end <= w.start) w.end = w.start + 0.05;
153
- const next = words[i + 1];
210
+ const next = kept[i + 1];
154
211
  if (next && next.start < w.end) next.start = w.end;
155
212
  }
156
- return { language: json.result?.language ?? "en", words };
213
+ return { language: json.result?.language ?? "en", words: kept };
157
214
  }
158
215
 
159
216
  export interface WhisperOptions {
@@ -192,6 +249,15 @@ export function whisperArgs(opts: WhisperOptions, wavPath: string): string[] {
192
249
  "-oj",
193
250
  "-of", opts.outBase,
194
251
  "-ml", "1",
252
+ // No text context across 30s decode windows (field case 2026-08-18): an
253
+ // Urdu take hit whisper's repetition loop — a whole sentence re-decoded
254
+ // as 261 zero-duration tokens — and carrying the previous window's text
255
+ // into the decoder is the known trigger. `-mc 0` is the standard
256
+ // mitigation and leaves `--prompt` (the dictionary bias) untouched.
257
+ // Cached transcript.json files decoded without it are knowingly still
258
+ // reused (transcriptCacheReusable's no-spurious-retranscribe rule);
259
+ // delete a workdir's transcript.json to re-decode with it.
260
+ "-mc", "0",
195
261
  "--no-prints",
196
262
  ];
197
263
  if (opts.language !== undefined) args.push("-l", opts.language);