@ossclip/core 0.1.25 → 0.1.27

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/overrides.ts CHANGED
@@ -152,6 +152,37 @@ export const CaptionEditSchema = z.object({
152
152
  });
153
153
  export type CaptionEdit = z.infer<typeof CaptionEditSchema>;
154
154
 
155
+ /**
156
+ * A free-text rewrite of a contiguous caption word RUN (2026-08-18) — the one
157
+ * deliberate relaxation of the 1:1 retype contract, for range edits only.
158
+ * Single-word retype (`CaptionEditSchema` above) is untouched, and
159
+ * `transcript.words` is NEVER spliced — scene anchors are raw indices into it
160
+ * — so everything happens on the derived `CaptionLine[]`
161
+ * (`applyCaptionRangeEdits` below).
162
+ *
163
+ * Endpoints are anchored by §137 source-time keys (`captionKeyFor`), so a
164
+ * user cut elsewhere cannot shift the run. `was` is the NFC-normalized,
165
+ * space-joined BASE text of the run — the `captionEditWas` base-truth rule,
166
+ * run-wide: the reducer scrubs every per-word retype inside the interval in
167
+ * the same commit that stores the entry, so the run `applyCaptionRangeEdits`
168
+ * reads at apply time IS the base run, and a live (post-retype) join would
169
+ * fail the guard forever. A WHOLE-RUN stale guard: if any word in the run is
170
+ * re-worded or cut later, the entire edit is reported dropped, never
171
+ * partially guessed at. Identity is the `(fromKey, toKey)`
172
+ * pair — retyping the run back to its `was` DELETES the entry (the
173
+ * clearVideo/`patchCaption` rule). An array like `cuts`, `.default([])` so
174
+ * every pre-existing overrides.json parses byte-identically. NEVER
175
+ * legacy-keyed: the field postdates §137, so `migrateCaptionKeys` must not
176
+ * process it — there are no positional range edits to upgrade.
177
+ */
178
+ export const CaptionRangeEditSchema = z.object({
179
+ fromKey: z.string().regex(/^w\d+$/),
180
+ toKey: z.string().regex(/^w\d+$/),
181
+ text: z.string().min(1).max(400),
182
+ was: z.string(),
183
+ });
184
+ export type CaptionRangeEdit = z.infer<typeof CaptionRangeEditSchema>;
185
+
155
186
  /**
156
187
  * The `was` a caption edit should store (R15 §59). The FIRST edit's `was` is
157
188
  * the base truth (the word as transcribed); every later re-edit of the same
@@ -169,6 +200,25 @@ export function captionEditWas(
169
200
  return captions[key]?.was ?? seen;
170
201
  }
171
202
 
203
+ /**
204
+ * The `was` a RANGE edit should store — `captionEditWas` for the
205
+ * `(fromKey, toKey)` pair. The first edit's `was` is the base truth; a
206
+ * re-edit of the SAME run (its endpoints are re-minted verbatim, see
207
+ * `applyCaptionRangeEdits`' srcStart minting) sees the LIVE, already-rewritten
208
+ * text, and storing that as `was` would stale the guard against the base
209
+ * lines the next apply runs on. Preserving the existing pair's `was` keeps
210
+ * the guard anchored to the base — and makes "retyped back to the original"
211
+ * detectable, which is when the entry should clear entirely.
212
+ */
213
+ export function captionRangeEditWas(
214
+ rangeEdits: readonly CaptionRangeEdit[],
215
+ fromKey: string,
216
+ toKey: string,
217
+ seen: string,
218
+ ): string {
219
+ return rangeEdits.find((e) => e.fromKey === fromKey && e.toKey === toKey)?.was ?? seen;
220
+ }
221
+
172
222
  /**
173
223
  * The id a pre-§137 split gets when it is upgraded: the output milliseconds of
174
224
  * whatever `at` the file holds NOW.
@@ -256,6 +306,76 @@ export const OverrideDocSchema = z.object({
256
306
  scenes: z.record(z.string(), SceneOverrideSchema).default({}),
257
307
  /** Retyped caption words, keyed by the word's source time (§137). */
258
308
  captions: z.record(z.string(), CaptionEditSchema).default({}),
309
+ /**
310
+ * Per-word caption HIDES ("delete word from captions") — non-destructive:
311
+ * the word stays in the transcript and in the video's audio; only the
312
+ * rendered caption drops it. Keyed by the word's source time
313
+ * (`captionKeyFor`, §137) like `captions` above, so a user cut never
314
+ * shifts a hide onto a different word. `was` is the LIVE (post-retype)
315
+ * text at hide time — hides apply AFTER retypes (`applyCaptionLayers`
316
+ * below) — the same stale-guard contract as `CaptionEditSchema.was`:
317
+ * a re-derived stream under a surviving anchor drops the hide WITH A
318
+ * REPORT rather than deleting the wrong word. Restore DELETES the key
319
+ * (the restoreScene/captionsHidden rule — an entry with nothing to say is
320
+ * still an override), and `.default({})` keeps every pre-existing
321
+ * overrides.json parsing byte-identically. This field NEVER existed in
322
+ * the legacy positional-key era, so `migrateCaptionKeys` must NOT process
323
+ * it — there are no legacy hides to upgrade.
324
+ */
325
+ captionWordsHidden: z.record(z.string(), z.object({ was: z.string() })).default({}),
326
+ /**
327
+ * Multi-word free-text rewrites — see `CaptionRangeEditSchema` for the
328
+ * whole contract (endpoint anchoring, the whole-run `was` guard, identity
329
+ * by pair, why it is never legacy-keyed). Applied between per-word retypes
330
+ * and hides (`applyCaptionLayers`).
331
+ */
332
+ captionRangeEdits: z.array(CaptionRangeEditSchema).default([]),
333
+ /**
334
+ * Per-LINE caption TIMING nudges — "when does this caption appear, and when
335
+ * does it leave". Stored as DELTAS against the DERIVED window (`lead` moves
336
+ * the line's OPENING seam, `tail` its CLOSING seam), keyed by the LINE's
337
+ * FIRST WORD's SOURCE time (`captionKeyFor`, §137). Deltas over source keys
338
+ * make the record recut-immune for free: a recut rebuilds every derived
339
+ * `start`/`end` through the new TimeMap and the deltas simply re-apply on
340
+ * top — zero work in `remapOverridesThroughRecut`, the same property every
341
+ * other caption record leans on (captions.ts:14-20: `srcStart` is the one
342
+ * field a re-cut cannot move). Restore DELETES the key, and a patch whose
343
+ * deltas are both under 1ms in magnitude also deletes (the clearVideo/
344
+ * patchCaption clear-override rule — a nudge of nothing is still an
345
+ * override). `.default({})` keeps every pre-existing overrides.json parsing
346
+ * byte-identically, and the field NEVER existed in the legacy
347
+ * positional-key era, so `migrateCaptionKeys` must not process it.
348
+ *
349
+ * PER LINE, NOT PER WORD, and that is the whole point of the field. It
350
+ * replaces `captionWordTiming` (deleted 2026-08-18), which stored the same
351
+ * shape against individual WORDS and was measured to be MATHEMATICALLY
352
+ * INERT: on a live workdir (117 lines / 301 words) 116/116 inter-line gaps
353
+ * were exactly 0.0, 184/184 intra-line word boundaries exactly 0.0,
354
+ * `line.start === words[0].start` 117/117 and `line.end === lastWord.end`
355
+ * 117/117 — `transcribe.ts` chains words (`next.start = w.end`) and
356
+ * `captions.ts:203-213`'s hold pass clamps each line's end to the next
357
+ * line's start, so the caption stream is a GAP-FREE PARTITION. A per-word
358
+ * clamp of `[max(lineStart, prevEnd), min(lineEnd, nextStart)]` therefore
359
+ * collapsed to exactly `[w.start, w.end]` for EVERY word: the user dragged,
360
+ * every stored delta came back zero, and the reducer's sub-ms rule deleted
361
+ * them again. Do not reintroduce word-level clamping against a packed
362
+ * stream. Word stamps also only drive the karaoke highlight INSIDE a line's
363
+ * `<Sequence>` window (CaptionTrack.tsx:228-229, 387) — "when a caption
364
+ * appears" IS `line.start`/`line.end`, so timing has to move LINE windows.
365
+ * The ±30s range is per SEAM, which is why it is wider than the old
366
+ * per-word ±10s: a line may be dragged well clear of its neighbours, and
367
+ * `applyCaptionLineTiming`'s sweep — not the schema — is what keeps seams
368
+ * ordered and inside the track.
369
+ */
370
+ captionLineTiming: z
371
+ .record(
372
+ z.string(),
373
+ z.object({
374
+ lead: z.number().min(-30).max(30),
375
+ tail: z.number().min(-30).max(30),
376
+ }),
377
+ )
378
+ .default({}),
259
379
  /**
260
380
  * Scene split points. `at` is ABSOLUTE output seconds (R16 §61 — Cmd/Ctrl+B
261
381
  * at the playhead) and moves when a re-cut re-anchors the doc; `id` is
@@ -978,6 +1098,614 @@ export function applyCaptionEdits(
978
1098
  return { lines: out, dropped };
979
1099
  }
980
1100
 
1101
+ /**
1102
+ * Re-time replacement tokens over ONE line's stretch of a rewritten run —
1103
+ * `repair.ts`'s `retime` model (producer/repair.ts:137-154), restated here
1104
+ * for CaptionWords: stamps distributed across the window weighted by token
1105
+ * length + 1, strictly increasing, the last token's `end` pinned to the
1106
+ * window end so the run never leaks past the span it replaced. The measured
1107
+ * window edges (first run word's start, last run word's end) are kept;
1108
+ * only the interior boundaries are interpolated — interpolated boundaries
1109
+ * are a guess, and `retime`'s comment is explicit that a guess must never
1110
+ * displace a measurement, which is why the equal-count fast path in
1111
+ * `applyCaptionRangeEdits` below bypasses this entirely.
1112
+ */
1113
+ function retimeCaptionTokens(
1114
+ tokens: readonly string[],
1115
+ windowStart: number,
1116
+ windowEnd: number,
1117
+ srcStarts: readonly number[],
1118
+ ): CaptionWord[] {
1119
+ const weights = tokens.map((t) => t.length + 1);
1120
+ const total = weights.reduce((a, b) => a + b, 0);
1121
+ const out: CaptionWord[] = [];
1122
+ let cursor = windowStart;
1123
+ for (let i = 0; i < tokens.length; i++) {
1124
+ const share = ((windowEnd - windowStart) * weights[i]!) / total;
1125
+ const end = i === tokens.length - 1 ? windowEnd : cursor + share;
1126
+ out.push({ text: tokens[i]!, start: cursor, end, srcStart: srcStarts[i]! });
1127
+ cursor = end;
1128
+ }
1129
+ return out;
1130
+ }
1131
+
1132
+ /**
1133
+ * Apply the free-text RANGE rewrites (`captionRangeEdits`) — the one layer
1134
+ * allowed to change word COUNT, which is why it exists at all: everything it
1135
+ * reshapes is the derived `CaptionLine[]`, never `transcript.words` (scene
1136
+ * anchors are raw indices into that array — splicing it is the forbidden
1137
+ * operation this whole edit family is built around).
1138
+ *
1139
+ * Same reporting shape as `applyCaptionEdits`; drop `key`s are the COMPOSITE
1140
+ * `${fromKey}..${toKey}` — the pair is the entry's identity, and either half
1141
+ * alone names only an endpoint. Each entry drops AT MOST ONCE (unlike the
1142
+ * per-word layers, where one key can be reported per extra claimant), which
1143
+ * is what lets `reconcileCaptionEdits` count applied entries by subtraction.
1144
+ *
1145
+ * Locating: `fromKey`'s first claimant across the flat word order (the
1146
+ * per-word first-claimant rule — ms-quantised keys CAN collide,
1147
+ * captions.ts:44-50), then a FORWARD walk to `toKey`; a missing endpoint, or
1148
+ * a `toKey` that only occurs before `fromKey`, is `found: null`. An entry
1149
+ * whose pair was already applied, or whose `fromKey` an earlier range edit's
1150
+ * run consumed, is `duplicate-anchor` — reachable only in a hand-edited doc,
1151
+ * since the reducer scrubs overlapping entries at creation, and reported
1152
+ * rather than guessed at like every other collision in this file.
1153
+ *
1154
+ * The whole-run stale guard: the run's live texts, NFC-normalized and
1155
+ * space-joined, must equal `was` byte for byte, or the WHOLE edit drops with
1156
+ * the joined text as `found` — never a partial rewrite of the words that
1157
+ * still match (a half-applied rewrite reads as garbage, and there is no
1158
+ * per-word truth to fall back on once the counts differ).
1159
+ *
1160
+ * Retiming across lines: the run may span several lines, and their `start`/
1161
+ * `end` WINDOWS are deliberately not re-packed — Sequence windows and
1162
+ * `buildCaptionLines`' breakpoint semantics stay exactly as produced.
1163
+ * Replacement tokens are distributed across the affected lines
1164
+ * proportionally to each line's share of the run's summed word duration,
1165
+ * rounded by largest remainder (deterministic — earlier line wins a tie) so
1166
+ * every token lands somewhere and the totals match. Within a line the stamps
1167
+ * follow `retimeCaptionTokens` above; a token count equal to the run's word
1168
+ * count skips all of it and keeps the measured per-word stamps AND srcStarts
1169
+ * verbatim (measured ASR boundaries beat interpolation — `retime`'s rule).
1170
+ * A line allotted zero tokens loses its run words, and if that empties it
1171
+ * the line is omitted (the `applyCaptionWordHides` rule — no zero-word
1172
+ * Sequence).
1173
+ *
1174
+ * srcStart minting for count-changed runs: linear across `[fromSrc, toSrc]`
1175
+ * (the endpoints' own source starts), endpoints re-minted verbatim — which
1176
+ * is what lets the user select a rewritten run again and edit it (its
1177
+ * endpoints still answer to the same pair). Strictly increasing whenever the
1178
+ * span is non-degenerate; when the span is too short for 1ms-distinct
1179
+ * quantised keys (`captionKeyFor` rounds to ms), later words SHARE quantised
1180
+ * keys — an accepted, documented duplicate-anchor case the existing
1181
+ * machinery reports if a per-word edit ever targets one.
1182
+ */
1183
+ export function applyCaptionRangeEdits(
1184
+ lines: readonly CaptionLine[],
1185
+ rangeEdits: readonly CaptionRangeEdit[],
1186
+ ): AppliedCaptionEdits {
1187
+ const dropped: AppliedCaptionEdits["dropped"] = [];
1188
+ if (rangeEdits.length === 0) return { lines: [...lines], dropped };
1189
+
1190
+ let out: CaptionLine[] = [...lines];
1191
+ const seenPairs = new Set<string>();
1192
+ const consumed = new Set<string>();
1193
+
1194
+ for (const entry of rangeEdits) {
1195
+ const key = `${entry.fromKey}..${entry.toKey}`;
1196
+ // Flatten the CURRENT lines — edits apply sequentially, so a later entry
1197
+ // addresses the stream as the earlier ones left it (that is how a
1198
+ // re-minted endpoint stays addressable at all).
1199
+ const flat: Array<{ line: number; word: number; w: CaptionWord }> = [];
1200
+ for (let li = 0; li < out.length; li++) {
1201
+ for (let wi = 0; wi < out[li]!.words.length; wi++) {
1202
+ flat.push({ line: li, word: wi, w: out[li]!.words[wi]! });
1203
+ }
1204
+ }
1205
+ const fromIdx = flat.findIndex((f) => captionAnchorOf(f.w) === entry.fromKey);
1206
+ if (seenPairs.has(key) || consumed.has(entry.fromKey)) {
1207
+ dropped.push({
1208
+ key,
1209
+ expected: entry.was,
1210
+ found: fromIdx === -1 ? null : flat[fromIdx]!.w.text,
1211
+ reason: "duplicate-anchor",
1212
+ });
1213
+ continue;
1214
+ }
1215
+ if (fromIdx === -1) {
1216
+ dropped.push({ key, expected: entry.was, found: null });
1217
+ continue;
1218
+ }
1219
+ // FORWARD only: a toKey sitting before fromKey is a run that crosses a
1220
+ // gap the stream no longer bridges — `found: null`, never a guess.
1221
+ let toIdx = -1;
1222
+ for (let i = fromIdx; i < flat.length; i++) {
1223
+ if (captionAnchorOf(flat[i]!.w) === entry.toKey) {
1224
+ toIdx = i;
1225
+ break;
1226
+ }
1227
+ }
1228
+ if (toIdx === -1) {
1229
+ dropped.push({ key, expected: entry.was, found: null });
1230
+ continue;
1231
+ }
1232
+ const run = flat.slice(fromIdx, toIdx + 1);
1233
+ const joined = run
1234
+ .map((f) => f.w.text)
1235
+ .join(" ")
1236
+ .normalize("NFC");
1237
+ if (joined !== entry.was.normalize("NFC")) {
1238
+ dropped.push({ key, expected: entry.was, found: joined });
1239
+ continue;
1240
+ }
1241
+ const tokens = entry.text.trim().split(/\s+/).filter(Boolean);
1242
+ if (tokens.length === 0) {
1243
+ // Defensive: zod's min(1) admits a whitespace-only string, and a run
1244
+ // rewritten to NOTHING is a delete, which is the hide layer's job —
1245
+ // treated as a stale-style drop rather than silently emptying the run.
1246
+ dropped.push({ key, expected: entry.was, found: joined });
1247
+ continue;
1248
+ }
1249
+ seenPairs.add(key);
1250
+ for (const f of run) {
1251
+ const a = captionAnchorOf(f.w);
1252
+ if (a !== null) consumed.add(a);
1253
+ }
1254
+
1255
+ if (tokens.length === run.length) {
1256
+ // Equal count: keep the measured stamps AND srcStarts verbatim —
1257
+ // `retime`'s fast path, for its reason (measured ASR onsets beat any
1258
+ // interpolation, and verbatim srcStarts keep every anchor addressable).
1259
+ const replaced = new Map(run.map((f, i) => [`${f.line}:${f.word}`, tokens[i]!]));
1260
+ out = out.map((line, li) => ({
1261
+ ...line,
1262
+ words: line.words.map((w, wi) => {
1263
+ const text = replaced.get(`${li}:${wi}`);
1264
+ return text === undefined ? w : { ...w, text };
1265
+ }),
1266
+ }));
1267
+ continue;
1268
+ }
1269
+
1270
+ // Count changed: distribute tokens across the affected lines by each
1271
+ // line's share of the run's total duration, largest-remainder rounded.
1272
+ const lineOrder: number[] = [];
1273
+ const runByLine = new Map<number, { first: number; last: number; words: CaptionWord[] }>();
1274
+ for (const f of run) {
1275
+ const seg = runByLine.get(f.line);
1276
+ if (seg) {
1277
+ seg.last = f.word;
1278
+ seg.words.push(f.w);
1279
+ } else {
1280
+ lineOrder.push(f.line);
1281
+ runByLine.set(f.line, { first: f.word, last: f.word, words: [f.w] });
1282
+ }
1283
+ }
1284
+ const shares = lineOrder.map((li) =>
1285
+ runByLine.get(li)!.words.reduce((a, w) => a + (w.end - w.start), 0),
1286
+ );
1287
+ const totalShare = shares.reduce((a, b) => a + b, 0);
1288
+ // Zero total duration (every run word zero-width) has no proportion to
1289
+ // honor — fall back to equal weights so the rounding below still lands
1290
+ // every token somewhere deterministic.
1291
+ const weights = totalShare > 0 ? shares : shares.map(() => 1);
1292
+ const weightTotal = totalShare > 0 ? totalShare : shares.length;
1293
+ const quotas = weights.map((s) => (tokens.length * s) / weightTotal);
1294
+ const counts = quotas.map((q) => Math.floor(q));
1295
+ let leftover = tokens.length - counts.reduce((a, b) => a + b, 0);
1296
+ // Largest remainder first; ties break to the EARLIER line — stated so
1297
+ // the distribution is reproducible from the doc alone, like every other
1298
+ // persisted derivation in this file.
1299
+ const byRemainder = quotas
1300
+ .map((q, i) => ({ i, rem: q - Math.floor(q) }))
1301
+ .sort((a, b) => b.rem - a.rem || a.i - b.i);
1302
+ for (let k = 0; leftover > 0; k = (k + 1) % byRemainder.length) {
1303
+ counts[byRemainder[k]!.i]!++;
1304
+ leftover--;
1305
+ }
1306
+
1307
+ const fromSrc = run[0]!.w.srcStart;
1308
+ const toSrc = run[run.length - 1]!.w.srcStart;
1309
+ const srcStarts = tokens.map((_, j) =>
1310
+ tokens.length === 1 ? fromSrc : fromSrc + ((toSrc - fromSrc) * j) / (tokens.length - 1),
1311
+ );
1312
+
1313
+ let tokenCursor = 0;
1314
+ const next: CaptionLine[] = [];
1315
+ for (let li = 0; li < out.length; li++) {
1316
+ const line = out[li]!;
1317
+ const seg = runByLine.get(li);
1318
+ if (!seg) {
1319
+ next.push(line);
1320
+ continue;
1321
+ }
1322
+ const n = counts[lineOrder.indexOf(li)]!;
1323
+ const lineTokens = tokens.slice(tokenCursor, tokenCursor + n);
1324
+ const lineSrcs = srcStarts.slice(tokenCursor, tokenCursor + n);
1325
+ tokenCursor += n;
1326
+ const minted =
1327
+ n === 0
1328
+ ? []
1329
+ : retimeCaptionTokens(
1330
+ lineTokens,
1331
+ seg.words[0]!.start,
1332
+ seg.words[seg.words.length - 1]!.end,
1333
+ lineSrcs,
1334
+ );
1335
+ const words = [...line.words.slice(0, seg.first), ...minted, ...line.words.slice(seg.last + 1)];
1336
+ // Window untouched (the no-re-pack rule above); an emptied line is
1337
+ // omitted, same as `applyCaptionWordHides`.
1338
+ if (words.length === 0) continue;
1339
+ next.push({ ...line, words });
1340
+ }
1341
+ out = next;
1342
+ }
1343
+ return { lines: out, dropped };
1344
+ }
1345
+
1346
+ /**
1347
+ * Drop hidden caption words (the `captionWordsHidden` layer). Same reporting
1348
+ * shape as `applyCaptionEdits` — callers must surface `dropped` for the same
1349
+ * reason: a hide that silently fails looks like the editor forgot it.
1350
+ *
1351
+ * Runs on the DERIVED `CaptionLine[]`, never on `transcript.words` — scene
1352
+ * anchors are raw word INDICES into the transcript, so splicing a word out of
1353
+ * it would shift every later anchor onto the wrong word: the forbidden
1354
+ * operation this whole layer exists to avoid. The transcript stays intact;
1355
+ * only the rendered caption stream loses the word.
1356
+ *
1357
+ * Line WINDOWS are recomputed here, deliberately: `buildCaptionLines` derives
1358
+ * `start` from the first word and `end` from the last word plus a hold
1359
+ * (captions.ts:203-213), so hiding a boundary word would otherwise leave the
1360
+ * line lingering on screen over silence — up for the hidden first word's
1361
+ * duration, or held past the hidden last word's end. A hidden FIRST word moves
1362
+ * `start` to the first survivor; a hidden LAST word re-bases the packer's hold
1363
+ * delta onto whichever word is now last (clamped so the line never ends before
1364
+ * its own last word); middle hides leave the window alone. A line whose words
1365
+ * are ALL hidden is omitted entirely, so the downstream CaptionTrack emits no
1366
+ * Sequence for it.
1367
+ *
1368
+ * `was` is the LIVE (post-retype) text at hide time — hides apply AFTER
1369
+ * retypes (`applyCaptionLayers` below) — so un-retyping a word under a hide
1370
+ * stales the hide, and it is REPORTED rather than guessed at. Same
1371
+ * first-claimant rule as `applyCaptionEdits`: ms-quantised keys CAN collide
1372
+ * (`captions.ts:44-50` manufactures duplicates by design), and one hide must
1373
+ * remove one word, not every word sharing its instant.
1374
+ */
1375
+ export function applyCaptionWordHides(
1376
+ lines: readonly CaptionLine[],
1377
+ hides: Record<string, { was: string }>,
1378
+ ): AppliedCaptionEdits {
1379
+ const dropped: AppliedCaptionEdits["dropped"] = [];
1380
+ if (Object.keys(hides).length === 0) return { lines: [...lines], dropped };
1381
+
1382
+ const seen = new Set<string>();
1383
+ const out: CaptionLine[] = [];
1384
+ for (const line of lines) {
1385
+ const kept: CaptionWord[] = [];
1386
+ for (const w of line.words) {
1387
+ // No anchor, no hide — same boundary rule as `applyCaptionEdits`: a
1388
+ // pre-§137 word cannot be addressed, and the stored hides then fall out
1389
+ // of the sweep below as `found: null`.
1390
+ const key = captionAnchorOf(w);
1391
+ const hide = key === null ? undefined : hides[key];
1392
+ if (key === null || !hide) {
1393
+ kept.push(w);
1394
+ continue;
1395
+ }
1396
+ // An earlier word already answered for this anchor — whichever way it
1397
+ // answered. Hiding here too would fan one delete onto a second word.
1398
+ if (seen.has(key)) {
1399
+ dropped.push({ key, expected: hide.was, found: w.text, reason: "duplicate-anchor" });
1400
+ kept.push(w);
1401
+ continue;
1402
+ }
1403
+ seen.add(key);
1404
+ if (w.text !== hide.was) {
1405
+ dropped.push({ key, expected: hide.was, found: w.text });
1406
+ kept.push(w);
1407
+ continue;
1408
+ }
1409
+ // Matched: the word is dropped from the line.
1410
+ }
1411
+ if (kept.length === line.words.length) {
1412
+ out.push(line);
1413
+ continue;
1414
+ }
1415
+ // Every word hidden — the line goes with them, rather than a zero-word
1416
+ // line the CaptionTrack would still mount a Sequence for.
1417
+ if (kept.length === 0) continue;
1418
+ const lastOriginal = line.words[line.words.length - 1]!;
1419
+ const firstKept = kept[0]!;
1420
+ const lastKept = kept[kept.length - 1]!;
1421
+ const start = firstKept === line.words[0] ? line.start : firstKept.start;
1422
+ // The packer's hold delta (captions.ts:203-213) rides on whichever word
1423
+ // is now last; clamped so the line never ends before its own last word
1424
+ // (the delta can be negative when the hold was clamped to outputDuration).
1425
+ const end =
1426
+ lastKept === lastOriginal
1427
+ ? line.end
1428
+ : Math.max(lastKept.end, lastKept.end + (line.end - lastOriginal.end));
1429
+ out.push({ words: kept, start, end });
1430
+ }
1431
+
1432
+ // An anchor no word carries any more — a later cut removed the word the
1433
+ // user hid. Silence here is the field-case failure mode, so say it.
1434
+ for (const [key, hide] of Object.entries(hides)) {
1435
+ if (!seen.has(key)) dropped.push({ key, expected: hide.was, found: null });
1436
+ }
1437
+ return { lines: out, dropped };
1438
+ }
1439
+
1440
+ /**
1441
+ * The floor a caption's window may shrink to. A caption nobody can read is a
1442
+ * delete wearing a timing nudge's clothes — deletes are the hide layer's
1443
+ * gesture, with its own guard and report. Also the minimum WIDTH of every
1444
+ * line's window, which is what keeps §115 (`packages/scenes/src/frames.ts`)
1445
+ * true: 50ms is more than one frame at any fps this renders at, so two
1446
+ * adjacent windows can never round onto the same frame.
1447
+ *
1448
+ * (Was `MIN_TIMED_WORD_SEC`, the same 0.05 measured against a WORD, until the
1449
+ * per-word layer was found inert — see `captionLineTiming`'s docstring.)
1450
+ *
1451
+ * Exported for the EDITOR's drag bounds (`captionDragBounds`,
1452
+ * apps/editor/src/TranscriptPanel.tsx): the popover has to stop a drag exactly
1453
+ * where this sweep would, and a second copy of the floor in the browser is how
1454
+ * the two would drift apart.
1455
+ */
1456
+ export const MIN_CAPTION_SEC = 0.05;
1457
+
1458
+ /**
1459
+ * Re-time a line's words from one window onto another, PROPORTIONALLY — the
1460
+ * arithmetic that keeps the karaoke highlight in sync when a line's
1461
+ * `<Sequence>` window moves under it (`CaptionTrack.tsx:228-229, 387` reads
1462
+ * the word stamps INSIDE the window; a window moved without them would light
1463
+ * the wrong words up, or none).
1464
+ *
1465
+ * The source is the WINDOW, not the words' own span: on the packed stream
1466
+ * both are the same interval (`line.start === words[0].start` and
1467
+ * `line.end === lastWord.end`, measured 117/117 — `captionLineTiming`'s
1468
+ * docstring), and on a line that DOES carry lead-in or hold (the hide layer
1469
+ * can re-base either edge) mapping the window preserves that slack instead of
1470
+ * stretching the words over it.
1471
+ *
1472
+ * Pure and exported so a caller previewing a drag and the apply pass below
1473
+ * share ONE piece of arithmetic (the openCommand/openInBrowser split).
1474
+ * Identity when the source window is degenerate — a zero-width or inverted
1475
+ * span has no ratio to scale by, and `0/0` would put NaN stamps in the render
1476
+ * props. The caller owns `toStart < toEnd`; a target handed backwards would
1477
+ * mirror the word order, which `applyCaptionLineTiming`'s edge sweep makes
1478
+ * unreachable.
1479
+ */
1480
+ export function scaleWordsIntoWindow(
1481
+ words: readonly CaptionWord[],
1482
+ fromStart: number,
1483
+ fromEnd: number,
1484
+ toStart: number,
1485
+ toEnd: number,
1486
+ ): CaptionWord[] {
1487
+ const span = fromEnd - fromStart;
1488
+ if (!(span > 0)) return words.map((w) => ({ ...w }));
1489
+ const ratio = (toEnd - toStart) / span;
1490
+ const at = (t: number): number => toStart + (t - fromStart) * ratio;
1491
+ return words.map((w) => ({ ...w, start: at(w.start), end: at(w.end) }));
1492
+ }
1493
+
1494
+ /**
1495
+ * Apply per-LINE caption TIMING nudges (`captionLineTiming`) — the LAST
1496
+ * layer, after hides, because it must operate on the SURVIVING lines: a hide
1497
+ * can move a line's window (or remove the line entirely), and a nudge stored
1498
+ * on a line the hides emptied has no window to move (it falls out of the
1499
+ * sweep as `found: null`, like every other orphaned caption record).
1500
+ *
1501
+ * EDGES, NOT ONE SHARED SEAM. Each line owns its `[start, end]` pair, and a
1502
+ * line's END and the next line's START are two separate numbers here — even
1503
+ * though on a real transcript they are always equal, because the packer chains
1504
+ * words (`transcribe.ts`: `next.start = w.end`) and clamps each line's end to
1505
+ * the next line's start (`captions.ts:203-213`), giving inter-line gaps of
1506
+ * exactly zero (measured 116/116, see `captionLineTiming`). A nudge CLOSES the
1507
+ * two onto one value only when they were already COINCIDENT: that is what
1508
+ * makes a lead on the packed stream move both sides of the boundary, one edit
1509
+ * and two windows, exactly as before.
1510
+ *
1511
+ * They are two numbers because GAPS ARE REAL: `applyCaptionWordHides` re-bases
1512
+ * a line's window onto its surviving words, `MAX_CAPTION_WORD_LEAD_SEC`
1513
+ * (captions.ts:147, 169) clamps a word's display start, and an overrides.json
1514
+ * can be hand-edited. This code
1515
+ * used to hold ONE `seams` array whose interior entry was read off the later
1516
+ * line's start, conflating the two: with lines `[0,2] [2,4] [5,6]`, a
1517
+ * lead-only drag of the middle line (`{lead: -0.05, tail: 0}`, exactly what
1518
+ * the editor writes) rebuilt the UNTOUCHED third caption as `[4,6]` — a full
1519
+ * second early, its words stretched 2x by `scaleWordsIntoWindow`, with no drop
1520
+ * reported (review 2026-08-19).
1521
+ *
1522
+ * The edge model still protects §115 (`packages/scenes/src/frames.ts:1-21` —
1523
+ * no two lines may share a frame) BY CONSTRUCTION, which is what the old "LINE
1524
+ * WINDOWS NEVER CHANGE" rule existed for: the sweep below leaves the edges
1525
+ * ORDERED (`start_0 <= end_0 <= start_1 <= ... <= end_n-1`) with every window
1526
+ * at least `MIN_CAPTION_SEC` wide, and ordered non-overlapping windows at
1527
+ * least 50ms wide cannot round onto a shared frame.
1528
+ *
1529
+ * THE SWEEP, forward: every edge is clamped into the track's ORIGINAL outer
1530
+ * bounds, no line may open before the previous line CLOSED, and no window may
1531
+ * be narrower than `MIN_CAPTION_SEC`. A backward pass then pulls lines left if
1532
+ * a track too short to hold every line at the floor made the forward pass run
1533
+ * into the end. Ordering is enforced against the NEIGHBOUR'S OWN edge, never a
1534
+ * derived seam: a nudge that runs past it is BLOCKED there rather than pushing
1535
+ * it, so a gap gets consumed but no untouched caption ever moves. (A nudge
1536
+ * takes time FROM a neighbour only through the coincidence rule above — the
1537
+ * packed case, where the two share the boundary being dragged.) The outer
1538
+ * bounds never GROW: a caption must not appear before the first caption of the
1539
+ * track or linger past the last, where there is no output left to show it
1540
+ * over.
1541
+ *
1542
+ * BOTH SIDES OF ONE BOUNDARY: line i's `tail` and line i+1's `lead` address
1543
+ * the same coincident boundary. The LATER line's `lead` wins,
1544
+ * deterministically — the UI writes both sides of a drag consistently, so this
1545
+ * only decides hand-edited docs, and a stated winner beats an
1546
+ * order-of-iteration accident. (A stored `lead: 0` still claims its edge; an
1547
+ * entry the user cleared is DELETED from the doc, not written as zeros.)
1548
+ *
1549
+ * Lines whose window the sweep did not move are returned VERBATIM — including
1550
+ * their word stamps — so a nudge on one caption cannot perturb the rest of
1551
+ * the track. The ones that did move (the nudged line AND, on a coincident
1552
+ * boundary, its neighbour) have their words scaled into the new window by
1553
+ * `scaleWordsIntoWindow`.
1554
+ *
1555
+ * DELIBERATELY NO `was` GUARD, unlike `captionWordsHidden`: timing is
1556
+ * text-orthogonal — a retype under a timing nudge changes what the caption
1557
+ * says, not when it is said, and staleness on text would drop nudges the user
1558
+ * never un-meant. `expected` in the drop reports is therefore always `""`
1559
+ * (the record stores no text to expect). Same first-claimant rule as every
1560
+ * per-word layer: ms-quantised anchors CAN collide (captions.ts:44-50), and
1561
+ * one nudge must move one line.
1562
+ */
1563
+ export function applyCaptionLineTiming(
1564
+ lines: readonly CaptionLine[],
1565
+ timing: Record<string, { lead: number; tail: number }>,
1566
+ ): AppliedCaptionEdits {
1567
+ const dropped: AppliedCaptionEdits["dropped"] = [];
1568
+ const n = lines.length;
1569
+ // NO LINES is not "no nudges to report": every stored key is an anchor that
1570
+ // no line starts on, which is exactly the `found: null` case the sweep at
1571
+ // the bottom exists to say out loud, and what this function's own docstring
1572
+ // promises. `applyCaptionEdits` and `applyCaptionWordHides` never took this
1573
+ // shortcut either. The editor's false-banner guard lives at the CALLER
1574
+ // (`App.tsx`: `if (!renderProps) return { lines: [], dropped: [] }`), where
1575
+ // "nothing loaded yet" is distinguishable from "this cut has no captions" —
1576
+ // silence here instead let produce report nudges as applied that never were.
1577
+ if (n === 0) {
1578
+ for (const key of Object.keys(timing)) dropped.push({ key, expected: "", found: null });
1579
+ return { lines: [], dropped };
1580
+ }
1581
+ if (Object.keys(timing).length === 0) return { lines: [...lines], dropped };
1582
+
1583
+ // One `[start, end]` pair PER LINE — never a shared seam array (see the
1584
+ // docstring: the conflation moved untouched captions on a gapped stream).
1585
+ const starts = lines.map((l) => l.start);
1586
+ const ends = lines.map((l) => l.end);
1587
+
1588
+ const seen = new Set<string>();
1589
+ for (let i = 0; i < n; i++) {
1590
+ const line = lines[i]!;
1591
+ // No anchor, no nudge — the same boundary rule as `applyCaptionEdits`: a
1592
+ // pre-§137 word cannot be addressed, and the stored nudges then fall out
1593
+ // of the sweep below as `found: null`.
1594
+ const key = captionAnchorOf(line.words[0]);
1595
+ const entry = key === null ? undefined : timing[key];
1596
+ if (key === null || !entry) continue;
1597
+ // An earlier line already answered for this anchor — nudging here too
1598
+ // would fan one nudge onto a second line.
1599
+ if (seen.has(key)) {
1600
+ dropped.push({ key, expected: "", found: line.words[0]!.text, reason: "duplicate-anchor" });
1601
+ continue;
1602
+ }
1603
+ seen.add(key);
1604
+ // Deltas ride on the line's OWN edges, so a gapped stream moves the edge
1605
+ // the user dragged rather than the neighbour's. Tail first, then lead:
1606
+ // lines are visited in order, so line i+1's lead lands on a shared
1607
+ // boundary AFTER line i's tail — the documented "later lead wins".
1608
+ ends[i] = line.end + entry.tail;
1609
+ // COINCIDENCE, tested against the ORIGINAL edges: only a boundary the two
1610
+ // lines already SHARED travels with the nudge (the packed stream, where
1611
+ // every one of them is shared). Across a gap the neighbour stays where it
1612
+ // is — the sweep below still stops the moved edge from crossing it.
1613
+ // Assigning the same number, not recomputing it, keeps the two exactly
1614
+ // equal: a float `+ delta` computed twice can differ in the last bit, and
1615
+ // an unequal pair is an overlap the sweep would then have to fix.
1616
+ if (i + 1 < n && lines[i + 1]!.start === line.end) starts[i + 1] = ends[i]!;
1617
+ starts[i] = line.start + entry.lead;
1618
+ if (i > 0 && lines[i - 1]!.end === line.start) ends[i - 1] = starts[i]!;
1619
+ }
1620
+
1621
+ const lo = lines[0]!.start;
1622
+ const hi = lines[n - 1]!.end;
1623
+ // Forward: into the track's bounds, never opening before the previous line
1624
+ // CLOSED (its own edge, not a derived seam), never narrower than the floor.
1625
+ for (let i = 0; i < n; i++) {
1626
+ const floor = i === 0 ? lo : Math.max(lo, ends[i - 1]!);
1627
+ starts[i] = Math.min(Math.max(starts[i]!, floor), hi);
1628
+ ends[i] = Math.min(Math.max(ends[i]!, starts[i]! + MIN_CAPTION_SEC), hi);
1629
+ }
1630
+ // The forward pass caps at `hi`, so a track with less room than
1631
+ // `n * MIN_CAPTION_SEC` can leave the last lines piled on the end. Pull them
1632
+ // back (never before `lo`) so the edges stay ordered.
1633
+ for (let i = n - 1; i >= 0; i--) {
1634
+ const ceil = i === n - 1 ? hi : Math.min(hi, starts[i + 1]!);
1635
+ ends[i] = Math.max(Math.min(ends[i]!, ceil), lo);
1636
+ starts[i] = Math.max(Math.min(starts[i]!, ends[i]! - MIN_CAPTION_SEC), lo);
1637
+ }
1638
+
1639
+ const out = lines.map((line, i) => {
1640
+ const start = starts[i]!;
1641
+ const end = ends[i]!;
1642
+ // Neither edge moved: VERBATIM, same reference and same word stamps.
1643
+ if (start === line.start && end === line.end) return line;
1644
+ return {
1645
+ ...line,
1646
+ start,
1647
+ end,
1648
+ words: scaleWordsIntoWindow(line.words, line.start, line.end, start, end),
1649
+ };
1650
+ });
1651
+
1652
+ // An anchor no line starts on any more — a later cut removed the word the
1653
+ // line was keyed to, or a hide emptied the line. Silence here is the
1654
+ // field-case failure mode, so say it.
1655
+ for (const key of Object.keys(timing)) {
1656
+ if (!seen.has(key)) dropped.push({ key, expected: "", found: null });
1657
+ }
1658
+ return { lines: out, dropped };
1659
+ }
1660
+
1661
+ export interface AppliedCaptionLayers {
1662
+ lines: CaptionLine[];
1663
+ /** Every layer's drop reports, tagged with which layer refused them. */
1664
+ dropped: Array<
1665
+ AppliedCaptionEdits["dropped"][number] & { layer: "edit" | "range" | "hide" | "timing" }
1666
+ >;
1667
+ }
1668
+
1669
+ /**
1670
+ * The caption edit layers, composed in their ONE authoritative order — the
1671
+ * single chokepoint both the editor preview and produce consume, so the two
1672
+ * can never disagree about caption content.
1673
+ *
1674
+ * Per-word edits → RANGE edits → hides → LINE TIMING. Edits BEFORE hides is the
1675
+ * `was` contract: a hide's `was` records the LIVE text the user saw when they
1676
+ * deleted the word, which is the post-retype text — running hides first
1677
+ * would stale every hide sitting on a retyped word. Range edits sit between
1678
+ * the two, but the order barely earns the word: the reducer's creation-time
1679
+ * scrubbing (`useEdits`' `patchCaptionRange`) removes every per-word edit
1680
+ * and hide inside a new range's interval, so a LIVE range edit never
1681
+ * coexists with either inside its own words — the order only matters for
1682
+ * hand-edited docs, where the layers' own guards report rather than guess.
1683
+ * Timing runs LAST because it must see the surviving LINES: the hide layer
1684
+ * re-bases a line's window onto its surviving words and drops a line whose
1685
+ * words are all hidden, and a nudge on a line that no longer exists has no
1686
+ * window to move (`applyCaptionLineTiming`). Drop reports carry which layer
1687
+ * refused them, since "the retype missed", "the rewrite missed" and "the
1688
+ * delete missed" send the user to different gestures.
1689
+ */
1690
+ export function applyCaptionLayers(
1691
+ lines: readonly CaptionLine[],
1692
+ doc: OverrideDoc,
1693
+ ): AppliedCaptionLayers {
1694
+ const edited = applyCaptionEdits(lines, doc.captions);
1695
+ const ranged = applyCaptionRangeEdits(edited.lines, doc.captionRangeEdits);
1696
+ const hidden = applyCaptionWordHides(ranged.lines, doc.captionWordsHidden);
1697
+ const timed = applyCaptionLineTiming(hidden.lines, doc.captionLineTiming);
1698
+ return {
1699
+ lines: timed.lines,
1700
+ dropped: [
1701
+ ...edited.dropped.map((d) => ({ ...d, layer: "edit" as const })),
1702
+ ...ranged.dropped.map((d) => ({ ...d, layer: "range" as const })),
1703
+ ...hidden.dropped.map((d) => ({ ...d, layer: "hide" as const })),
1704
+ ...timed.dropped.map((d) => ({ ...d, layer: "timing" as const })),
1705
+ ],
1706
+ };
1707
+ }
1708
+
981
1709
  /** Theme tokens the user set, over whatever the production already had. */
982
1710
  export function resolveTheme(base: Theme, doc: OverrideDoc): Theme {
983
1711
  return ThemeSchema.parse({ ...base, ...doc.theme });