@ossclip/core 0.1.25 → 0.1.27

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/phonetics.ts CHANGED
@@ -12,6 +12,49 @@
12
12
  * "unrelated", and it must stay dependency-free and deterministic.
13
13
  */
14
14
 
15
+ /**
16
+ * Marks Arabic-script text carries that two transcribers disagree about
17
+ * without disagreeing about the WORD: harakat/vowel diacritics
18
+ * (U+064B–U+065F, U+0670), tatweel (U+0640, a pure typographic stretch), and
19
+ * the zero-width joiners (U+200C/U+200D). Whisper emits them inconsistently
20
+ * and an LLM writing a correction rarely reproduces them, so leaving them in
21
+ * makes a correct repair either miss `locate()` outright or read as
22
+ * "different from what was heard" purely on invisible marks. Only tatweel
23
+ * survives the `\p{L}\p{N}` filter below (it is Lm, a letter); the rest are
24
+ * stripped here so the intent is legible rather than an accident of Unicode
25
+ * categories.
26
+ */
27
+ // Escaped, not literal: three of these code points are invisible in an editor.
28
+ const ARABIC_NOISE = /[\u064B-\u065F\u0670\u0640\u200C\u200D]/g;
29
+
30
+ /**
31
+ * Comparable form for two pieces of text: case-, punctuation- and
32
+ * whitespace-insensitive, in ANY script.
33
+ *
34
+ * Shared by `phonetics.ts` and `producer/repair.ts` on purpose — they used to
35
+ * hold two copies and the copy in `repair.ts` was `[^a-z0-9\s]`, i.e.
36
+ * Latin-only. Field case (2026-08-18): every one of 11 recorded Urdu repairs
37
+ * normalized to the empty string, so `norm(heard) === norm(correction)` was
38
+ * `"" === ""` and ALL 11 were refused as "identical to what was heard" —
39
+ * including `پرسٹ` → `فرسٹ`, which shares no letters with what it replaced.
40
+ *
41
+ * Keeping letters and digits of every script also KEEPS accented Latin
42
+ * ("café", "über") where the old expression deleted it. That is the same bug
43
+ * in miniature — a French word normalized to "caf" — so it is a fix, not a
44
+ * regression. Pure-ASCII input is byte-identical to the old behaviour, which
45
+ * is pinned by a test.
46
+ */
47
+ export function normalizeForCompare(s: string): string {
48
+ return s
49
+ .normalize("NFC")
50
+ .toLowerCase()
51
+ .replace(ARABIC_NOISE, "")
52
+ .replace(/[^\p{L}\p{N}\s]/gu, "")
53
+ .split(/\s+/)
54
+ .filter(Boolean)
55
+ .join(" ");
56
+ }
57
+
15
58
  /**
16
59
  * Digraphs collapsed before single letters, longest first. The ch/sh and th
17
60
  * sounds get DIGIT placeholders on purpose: a letter placeholder would be
@@ -106,6 +149,47 @@ export function soundsLike(a: string, b: string): number {
106
149
  return Math.max(0, 1 - dist / Math.max(ka.length, kb.length));
107
150
  }
108
151
 
152
+ /**
153
+ * 0..1 similarity of the TEXT itself, for scripts the phonetic key cannot
154
+ * represent (`phoneticKey` is defined over a-z, so anything non-Latin keys to
155
+ * ""). Same shape as `soundsLike` — normalized edit distance over the longer
156
+ * string — but run on the normalized text rather than a consonant skeleton.
157
+ *
158
+ * This is a weaker signal than a phonetic key and it is meant to be: an
159
+ * Urdu-script mishearing differs from the truth by a letter or two of the same
160
+ * script, so edit distance still separates it from an unrelated phrase. What
161
+ * it cannot do is fold vowels, which is why the floor below is calibrated
162
+ * against real data instead of borrowing SOUNDS_LIKE_FLOOR.
163
+ */
164
+ export function textSimilarity(a: string, b: string): number {
165
+ const na = normalizeForCompare(a);
166
+ const nb = normalizeForCompare(b);
167
+ if (na.length === 0 && nb.length === 0) return 1;
168
+ if (na.length === 0 || nb.length === 0) return 0;
169
+ return Math.max(0, 1 - levenshtein(na, nb) / Math.max(na.length, nb.length));
170
+ }
171
+
172
+ /**
173
+ * Floor for the non-Latin fallback, MEASURED rather than guessed.
174
+ *
175
+ * The 11 Urdu repairs recorded in a real production.json (2026-08-18) score,
176
+ * sorted: 0.333, 0.400, 0.500, 0.500, 0.545, 0.583, 0.636, 0.750, 0.750,
177
+ * 0.800, 0.800. The 0.333 is `حقیقہ ٹون` → `ہیکاتھون` ("hackathon"), a genuine
178
+ * repair and the worst of the set because the recognizer both re-segmented the
179
+ * word and changed its opening letter. Admitting it sets the ceiling on the
180
+ * floor; 0.33 is the largest value that does.
181
+ *
182
+ * Against that, unrelated four-word spans lifted from the same transcript
183
+ * score 0.167–0.250 and are refused. The band is narrow, and it is narrow for
184
+ * the same reason the Latin one is (see SOUNDS_LIKE_FLOOR): two SHORT
185
+ * unrelated Urdu spans can still land above it — measured, `پرسٹ ہیک` vs
186
+ * `ٹرس می` scores 0.500. There is no onset test here to catch that, because
187
+ * two of the 11 genuine repairs change their first letter. So this gate is
188
+ * real but shallow; the span, token-count and length guards in
189
+ * `applyRepairs` are what keep it from being a rewrite licence.
190
+ */
191
+ export const TEXT_SIMILARITY_FLOOR = 0.33;
192
+
109
193
  /**
110
194
  * Default floor for "this is a repair, not a rewrite". Deliberately low,
111
195
  * because a real mishearing can move word boundaries ("code churn" → "coach
@@ -124,11 +208,27 @@ export const SOUNDS_LIKE_FLOOR = 0.34;
124
208
  * a phrase, essentially never its onset. This is what rejects a rewrite:
125
209
  * "revenue" for "churn" and "monetization" for "agents" both score in the
126
210
  * same range as a true repair, and both fail the onset test.
211
+ *
212
+ * When either side has no Latin letters there is no key to compare, and this
213
+ * used to answer `ka === kb` — `"" === ""`, i.e. YES for any two non-Latin
214
+ * strings however unrelated. That is no gate at all for an Urdu transcript, so
215
+ * those pairs route to `textSimilarity` instead (2026-08-18 field case).
216
+ * Latin-to-Latin comparisons never reach that branch and are unchanged.
127
217
  */
128
218
  export function soundsSimilar(a: string, b: string, floor = SOUNDS_LIKE_FLOOR): boolean {
129
219
  const ka = phraseKey(a);
130
220
  const kb = phraseKey(b);
131
- if (ka.length === 0 || kb.length === 0) return ka === kb;
221
+ if (ka.length === 0 || kb.length === 0) {
222
+ // `floor` is deliberately NOT reused here: it is calibrated against
223
+ // consonant skeletons, which are shorter and coarser than the text this
224
+ // branch compares, so the same number means something else. Taking the
225
+ // larger of the two was tried and is wrong — the default 0.34 alone
226
+ // rejects a measured genuine repair scoring 0.333. The only caller that
227
+ // raises the floor (reconcileCopy, 0.6) cannot reach this branch anyway:
228
+ // its candidate tokens are stripped to `[A-Za-z]`, so its key is never
229
+ // empty and a non-Latin spoken word scores ~0 against it regardless.
230
+ return textSimilarity(a, b) >= TEXT_SIMILARITY_FLOOR;
231
+ }
132
232
  if (ka[0] !== kb[0]) return false;
133
233
  return soundsLike(a, b) >= floor;
134
234
  }
@@ -1,7 +1,7 @@
1
1
  import { z } from "zod/v4";
2
2
  import type { Transcript, Word } from "../schema";
3
3
  import type { Scene } from "../scene-schema";
4
- import { soundsSimilar } from "../phonetics";
4
+ import { normalizeForCompare, soundsSimilar } from "../phonetics";
5
5
  import type { LlmProvider } from "./provider";
6
6
 
7
7
  /**
@@ -105,15 +105,18 @@ export function buildRepairUserPrompt(
105
105
  );
106
106
  }
107
107
 
108
- /** Comparable form: case- and punctuation-insensitive. */
109
- function norm(s: string): string {
110
- return s
111
- .toLowerCase()
112
- .replace(/[^a-z0-9\s]/g, "")
113
- .split(/\s+/)
114
- .filter(Boolean)
115
- .join(" ");
116
- }
108
+ /**
109
+ * Comparable form: case- and punctuation-insensitive, in any script.
110
+ *
111
+ * This was a second, Latin-only copy (`[^a-z0-9\s]`) of what is now
112
+ * `normalizeForCompare`. The two drifted in the worst possible way: on an Urdu
113
+ * transcript every string normalized to "", so `norm(actual) ===
114
+ * norm(r.correction)` was true for every proposal and all 11 repairs in a real
115
+ * run were refused as "identical to what was heard" (2026-08-18) — while
116
+ * `locate()`, comparing "" to "", "matched" the first span it tried without
117
+ * verifying anything. One shared helper so they cannot drift again.
118
+ */
119
+ const norm = normalizeForCompare;
117
120
 
118
121
  function spanText(transcript: Transcript, startWord: number, endWord: number): string {
119
122
  return transcript.words
@@ -228,6 +231,12 @@ export function applyRepairs(
228
231
  */
229
232
  const locate = (r: TranscriptRepair): { startWord: number; endWord: number } | null => {
230
233
  const want = norm(r.heard);
234
+ // A quote that normalizes to nothing is not an anchor: "" compares equal
235
+ // to the first span whose own normalisation is empty, so the search would
236
+ // "find" a span it never verified. That was live for every non-Latin
237
+ // transcript until norm() was fixed above; refuse it explicitly so it
238
+ // cannot come back through some other all-punctuation quote.
239
+ if (want.length === 0) return null;
231
240
  // Widths to try, in order of trust. The QUOTED TEXT is the reliable part
232
241
  // of a proposal, so its own token count leads; the claimed span is a
233
242
  // fallback for a quote whose normalisation splits differently. Trusting
package/src/recut.ts CHANGED
@@ -391,3 +391,64 @@ export function applyUserCuts(
391
391
  removedSec,
392
392
  };
393
393
  }
394
+
395
+ /** What `pruneHidesInsideCuts` hands back: the doc (same reference when
396
+ * nothing was pruned — the caller's changed-gate reads `pruned.length`), and
397
+ * the retired keys so produce can SAY what it retired. */
398
+ export interface PrunedHides {
399
+ doc: OverrideDoc;
400
+ pruned: string[];
401
+ }
402
+
403
+ /**
404
+ * Retire `captionWordsHidden` entries whose word the final cutlist REMOVES
405
+ * (§59b revisited 2026-08-18 — the "captions + video" delete gesture writes
406
+ * both a hide and a cut in one commit).
407
+ *
408
+ * Once the cut lands, `buildCaptionLines` drops the word before the hide
409
+ * layer ever sees it, so the hide key would report `found: null` ("the cut
410
+ * removed it", `captionHideDropLine`) on every subsequent run forever — the
411
+ * cut SUPERSEDES the hide, the same superseded philosophy `overrides.ts`'s
412
+ * caption-key migration applies. Hides whose source instant is OUTSIDE every
413
+ * removed segment are kept verbatim — as are keys that are not §137 `w<ms>`
414
+ * anchors at all, which name no instant this can test (see the guard below).
415
+ *
416
+ * HALF-OPEN interval (`srcIn <= src < srcOut`), on purpose — the two edges
417
+ * are NOT symmetric. A word starting exactly at `srcIn` IS cut: that is
418
+ * precisely where the FIRST word of a captions+video delete lands (its
419
+ * srcStart round-trips through the TimeMap to the resolved cut's own srcIn),
420
+ * and `mapWord` clamps that instant into the removal and drops the word — a
421
+ * strictly-inside test never retired the gesture's own first hide, leaving
422
+ * it a permanent `found: null` drop report. A word starting exactly at
423
+ * `srcOut` belongs to the NEXT kept span (`buildCaptionLines` still emits it
424
+ * — a seam instant has a kept-side preimage, `timemap.ts`), so its hide is
425
+ * still doing work and must survive.
426
+ */
427
+ export function pruneHidesInsideCuts(doc: OverrideDoc, cutlist: readonly Segment[]): PrunedHides {
428
+ const pruned: string[] = [];
429
+ const kept: OverrideDoc["captionWordsHidden"] = {};
430
+ for (const [key, entry] of Object.entries(doc.captionWordsHidden)) {
431
+ // SOURCE-KEYED ONLY, parsed and not coerced. `captionWordsHidden` is an
432
+ // unpinned `z.record` (unlike `CaptionRangeEditSchema`'s `/^w\d+$/`), so a
433
+ // hand-edited or legacy-keyed doc reaches here: a POSITIONAL key like "17"
434
+ // would slice to "7", parse as 7ms, land inside any early cut and be
435
+ // deleted with nothing said. The editor guards the identical case and
436
+ // states the rule (`apps/editor/src/useEdits.ts:587-600`): only §137
437
+ // `w<ms>` keys carry an interval-testable instant, and an entry this
438
+ // function cannot honestly locate is KEPT.
439
+ if (!/^w\d+$/.test(key)) {
440
+ kept[key] = entry;
441
+ continue;
442
+ }
443
+ // `captionKeyFor`'s quantization inverted (`w${Math.round(sec * 1000)}`,
444
+ // overrides.ts): the key IS the word's source instant, ms-quantized.
445
+ const srcSec = parseInt(key.slice(1), 10) / 1000;
446
+ const removed = cutlist.some(
447
+ (seg) => seg.kind === "remove" && srcSec >= seg.srcIn && srcSec < seg.srcOut,
448
+ );
449
+ if (removed) pruned.push(key);
450
+ else kept[key] = entry;
451
+ }
452
+ if (pruned.length === 0) return { doc, pruned };
453
+ return { doc: { ...doc, captionWordsHidden: kept }, pruned };
454
+ }
package/src/transcribe.ts CHANGED
@@ -78,6 +78,51 @@ function repairSplitSegments(json: WhisperJson): WhisperJson {
78
78
  };
79
79
  }
80
80
 
81
+ /**
82
+ * Run length at which a stack of zero-length words at ONE instant stops being
83
+ * a rounding artifact and becomes a repetition-loop hallucination. Real speech
84
+ * never emits 8 tokens at a single instant; the field case emitted 118.
85
+ */
86
+ export const REPETITION_BURST_MIN = 8;
87
+
88
+ /**
89
+ * Drop whisper repetition-loop bursts (field case 2026-08-18): an Urdu take
90
+ * re-decoded a whole phrase as 118 CONSECUTIVE tokens all stamped
91
+ * `from === to === 31040` — zero length, at one instant. The stamp repair
92
+ * below then fans such a burst out into 118 fabricated 50ms words marching
93
+ * forward from 31.04s, so the phrase ships TWICE in the captions (31.04s and
94
+ * 33.54s) and a fifth of the transcript carries the tell-tale exactly-0.05s
95
+ * duration. `-mc 0` in whisperArgs is the decoder-side mitigation for the same
96
+ * failure; it did not prevent this occurrence, and it can never repair an
97
+ * already-cached transcript.json — hence a parse-side guard too.
98
+ *
99
+ * A burst is a MAXIMAL run of consecutive zero-length/inverted words sharing
100
+ * one `start`. Equality is exact, not epsilon: these stamps are integer
101
+ * milliseconds divided by 1000, so members of one burst are the same double
102
+ * bit-for-bit, and a tolerance would only start swallowing real neighbors.
103
+ * Runs shorter than REPETITION_BURST_MIN fall through untouched — a lone
104
+ * zero-length stamp is a rounding artifact, not a hallucination. The drop is
105
+ * silent by design: this function is pure and total, and there is no logging
106
+ * channel in the parse path to warn on.
107
+ */
108
+ export function dropRepetitionBursts(words: readonly Word[]): Word[] {
109
+ const out: Word[] = [];
110
+ let i = 0;
111
+ while (i < words.length) {
112
+ const w = words[i]!;
113
+ if (w.end > w.start) {
114
+ out.push(w);
115
+ i++;
116
+ continue;
117
+ }
118
+ let j = i + 1;
119
+ while (j < words.length && words[j]!.end <= words[j]!.start && words[j]!.start === w.start) j++;
120
+ if (j - i < REPETITION_BURST_MIN) for (let k = i; k < j; k++) out.push(words[k]!);
121
+ i = j;
122
+ }
123
+ return out;
124
+ }
125
+
81
126
  const STRICT_UTF8 = new TextDecoder("utf-8", { fatal: true });
82
127
 
83
128
  /**
@@ -115,11 +160,19 @@ export function parseWhisperJson(json: WhisperJson): Transcript {
115
160
  if (!raw || !raw.trim()) continue;
116
161
  const text = raw.trim();
117
162
  if (NOISE_TOKEN.test(text)) continue;
163
+ // A word already CLOSED by Arabic-script sentence punctuation refuses
164
+ // continuations (field case 2026-08-18): whisper emits `۔` and the next
165
+ // sentence's first token with no leading whitespace, and the plain
166
+ // whitespace rule fused them into one unsplittable word ("ہوں۔اس").
167
+ // Deliberately NOT the Latin `.`/`!`/`?` — whisper tokenizes decimals
168
+ // ("3", ".", "5") and abbreviations as bare continuations too, and
169
+ // splitting those would shred "3.5" into two words. ۔ (U+06D4) and
170
+ // ؟ (U+061F) have no such second job.
118
171
  const startsWord = /^\s/.test(raw) || words.length === 0;
119
172
  const start = seg.offsets.from / 1000;
120
173
  const end = seg.offsets.to / 1000;
121
174
  const last = words[words.length - 1];
122
- if (!startsWord && last) {
175
+ if (!startsWord && last && !/[۔؟]$/.test(last.text)) {
123
176
  last.text += text;
124
177
  last.end = Math.max(last.end, end);
125
178
  } else {
@@ -146,14 +199,18 @@ export function parseWhisperJson(json: WhisperJson): Transcript {
146
199
  else if (next) next.start = Math.min(next.start, w.start);
147
200
  words.splice(i, 1);
148
201
  }
202
+ // BEFORE the repair loop, never after: the repair rewrites every burst
203
+ // member into a distinct monotone stamp, so once it has run the shared
204
+ // timestamp — the only evidence a burst existed — is gone.
205
+ const kept = dropRepetitionBursts(words);
149
206
  // Whisper occasionally emits zero-length or inverted stamps; repair minimally.
150
- for (let i = 0; i < words.length; i++) {
151
- const w = words[i]!;
207
+ for (let i = 0; i < kept.length; i++) {
208
+ const w = kept[i]!;
152
209
  if (w.end <= w.start) w.end = w.start + 0.05;
153
- const next = words[i + 1];
210
+ const next = kept[i + 1];
154
211
  if (next && next.start < w.end) next.start = w.end;
155
212
  }
156
- return { language: json.result?.language ?? "en", words };
213
+ return { language: json.result?.language ?? "en", words: kept };
157
214
  }
158
215
 
159
216
  export interface WhisperOptions {
@@ -192,6 +249,15 @@ export function whisperArgs(opts: WhisperOptions, wavPath: string): string[] {
192
249
  "-oj",
193
250
  "-of", opts.outBase,
194
251
  "-ml", "1",
252
+ // No text context across 30s decode windows (field case 2026-08-18): an
253
+ // Urdu take hit whisper's repetition loop — a whole sentence re-decoded
254
+ // as 261 zero-duration tokens — and carrying the previous window's text
255
+ // into the decoder is the known trigger. `-mc 0` is the standard
256
+ // mitigation and leaves `--prompt` (the dictionary bias) untouched.
257
+ // Cached transcript.json files decoded without it are knowingly still
258
+ // reused (transcriptCacheReusable's no-spurious-retranscribe rule);
259
+ // delete a workdir's transcript.json to re-decode with it.
260
+ "-mc", "0",
195
261
  "--no-prints",
196
262
  ];
197
263
  if (opts.language !== undefined) args.push("-l", opts.language);