@ossclip/core 0.1.24 → 0.1.26

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/phonetics.ts CHANGED
@@ -12,6 +12,49 @@
12
12
  * "unrelated", and it must stay dependency-free and deterministic.
13
13
  */
14
14
 
15
+ /**
16
+ * Marks Arabic-script text carries that two transcribers disagree about
17
+ * without disagreeing about the WORD: harakat/vowel diacritics
18
+ * (U+064B–U+065F, U+0670), tatweel (U+0640, a pure typographic stretch), and
19
+ * the zero-width joiners (U+200C/U+200D). Whisper emits them inconsistently
20
+ * and an LLM writing a correction rarely reproduces them, so leaving them in
21
+ * makes a correct repair either miss `locate()` outright or read as
22
+ * "different from what was heard" purely on invisible marks. Only tatweel
23
+ * survives the `\p{L}\p{N}` filter below (it is Lm, a letter); the rest are
24
+ * stripped here so the intent is legible rather than an accident of Unicode
25
+ * categories.
26
+ */
27
+ // Escaped, not literal: three of these code points are invisible in an editor.
28
+ const ARABIC_NOISE = /[\u064B-\u065F\u0670\u0640\u200C\u200D]/g;
29
+
30
+ /**
31
+ * Comparable form for two pieces of text: case-, punctuation- and
32
+ * whitespace-insensitive, in ANY script.
33
+ *
34
+ * Shared by `phonetics.ts` and `producer/repair.ts` on purpose — they used to
35
+ * hold two copies and the copy in `repair.ts` was `[^a-z0-9\s]`, i.e.
36
+ * Latin-only. Field case (2026-08-18): every one of 11 recorded Urdu repairs
37
+ * normalized to the empty string, so `norm(heard) === norm(correction)` was
38
+ * `"" === ""` and ALL 11 were refused as "identical to what was heard" —
39
+ * including `پرسٹ` → `فرسٹ`, which shares no letters with what it replaced.
40
+ *
41
+ * Keeping letters and digits of every script also KEEPS accented Latin
42
+ * ("café", "über") where the old expression deleted it. That is the same bug
43
+ * in miniature — a French word normalized to "caf" — so it is a fix, not a
44
+ * regression. Pure-ASCII input is byte-identical to the old behaviour, which
45
+ * is pinned by a test.
46
+ */
47
+ export function normalizeForCompare(s: string): string {
48
+ return s
49
+ .normalize("NFC")
50
+ .toLowerCase()
51
+ .replace(ARABIC_NOISE, "")
52
+ .replace(/[^\p{L}\p{N}\s]/gu, "")
53
+ .split(/\s+/)
54
+ .filter(Boolean)
55
+ .join(" ");
56
+ }
57
+
15
58
  /**
16
59
  * Digraphs collapsed before single letters, longest first. The ch/sh and th
17
60
  * sounds get DIGIT placeholders on purpose: a letter placeholder would be
@@ -106,6 +149,47 @@ export function soundsLike(a: string, b: string): number {
106
149
  return Math.max(0, 1 - dist / Math.max(ka.length, kb.length));
107
150
  }
108
151
 
152
+ /**
153
+ * 0..1 similarity of the TEXT itself, for scripts the phonetic key cannot
154
+ * represent (`phoneticKey` is defined over a-z, so anything non-Latin keys to
155
+ * ""). Same shape as `soundsLike` — normalized edit distance over the longer
156
+ * string — but run on the normalized text rather than a consonant skeleton.
157
+ *
158
+ * This is a weaker signal than a phonetic key and it is meant to be: an
159
+ * Urdu-script mishearing differs from the truth by a letter or two of the same
160
+ * script, so edit distance still separates it from an unrelated phrase. What
161
+ * it cannot do is fold vowels, which is why the floor below is calibrated
162
+ * against real data instead of borrowing SOUNDS_LIKE_FLOOR.
163
+ */
164
+ export function textSimilarity(a: string, b: string): number {
165
+ const na = normalizeForCompare(a);
166
+ const nb = normalizeForCompare(b);
167
+ if (na.length === 0 && nb.length === 0) return 1;
168
+ if (na.length === 0 || nb.length === 0) return 0;
169
+ return Math.max(0, 1 - levenshtein(na, nb) / Math.max(na.length, nb.length));
170
+ }
171
+
172
+ /**
173
+ * Floor for the non-Latin fallback, MEASURED rather than guessed.
174
+ *
175
+ * The 11 Urdu repairs recorded in a real production.json (2026-08-18) score,
176
+ * sorted: 0.333, 0.400, 0.500, 0.500, 0.545, 0.583, 0.636, 0.750, 0.750,
177
+ * 0.800, 0.800. The 0.333 is `حقیقہ ٹون` → `ہیکاتھون` ("hackathon"), a genuine
178
+ * repair and the worst of the set because the recognizer both re-segmented the
179
+ * word and changed its opening letter. Admitting it sets the ceiling on the
180
+ * floor; 0.33 is the largest value that does.
181
+ *
182
+ * Against that, unrelated four-word spans lifted from the same transcript
183
+ * score 0.167–0.250 and are refused. The band is narrow, and it is narrow for
184
+ * the same reason the Latin one is (see SOUNDS_LIKE_FLOOR): two SHORT
185
+ * unrelated Urdu spans can still land above it — measured, `پرسٹ ہیک` vs
186
+ * `ٹرس می` scores 0.500. There is no onset test here to catch that, because
187
+ * two of the 11 genuine repairs change their first letter. So this gate is
188
+ * real but shallow; the span, token-count and length guards in
189
+ * `applyRepairs` are what keep it from being a rewrite licence.
190
+ */
191
+ export const TEXT_SIMILARITY_FLOOR = 0.33;
192
+
109
193
  /**
110
194
  * Default floor for "this is a repair, not a rewrite". Deliberately low,
111
195
  * because a real mishearing can move word boundaries ("code churn" → "coach
@@ -124,11 +208,27 @@ export const SOUNDS_LIKE_FLOOR = 0.34;
124
208
  * a phrase, essentially never its onset. This is what rejects a rewrite:
125
209
  * "revenue" for "churn" and "monetization" for "agents" both score in the
126
210
  * same range as a true repair, and both fail the onset test.
211
+ *
212
+ * When either side has no Latin letters there is no key to compare, and this
213
+ * used to answer `ka === kb` — `"" === ""`, i.e. YES for any two non-Latin
214
+ * strings however unrelated. That is no gate at all for an Urdu transcript, so
215
+ * those pairs route to `textSimilarity` instead (2026-08-18 field case).
216
+ * Latin-to-Latin comparisons never reach that branch and are unchanged.
127
217
  */
128
218
  export function soundsSimilar(a: string, b: string, floor = SOUNDS_LIKE_FLOOR): boolean {
129
219
  const ka = phraseKey(a);
130
220
  const kb = phraseKey(b);
131
- if (ka.length === 0 || kb.length === 0) return ka === kb;
221
+ if (ka.length === 0 || kb.length === 0) {
222
+ // `floor` is deliberately NOT reused here: it is calibrated against
223
+ // consonant skeletons, which are shorter and coarser than the text this
224
+ // branch compares, so the same number means something else. Taking the
225
+ // larger of the two was tried and is wrong — the default 0.34 alone
226
+ // rejects a measured genuine repair scoring 0.333. The only caller that
227
+ // raises the floor (reconcileCopy, 0.6) cannot reach this branch anyway:
228
+ // its candidate tokens are stripped to `[A-Za-z]`, so its key is never
229
+ // empty and a non-Latin spoken word scores ~0 against it regardless.
230
+ return textSimilarity(a, b) >= TEXT_SIMILARITY_FLOOR;
231
+ }
132
232
  if (ka[0] !== kb[0]) return false;
133
233
  return soundsLike(a, b) >= floor;
134
234
  }
@@ -25,6 +25,7 @@ import {
25
25
  export * from "./provider";
26
26
  export * from "./usage";
27
27
  export * from "./beats";
28
+ export * from "./youtube";
28
29
  export * from "./scene-props";
29
30
  export * from "./repair";
30
31
  export { AnthropicProvider, DEFAULT_CLAUDE_MODEL } from "./anthropic";
@@ -1,7 +1,7 @@
1
1
  import { z } from "zod/v4";
2
2
  import type { Transcript, Word } from "../schema";
3
3
  import type { Scene } from "../scene-schema";
4
- import { soundsSimilar } from "../phonetics";
4
+ import { normalizeForCompare, soundsSimilar } from "../phonetics";
5
5
  import type { LlmProvider } from "./provider";
6
6
 
7
7
  /**
@@ -76,7 +76,11 @@ A correction must sound essentially identical to what was heard. If a span is no
76
76
 
77
77
  For each fix give the word-index span, the exact text you are replacing (\`heard\`), and the corrected text.`;
78
78
 
79
- export function buildRepairUserPrompt(transcript: Transcript, speaker?: string): string {
79
+ export function buildRepairUserPrompt(
80
+ transcript: Transcript,
81
+ speaker?: string,
82
+ dictionary?: readonly string[],
83
+ ): string {
80
84
  const words = transcript.words.map((w, i) => `[${i}]${w.text}`).join(" ");
81
85
  return (
82
86
  (speaker
@@ -89,20 +93,30 @@ export function buildRepairUserPrompt(transcript: Transcript, speaker?: string):
89
93
  `About the speaker (use this to recognise names the recognizer mangled, ` +
90
94
  `never to introduce facts): ${speaker}\n\n`
91
95
  : "") +
96
+ (dictionary && dictionary.length > 0
97
+ ? // The user's dictionary (F4, 2026-08-16): same failure class as the
98
+ // speaker hint — "Jason" for JSON is a lookup once the model knows
99
+ // the term, a guess otherwise.
100
+ `Vouched terms the speaker uses — prefer these spellings when the ` +
101
+ `audio matches: ${dictionary.join(", ")}\n\n`
102
+ : "") +
92
103
  `Word-indexed transcript (indices refer to THIS list):\n${words}\n\n` +
93
104
  `Report only spans that are clearly mishearings.`
94
105
  );
95
106
  }
96
107
 
97
- /** Comparable form: case- and punctuation-insensitive. */
98
- function norm(s: string): string {
99
- return s
100
- .toLowerCase()
101
- .replace(/[^a-z0-9\s]/g, "")
102
- .split(/\s+/)
103
- .filter(Boolean)
104
- .join(" ");
105
- }
108
+ /**
109
+ * Comparable form: case- and punctuation-insensitive, in any script.
110
+ *
111
+ * This was a second, Latin-only copy (`[^a-z0-9\s]`) of what is now
112
+ * `normalizeForCompare`. The two drifted in the worst possible way: on an Urdu
113
+ * transcript every string normalized to "", so `norm(actual) ===
114
+ * norm(r.correction)` was true for every proposal and all 11 repairs in a real
115
+ * run were refused as "identical to what was heard" (2026-08-18) — while
116
+ * `locate()`, comparing "" to "", "matched" the first span it tried without
117
+ * verifying anything. One shared helper so they cannot drift again.
118
+ */
119
+ const norm = normalizeForCompare;
106
120
 
107
121
  function spanText(transcript: Transcript, startWord: number, endWord: number): string {
108
122
  return transcript.words
@@ -171,6 +185,13 @@ export interface ApplyRepairsOptions {
171
185
  * phonetic gate. Everything else about it is still checked.
172
186
  */
173
187
  speaker?: string;
188
+ /**
189
+ * The user's dictionary (F4, 2026-08-16) — vouched the same way the
190
+ * speaker hint is: terms the user typed themselves join the vouched set,
191
+ * so a fully-vouched correction ("Jason" → "JSON") may pass the phonetic
192
+ * gate. Every other guard still applies.
193
+ */
194
+ dictionary?: readonly string[];
174
195
  }
175
196
 
176
197
  export function applyRepairs(
@@ -180,18 +201,23 @@ export function applyRepairs(
180
201
  ): { transcript: Transcript; applied: AppliedRepair[] } {
181
202
  const maxIndex = transcript.words.length - 1;
182
203
  /**
183
- * Names the user vouched for via `--speaker`. A recognizer that turned
184
- * "Ahsan" into the initialism "SM" produces a correction no phonetic
185
- * measure will accept — and refusing it leaves the wrong name in the
186
- * captions, which is the failure the hint exists to prevent. So a correction
187
- * built ENTIRELY from words the user supplied is exempt from that one gate.
188
- * Every other guard still applies, and the exemption can only ever
189
- * substitute text the user typed themselves.
204
+ * Words the user vouched for — via `--speaker` and, since F4 (2026-08-16),
205
+ * via the dictionary. A recognizer that turned "Ahsan" into the initialism
206
+ * "SM" produces a correction no phonetic measure will accept — and refusing
207
+ * it leaves the wrong name in the captions, which is the failure the hint
208
+ * exists to prevent. So a correction built ENTIRELY from words the user
209
+ * supplied is exempt from that one gate. Every other guard still applies,
210
+ * and the exemption can only ever substitute text the user typed themselves.
190
211
  */
191
- const speakerWords = new Set(norm(opts.speaker ?? "").split(" ").filter(Boolean));
192
- const speakerVouched = (correction: string): boolean => {
212
+ const vouchedWords = new Set(
213
+ [
214
+ ...norm(opts.speaker ?? "").split(" "),
215
+ ...(opts.dictionary ?? []).flatMap((term) => norm(term).split(" ")),
216
+ ].filter(Boolean),
217
+ );
218
+ const userVouched = (correction: string): boolean => {
193
219
  const tokens = norm(correction).split(" ").filter(Boolean);
194
- return tokens.length > 0 && tokens.every((t) => speakerWords.has(t));
220
+ return tokens.length > 0 && tokens.every((t) => vouchedWords.has(t));
195
221
  };
196
222
  const results: AppliedRepair[] = [];
197
223
  const accepted: Array<{ startWord: number; endWord: number; tokens: string[] }> = [];
@@ -205,6 +231,12 @@ export function applyRepairs(
205
231
  */
206
232
  const locate = (r: TranscriptRepair): { startWord: number; endWord: number } | null => {
207
233
  const want = norm(r.heard);
234
+ // A quote that normalizes to nothing is not an anchor: "" compares equal
235
+ // to the first span whose own normalisation is empty, so the search would
236
+ // "find" a span it never verified. That was live for every non-Latin
237
+ // transcript until norm() was fixed above; refuse it explicitly so it
238
+ // cannot come back through some other all-punctuation quote.
239
+ if (want.length === 0) return null;
208
240
  // Widths to try, in order of trust. The QUOTED TEXT is the reliable part
209
241
  // of a proposal, so its own token count leads; the claimed span is a
210
242
  // fallback for a quote whose normalisation splits differently. Trusting
@@ -285,7 +317,7 @@ export function applyRepairs(
285
317
  record(`"${r.correction}" is too different in length from "${actual}"`);
286
318
  continue;
287
319
  }
288
- if (!soundsSimilar(actual, r.correction) && !speakerVouched(r.correction)) {
320
+ if (!soundsSimilar(actual, r.correction) && !userVouched(r.correction)) {
289
321
  // The gate that keeps this a repair pass and not a rewrite pass.
290
322
  record(`"${r.correction}" does not sound like "${actual}" — rewrite, not a repair`);
291
323
  continue;
@@ -340,7 +372,7 @@ export async function repairTranscript(
340
372
  try {
341
373
  const result = await provider.complete({
342
374
  system: REPAIR_SYSTEM,
343
- user: buildRepairUserPrompt(transcript, opts.speaker),
375
+ user: buildRepairUserPrompt(transcript, opts.speaker, opts.dictionary),
344
376
  schema: TranscriptRepairSchema,
345
377
  schemaName: "transcript_repair",
346
378
  // EDITORIAL on purpose, despite looking mechanical. Measured on the real