@ossclip/core 0.1.24 → 0.1.26
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/fonts/NotoNastaliqUrdu-Bold.ttf +0 -0
- package/assets/fonts/OFL.txt +93 -0
- package/assets/fonts/README.md +15 -0
- package/package.json +2 -1
- package/src/blooper.ts +91 -8
- package/src/browser.ts +8 -0
- package/src/captions.ts +45 -2
- package/src/concat.ts +95 -8
- package/src/config.ts +110 -0
- package/src/content-rect-detect.ts +16 -4
- package/src/content-rect.ts +211 -0
- package/src/cover.ts +21 -5
- package/src/cutlist.ts +38 -6
- package/src/dictionary.ts +56 -0
- package/src/export-premiere-project.ts +26 -9
- package/src/fonts.ts +17 -0
- package/src/index.ts +3 -0
- package/src/ingest.ts +88 -2
- package/src/normalize.ts +273 -127
- package/src/overrides.ts +728 -0
- package/src/phonetics.ts +101 -1
- package/src/producer/index.ts +1 -0
- package/src/producer/repair.ts +55 -23
- package/src/producer/youtube.ts +434 -0
- package/src/recut.ts +61 -0
- package/src/retake.ts +104 -2
- package/src/scene-schema.ts +10 -0
- package/src/thumbnail.ts +412 -0
- package/src/transcribe.ts +93 -5
- package/src/zoom.ts +63 -12
package/src/phonetics.ts
CHANGED
|
@@ -12,6 +12,49 @@
|
|
|
12
12
|
* "unrelated", and it must stay dependency-free and deterministic.
|
|
13
13
|
*/
|
|
14
14
|
|
|
15
|
+
/**
|
|
16
|
+
* Marks Arabic-script text carries that two transcribers disagree about
|
|
17
|
+
* without disagreeing about the WORD: harakat/vowel diacritics
|
|
18
|
+
* (U+064B–U+065F, U+0670), tatweel (U+0640, a pure typographic stretch), and
|
|
19
|
+
* the zero-width joiners (U+200C/U+200D). Whisper emits them inconsistently
|
|
20
|
+
* and an LLM writing a correction rarely reproduces them, so leaving them in
|
|
21
|
+
* makes a correct repair either miss `locate()` outright or read as
|
|
22
|
+
* "different from what was heard" purely on invisible marks. Only tatweel
|
|
23
|
+
* survives the `\p{L}\p{N}` filter below (it is Lm, a letter); the rest are
|
|
24
|
+
* stripped here so the intent is legible rather than an accident of Unicode
|
|
25
|
+
* categories.
|
|
26
|
+
*/
|
|
27
|
+
// Escaped, not literal: three of these code points are invisible in an editor.
|
|
28
|
+
const ARABIC_NOISE = /[\u064B-\u065F\u0670\u0640\u200C\u200D]/g;
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Comparable form for two pieces of text: case-, punctuation- and
|
|
32
|
+
* whitespace-insensitive, in ANY script.
|
|
33
|
+
*
|
|
34
|
+
* Shared by `phonetics.ts` and `producer/repair.ts` on purpose — they used to
|
|
35
|
+
* hold two copies and the copy in `repair.ts` was `[^a-z0-9\s]`, i.e.
|
|
36
|
+
* Latin-only. Field case (2026-08-18): every one of 11 recorded Urdu repairs
|
|
37
|
+
* normalized to the empty string, so `norm(heard) === norm(correction)` was
|
|
38
|
+
* `"" === ""` and ALL 11 were refused as "identical to what was heard" —
|
|
39
|
+
* including `پرسٹ` → `فرسٹ`, which shares no letters with what it replaced.
|
|
40
|
+
*
|
|
41
|
+
* Keeping letters and digits of every script also KEEPS accented Latin
|
|
42
|
+
* ("café", "über") where the old expression deleted it. That is the same bug
|
|
43
|
+
* in miniature — a French word normalized to "caf" — so it is a fix, not a
|
|
44
|
+
* regression. Pure-ASCII input is byte-identical to the old behaviour, which
|
|
45
|
+
* is pinned by a test.
|
|
46
|
+
*/
|
|
47
|
+
export function normalizeForCompare(s: string): string {
|
|
48
|
+
return s
|
|
49
|
+
.normalize("NFC")
|
|
50
|
+
.toLowerCase()
|
|
51
|
+
.replace(ARABIC_NOISE, "")
|
|
52
|
+
.replace(/[^\p{L}\p{N}\s]/gu, "")
|
|
53
|
+
.split(/\s+/)
|
|
54
|
+
.filter(Boolean)
|
|
55
|
+
.join(" ");
|
|
56
|
+
}
|
|
57
|
+
|
|
15
58
|
/**
|
|
16
59
|
* Digraphs collapsed before single letters, longest first. The ch/sh and th
|
|
17
60
|
* sounds get DIGIT placeholders on purpose: a letter placeholder would be
|
|
@@ -106,6 +149,47 @@ export function soundsLike(a: string, b: string): number {
|
|
|
106
149
|
return Math.max(0, 1 - dist / Math.max(ka.length, kb.length));
|
|
107
150
|
}
|
|
108
151
|
|
|
152
|
+
/**
|
|
153
|
+
* 0..1 similarity of the TEXT itself, for scripts the phonetic key cannot
|
|
154
|
+
* represent (`phoneticKey` is defined over a-z, so anything non-Latin keys to
|
|
155
|
+
* ""). Same shape as `soundsLike` — normalized edit distance over the longer
|
|
156
|
+
* string — but run on the normalized text rather than a consonant skeleton.
|
|
157
|
+
*
|
|
158
|
+
* This is a weaker signal than a phonetic key and it is meant to be: an
|
|
159
|
+
* Urdu-script mishearing differs from the truth by a letter or two of the same
|
|
160
|
+
* script, so edit distance still separates it from an unrelated phrase. What
|
|
161
|
+
* it cannot do is fold vowels, which is why the floor below is calibrated
|
|
162
|
+
* against real data instead of borrowing SOUNDS_LIKE_FLOOR.
|
|
163
|
+
*/
|
|
164
|
+
export function textSimilarity(a: string, b: string): number {
|
|
165
|
+
const na = normalizeForCompare(a);
|
|
166
|
+
const nb = normalizeForCompare(b);
|
|
167
|
+
if (na.length === 0 && nb.length === 0) return 1;
|
|
168
|
+
if (na.length === 0 || nb.length === 0) return 0;
|
|
169
|
+
return Math.max(0, 1 - levenshtein(na, nb) / Math.max(na.length, nb.length));
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Floor for the non-Latin fallback, MEASURED rather than guessed.
|
|
174
|
+
*
|
|
175
|
+
* The 11 Urdu repairs recorded in a real production.json (2026-08-18) score,
|
|
176
|
+
* sorted: 0.333, 0.400, 0.500, 0.500, 0.545, 0.583, 0.636, 0.750, 0.750,
|
|
177
|
+
* 0.800, 0.800. The 0.333 is `حقیقہ ٹون` → `ہیکاتھون` ("hackathon"), a genuine
|
|
178
|
+
* repair and the worst of the set because the recognizer both re-segmented the
|
|
179
|
+
* word and changed its opening letter. Admitting it sets the ceiling on the
|
|
180
|
+
* floor; 0.33 is the largest value that does.
|
|
181
|
+
*
|
|
182
|
+
* Against that, unrelated four-word spans lifted from the same transcript
|
|
183
|
+
* score 0.167–0.250 and are refused. The band is narrow, and it is narrow for
|
|
184
|
+
* the same reason the Latin one is (see SOUNDS_LIKE_FLOOR): two SHORT
|
|
185
|
+
* unrelated Urdu spans can still land above it — measured, `پرسٹ ہیک` vs
|
|
186
|
+
* `ٹرس می` scores 0.500. There is no onset test here to catch that, because
|
|
187
|
+
* two of the 11 genuine repairs change their first letter. So this gate is
|
|
188
|
+
* real but shallow; the span, token-count and length guards in
|
|
189
|
+
* `applyRepairs` are what keep it from being a rewrite licence.
|
|
190
|
+
*/
|
|
191
|
+
export const TEXT_SIMILARITY_FLOOR = 0.33;
|
|
192
|
+
|
|
109
193
|
/**
|
|
110
194
|
* Default floor for "this is a repair, not a rewrite". Deliberately low,
|
|
111
195
|
* because a real mishearing can move word boundaries ("code churn" → "coach
|
|
@@ -124,11 +208,27 @@ export const SOUNDS_LIKE_FLOOR = 0.34;
|
|
|
124
208
|
* a phrase, essentially never its onset. This is what rejects a rewrite:
|
|
125
209
|
* "revenue" for "churn" and "monetization" for "agents" both score in the
|
|
126
210
|
* same range as a true repair, and both fail the onset test.
|
|
211
|
+
*
|
|
212
|
+
* When either side has no Latin letters there is no key to compare, and this
|
|
213
|
+
* used to answer `ka === kb` — `"" === ""`, i.e. YES for any two non-Latin
|
|
214
|
+
* strings however unrelated. That is no gate at all for an Urdu transcript, so
|
|
215
|
+
* those pairs route to `textSimilarity` instead (2026-08-18 field case).
|
|
216
|
+
* Latin-to-Latin comparisons never reach that branch and are unchanged.
|
|
127
217
|
*/
|
|
128
218
|
export function soundsSimilar(a: string, b: string, floor = SOUNDS_LIKE_FLOOR): boolean {
|
|
129
219
|
const ka = phraseKey(a);
|
|
130
220
|
const kb = phraseKey(b);
|
|
131
|
-
if (ka.length === 0 || kb.length === 0)
|
|
221
|
+
if (ka.length === 0 || kb.length === 0) {
|
|
222
|
+
// `floor` is deliberately NOT reused here: it is calibrated against
|
|
223
|
+
// consonant skeletons, which are shorter and coarser than the text this
|
|
224
|
+
// branch compares, so the same number means something else. Taking the
|
|
225
|
+
// larger of the two was tried and is wrong — the default 0.34 alone
|
|
226
|
+
// rejects a measured genuine repair scoring 0.333. The only caller that
|
|
227
|
+
// raises the floor (reconcileCopy, 0.6) cannot reach this branch anyway:
|
|
228
|
+
// its candidate tokens are stripped to `[A-Za-z]`, so its key is never
|
|
229
|
+
// empty and a non-Latin spoken word scores ~0 against it regardless.
|
|
230
|
+
return textSimilarity(a, b) >= TEXT_SIMILARITY_FLOOR;
|
|
231
|
+
}
|
|
132
232
|
if (ka[0] !== kb[0]) return false;
|
|
133
233
|
return soundsLike(a, b) >= floor;
|
|
134
234
|
}
|
package/src/producer/index.ts
CHANGED
|
@@ -25,6 +25,7 @@ import {
|
|
|
25
25
|
export * from "./provider";
|
|
26
26
|
export * from "./usage";
|
|
27
27
|
export * from "./beats";
|
|
28
|
+
export * from "./youtube";
|
|
28
29
|
export * from "./scene-props";
|
|
29
30
|
export * from "./repair";
|
|
30
31
|
export { AnthropicProvider, DEFAULT_CLAUDE_MODEL } from "./anthropic";
|
package/src/producer/repair.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { z } from "zod/v4";
|
|
2
2
|
import type { Transcript, Word } from "../schema";
|
|
3
3
|
import type { Scene } from "../scene-schema";
|
|
4
|
-
import { soundsSimilar } from "../phonetics";
|
|
4
|
+
import { normalizeForCompare, soundsSimilar } from "../phonetics";
|
|
5
5
|
import type { LlmProvider } from "./provider";
|
|
6
6
|
|
|
7
7
|
/**
|
|
@@ -76,7 +76,11 @@ A correction must sound essentially identical to what was heard. If a span is no
|
|
|
76
76
|
|
|
77
77
|
For each fix give the word-index span, the exact text you are replacing (\`heard\`), and the corrected text.`;
|
|
78
78
|
|
|
79
|
-
export function buildRepairUserPrompt(
|
|
79
|
+
export function buildRepairUserPrompt(
|
|
80
|
+
transcript: Transcript,
|
|
81
|
+
speaker?: string,
|
|
82
|
+
dictionary?: readonly string[],
|
|
83
|
+
): string {
|
|
80
84
|
const words = transcript.words.map((w, i) => `[${i}]${w.text}`).join(" ");
|
|
81
85
|
return (
|
|
82
86
|
(speaker
|
|
@@ -89,20 +93,30 @@ export function buildRepairUserPrompt(transcript: Transcript, speaker?: string):
|
|
|
89
93
|
`About the speaker (use this to recognise names the recognizer mangled, ` +
|
|
90
94
|
`never to introduce facts): ${speaker}\n\n`
|
|
91
95
|
: "") +
|
|
96
|
+
(dictionary && dictionary.length > 0
|
|
97
|
+
? // The user's dictionary (F4, 2026-08-16): same failure class as the
|
|
98
|
+
// speaker hint — "Jason" for JSON is a lookup once the model knows
|
|
99
|
+
// the term, a guess otherwise.
|
|
100
|
+
`Vouched terms the speaker uses — prefer these spellings when the ` +
|
|
101
|
+
`audio matches: ${dictionary.join(", ")}\n\n`
|
|
102
|
+
: "") +
|
|
92
103
|
`Word-indexed transcript (indices refer to THIS list):\n${words}\n\n` +
|
|
93
104
|
`Report only spans that are clearly mishearings.`
|
|
94
105
|
);
|
|
95
106
|
}
|
|
96
107
|
|
|
97
|
-
/**
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
108
|
+
/**
|
|
109
|
+
* Comparable form: case- and punctuation-insensitive, in any script.
|
|
110
|
+
*
|
|
111
|
+
* This was a second, Latin-only copy (`[^a-z0-9\s]`) of what is now
|
|
112
|
+
* `normalizeForCompare`. The two drifted in the worst possible way: on an Urdu
|
|
113
|
+
* transcript every string normalized to "", so `norm(actual) ===
|
|
114
|
+
* norm(r.correction)` was true for every proposal and all 11 repairs in a real
|
|
115
|
+
* run were refused as "identical to what was heard" (2026-08-18) — while
|
|
116
|
+
* `locate()`, comparing "" to "", "matched" the first span it tried without
|
|
117
|
+
* verifying anything. One shared helper so they cannot drift again.
|
|
118
|
+
*/
|
|
119
|
+
const norm = normalizeForCompare;
|
|
106
120
|
|
|
107
121
|
function spanText(transcript: Transcript, startWord: number, endWord: number): string {
|
|
108
122
|
return transcript.words
|
|
@@ -171,6 +185,13 @@ export interface ApplyRepairsOptions {
|
|
|
171
185
|
* phonetic gate. Everything else about it is still checked.
|
|
172
186
|
*/
|
|
173
187
|
speaker?: string;
|
|
188
|
+
/**
|
|
189
|
+
* The user's dictionary (F4, 2026-08-16) — vouched the same way the
|
|
190
|
+
* speaker hint is: terms the user typed themselves join the vouched set,
|
|
191
|
+
* so a fully-vouched correction ("Jason" → "JSON") may pass the phonetic
|
|
192
|
+
* gate. Every other guard still applies.
|
|
193
|
+
*/
|
|
194
|
+
dictionary?: readonly string[];
|
|
174
195
|
}
|
|
175
196
|
|
|
176
197
|
export function applyRepairs(
|
|
@@ -180,18 +201,23 @@ export function applyRepairs(
|
|
|
180
201
|
): { transcript: Transcript; applied: AppliedRepair[] } {
|
|
181
202
|
const maxIndex = transcript.words.length - 1;
|
|
182
203
|
/**
|
|
183
|
-
*
|
|
184
|
-
* "Ahsan" into the initialism
|
|
185
|
-
* measure will accept — and refusing
|
|
186
|
-
* captions, which is the failure the hint
|
|
187
|
-
*
|
|
188
|
-
* Every other guard still applies,
|
|
189
|
-
* substitute text the user typed themselves.
|
|
204
|
+
* Words the user vouched for — via `--speaker` and, since F4 (2026-08-16),
|
|
205
|
+
* via the dictionary. A recognizer that turned "Ahsan" into the initialism
|
|
206
|
+
* "SM" produces a correction no phonetic measure will accept — and refusing
|
|
207
|
+
* it leaves the wrong name in the captions, which is the failure the hint
|
|
208
|
+
* exists to prevent. So a correction built ENTIRELY from words the user
|
|
209
|
+
* supplied is exempt from that one gate. Every other guard still applies,
|
|
210
|
+
* and the exemption can only ever substitute text the user typed themselves.
|
|
190
211
|
*/
|
|
191
|
-
const
|
|
192
|
-
|
|
212
|
+
const vouchedWords = new Set(
|
|
213
|
+
[
|
|
214
|
+
...norm(opts.speaker ?? "").split(" "),
|
|
215
|
+
...(opts.dictionary ?? []).flatMap((term) => norm(term).split(" ")),
|
|
216
|
+
].filter(Boolean),
|
|
217
|
+
);
|
|
218
|
+
const userVouched = (correction: string): boolean => {
|
|
193
219
|
const tokens = norm(correction).split(" ").filter(Boolean);
|
|
194
|
-
return tokens.length > 0 && tokens.every((t) =>
|
|
220
|
+
return tokens.length > 0 && tokens.every((t) => vouchedWords.has(t));
|
|
195
221
|
};
|
|
196
222
|
const results: AppliedRepair[] = [];
|
|
197
223
|
const accepted: Array<{ startWord: number; endWord: number; tokens: string[] }> = [];
|
|
@@ -205,6 +231,12 @@ export function applyRepairs(
|
|
|
205
231
|
*/
|
|
206
232
|
const locate = (r: TranscriptRepair): { startWord: number; endWord: number } | null => {
|
|
207
233
|
const want = norm(r.heard);
|
|
234
|
+
// A quote that normalizes to nothing is not an anchor: "" compares equal
|
|
235
|
+
// to the first span whose own normalisation is empty, so the search would
|
|
236
|
+
// "find" a span it never verified. That was live for every non-Latin
|
|
237
|
+
// transcript until norm() was fixed above; refuse it explicitly so it
|
|
238
|
+
// cannot come back through some other all-punctuation quote.
|
|
239
|
+
if (want.length === 0) return null;
|
|
208
240
|
// Widths to try, in order of trust. The QUOTED TEXT is the reliable part
|
|
209
241
|
// of a proposal, so its own token count leads; the claimed span is a
|
|
210
242
|
// fallback for a quote whose normalisation splits differently. Trusting
|
|
@@ -285,7 +317,7 @@ export function applyRepairs(
|
|
|
285
317
|
record(`"${r.correction}" is too different in length from "${actual}"`);
|
|
286
318
|
continue;
|
|
287
319
|
}
|
|
288
|
-
if (!soundsSimilar(actual, r.correction) && !
|
|
320
|
+
if (!soundsSimilar(actual, r.correction) && !userVouched(r.correction)) {
|
|
289
321
|
// The gate that keeps this a repair pass and not a rewrite pass.
|
|
290
322
|
record(`"${r.correction}" does not sound like "${actual}" — rewrite, not a repair`);
|
|
291
323
|
continue;
|
|
@@ -340,7 +372,7 @@ export async function repairTranscript(
|
|
|
340
372
|
try {
|
|
341
373
|
const result = await provider.complete({
|
|
342
374
|
system: REPAIR_SYSTEM,
|
|
343
|
-
user: buildRepairUserPrompt(transcript, opts.speaker),
|
|
375
|
+
user: buildRepairUserPrompt(transcript, opts.speaker, opts.dictionary),
|
|
344
376
|
schema: TranscriptRepairSchema,
|
|
345
377
|
schemaName: "transcript_repair",
|
|
346
378
|
// EDITORIAL on purpose, despite looking mechanical. Measured on the real
|