@ossclip/core 0.1.10 → 0.1.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/browser.ts +4 -1
- package/src/captions.ts +20 -0
- package/src/config.ts +14 -0
- package/src/transcribe.ts +134 -5
package/package.json
CHANGED
package/src/browser.ts
CHANGED
|
@@ -19,6 +19,9 @@ export {
|
|
|
19
19
|
type ContentRect,
|
|
20
20
|
type ContentRectSegment,
|
|
21
21
|
} from "./content-rect";
|
|
22
|
-
|
|
22
|
+
// lineDirection is a VALUE export but stays browser-safe: captions.ts
|
|
23
|
+
// imports types only. CaptionTrack needs it at render time (Urdu field test
|
|
24
|
+
// 2026-08-05 — RTL lines were laying out LTR).
|
|
25
|
+
export { lineDirection, type CaptionLine, type CaptionWord } from "./captions";
|
|
23
26
|
export type { KeptSpan } from "./timemap";
|
|
24
27
|
export type { Probe, Production, RenderSettings, Segment, Transcript, Word } from "./schema";
|
package/src/captions.ts
CHANGED
|
@@ -14,6 +14,26 @@ export interface CaptionLine {
|
|
|
14
14
|
end: number;
|
|
15
15
|
}
|
|
16
16
|
|
|
17
|
+
/**
|
|
18
|
+
* Per-LINE direction from the text itself — the first-strong-character
|
|
19
|
+
* heuristic (Unicode UAX #9 rules P2/P3), not the transcript's language
|
|
20
|
+
* code. The Urdu field transcript (2026-08-05) code-switches: lines opening
|
|
21
|
+
* with a Latin loanword ("Fulfillment …") exist alongside pure Urdu lines,
|
|
22
|
+
* and first-strong is the standard resolution for exactly that — the line
|
|
23
|
+
* lays out the way its own leading text reads. Digits and punctuation are
|
|
24
|
+
* bidi-weak/neutral and skipped, so "2026 میں …" still resolves RTL.
|
|
25
|
+
*/
|
|
26
|
+
const STRONG_RTL = /[\p{Script=Arabic}\p{Script=Hebrew}\p{Script=Syriac}\p{Script=Thaana}\p{Script=Nko}]/u;
|
|
27
|
+
const STRONG_LTR = /\p{L}/u; // checked AFTER the RTL scripts, which are also \p{L}
|
|
28
|
+
|
|
29
|
+
export function lineDirection(text: string): "rtl" | "ltr" {
|
|
30
|
+
for (const ch of text) {
|
|
31
|
+
if (STRONG_RTL.test(ch)) return "rtl";
|
|
32
|
+
if (STRONG_LTR.test(ch)) return "ltr";
|
|
33
|
+
}
|
|
34
|
+
return "ltr";
|
|
35
|
+
}
|
|
36
|
+
|
|
17
37
|
export interface CaptionOptions {
|
|
18
38
|
maxWordsPerLine?: number;
|
|
19
39
|
maxLineDuration?: number;
|
package/src/config.ts
CHANGED
|
@@ -30,6 +30,14 @@ export interface OssclipConfig {
|
|
|
30
30
|
* user picks one of its "stop asking" answers.
|
|
31
31
|
*/
|
|
32
32
|
openEditorAfterProduce?: OpenEditorPref;
|
|
33
|
+
/**
|
|
34
|
+
* Opt-in "made with ossclip" wordmark on every produce run, so voluntary
|
|
35
|
+
* attribution is a one-time config write instead of a flag remembered per
|
|
36
|
+
* run. DEFAULT OFF for everyone — a forced watermark on an open-source
|
|
37
|
+
* tool reads as a free-tier limitation, and this is a credit, not one.
|
|
38
|
+
* `--watermark` / `--no-watermark` win over this per run.
|
|
39
|
+
*/
|
|
40
|
+
watermark?: boolean;
|
|
33
41
|
browserExecutable?: string;
|
|
34
42
|
/**
|
|
35
43
|
* USD per million tokens, keyed by model id or family substring — overrides
|
|
@@ -108,6 +116,12 @@ export function loadConfig(): OssclipConfig {
|
|
|
108
116
|
openEditorAfterProduce: (process.env.OSSCLIP_OPEN_EDITOR ??
|
|
109
117
|
fileCfg.openEditorAfterProduce) as OpenEditorPref | undefined,
|
|
110
118
|
browserExecutable: process.env.OSSCLIP_BROWSER ?? fileCfg.browserExecutable,
|
|
119
|
+
// File-only, like `pricing`: an env spelling would arrive as a string,
|
|
120
|
+
// and "false" is truthy — parse-don't-coerce says no such trap. The
|
|
121
|
+
// strict `=== true` check lives at the consumer (produce's
|
|
122
|
+
// resolveWatermark), so a hand-edited non-boolean stays OFF, the safe
|
|
123
|
+
// default for a credit.
|
|
124
|
+
watermark: fileCfg.watermark,
|
|
111
125
|
pricing: fileCfg.pricing,
|
|
112
126
|
};
|
|
113
127
|
}
|
package/src/transcribe.ts
CHANGED
|
@@ -13,6 +13,96 @@ export interface WhisperJson {
|
|
|
13
13
|
|
|
14
14
|
const NOISE_TOKEN = /^[[(].*[\])]$/; // [BLANK_AUDIO], (buzzing), [MUSIC] …
|
|
15
15
|
|
|
16
|
+
/**
|
|
17
|
+
* How many bytes at the END of `bytes` form the start of a multi-byte UTF-8
|
|
18
|
+
* character whose continuation bytes are missing (§130: whisper.cpp `-ml 1`
|
|
19
|
+
* splits byte-level BPE tokens mid-character, so a segment's text can end on
|
|
20
|
+
* a bare lead byte — the field file ended one segment `0x20 0xD9` and started
|
|
21
|
+
* the next `0xB9 …`, the two halves of ٹ). 0 when the tail is complete.
|
|
22
|
+
*/
|
|
23
|
+
function utf8DanglingTailLen(bytes: Buffer): number {
|
|
24
|
+
let i = bytes.length - 1;
|
|
25
|
+
let cont = 0;
|
|
26
|
+
while (i >= 0 && cont < 3 && (bytes[i]! & 0xc0) === 0x80) {
|
|
27
|
+
i--;
|
|
28
|
+
cont++;
|
|
29
|
+
}
|
|
30
|
+
// All continuation bytes: a HEAD fragment, someone else's tail — not ours.
|
|
31
|
+
if (i < 0) return 0;
|
|
32
|
+
const lead = bytes[i]!;
|
|
33
|
+
const need = lead >= 0xf0 ? 4 : lead >= 0xe0 ? 3 : lead >= 0xc0 ? 2 : 1;
|
|
34
|
+
// ASCII (or a stray byte no neighbor could complete) — nothing dangling.
|
|
35
|
+
if (need === 1) return 0;
|
|
36
|
+
return cont < need - 1 ? bytes.length - i : 0;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Repair segments whose text was split MID-CHARACTER by byte-level BPE
|
|
41
|
+
* (§130). Operates in byte space: `text` values here are latin1-decoded, one
|
|
42
|
+
* code point per original byte, because the fix must see the real bytes —
|
|
43
|
+
* reading the file as utf8 first would already have destroyed them (Node
|
|
44
|
+
* substitutes U+FFFD, and the two halves of the character are gone for good).
|
|
45
|
+
* A segment ending on an incomplete sequence merges with a following segment
|
|
46
|
+
* that begins with continuation bytes, spanning both segments' offsets; the
|
|
47
|
+
* loop re-checks the merged tail so a 3–4 byte character split across three
|
|
48
|
+
* segments still heals. Only that exact shape merges — a dangling tail whose
|
|
49
|
+
* neighbor does NOT continue it is unrecoverable and left to decode to
|
|
50
|
+
* U+FFFD, which parseWhisperJson then folds into a neighboring word.
|
|
51
|
+
*/
|
|
52
|
+
function repairSplitSegments(json: WhisperJson): WhisperJson {
|
|
53
|
+
const out: WhisperJson["transcription"] = [];
|
|
54
|
+
for (const seg of json.transcription ?? []) {
|
|
55
|
+
const prev = out[out.length - 1];
|
|
56
|
+
const bytes = Buffer.from(seg.text, "latin1");
|
|
57
|
+
if (
|
|
58
|
+
prev &&
|
|
59
|
+
bytes.length > 0 &&
|
|
60
|
+
(bytes[0]! & 0xc0) === 0x80 &&
|
|
61
|
+
utf8DanglingTailLen(Buffer.from(prev.text, "latin1")) > 0
|
|
62
|
+
) {
|
|
63
|
+
prev.text += seg.text;
|
|
64
|
+
prev.offsets = {
|
|
65
|
+
from: Math.min(prev.offsets.from, seg.offsets.from),
|
|
66
|
+
to: Math.max(prev.offsets.to, seg.offsets.to),
|
|
67
|
+
};
|
|
68
|
+
} else {
|
|
69
|
+
out.push({ ...seg, offsets: { ...seg.offsets } });
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
return {
|
|
73
|
+
...json,
|
|
74
|
+
transcription: out.map((s) => ({
|
|
75
|
+
...s,
|
|
76
|
+
text: Buffer.from(s.text, "latin1").toString("utf8"),
|
|
77
|
+
})),
|
|
78
|
+
};
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
const STRICT_UTF8 = new TextDecoder("utf-8", { fatal: true });
|
|
82
|
+
|
|
83
|
+
/**
|
|
84
|
+
* Parse whisper.cpp's `-oj` output from its raw BYTES. The file is not
|
|
85
|
+
* guaranteed to be valid UTF-8 (§130): with `-ml 1` a multi-byte character
|
|
86
|
+
* can be split across two segments' text fields at the byte level, and a
|
|
87
|
+
* plain `readFile(path, "utf8")` silently replaces both halves with U+FFFD —
|
|
88
|
+
* which is exactly the `��اپک` that shipped into the Urdu field captions
|
|
89
|
+
* (Urdu field test 2026-08-05). Valid files take the strict-decode path and
|
|
90
|
+
* behave byte-identically to before; invalid ones round-trip through latin1
|
|
91
|
+
* (byte-transparent, and UTF-8 continuation bytes are ≥0x80 so JSON's ASCII
|
|
92
|
+
* structure is untouched) so the split can be healed with the bytes intact.
|
|
93
|
+
*/
|
|
94
|
+
export function parseWhisperOutput(raw: Buffer): Transcript {
|
|
95
|
+
let text: string | null = null;
|
|
96
|
+
try {
|
|
97
|
+
text = STRICT_UTF8.decode(raw);
|
|
98
|
+
} catch {
|
|
99
|
+
// Invalid UTF-8: the token-split shape. Fall through to byte repair —
|
|
100
|
+
// but only for DECODE failures; JSON syntax errors below still throw.
|
|
101
|
+
}
|
|
102
|
+
if (text !== null) return parseWhisperJson(JSON.parse(text) as WhisperJson);
|
|
103
|
+
return parseWhisperJson(repairSplitSegments(JSON.parse(raw.toString("latin1")) as WhisperJson));
|
|
104
|
+
}
|
|
105
|
+
|
|
16
106
|
/**
|
|
17
107
|
* Convert whisper.cpp `-ml 1` segments (≈ one token each) into words.
|
|
18
108
|
* Tokens beginning with whitespace start a new word; bare continuations
|
|
@@ -36,6 +126,26 @@ export function parseWhisperJson(json: WhisperJson): Transcript {
|
|
|
36
126
|
words.push({ text, start, end });
|
|
37
127
|
}
|
|
38
128
|
}
|
|
129
|
+
// U+FFFD is never displayable speech — it is a byte the split repair could
|
|
130
|
+
// not heal (§130: the field file also holds a lone lead byte whose
|
|
131
|
+
// continuation whisper never emitted at all, mid-word between "ی" and "ج"),
|
|
132
|
+
// and shipping it paints a literal � caption. Strip it from mixed words; a
|
|
133
|
+
// word left EMPTY by the strip folds its time span into a neighbor instead
|
|
134
|
+
// of vanishing — the span still belongs to speech.
|
|
135
|
+
for (let i = words.length - 1; i >= 0; i--) {
|
|
136
|
+
const w = words[i]!;
|
|
137
|
+
if (!w.text.includes("�")) continue;
|
|
138
|
+
const cleaned = w.text.replaceAll("�", "");
|
|
139
|
+
if (cleaned) {
|
|
140
|
+
w.text = cleaned;
|
|
141
|
+
continue;
|
|
142
|
+
}
|
|
143
|
+
const prev = words[i - 1];
|
|
144
|
+
const next = words[i + 1];
|
|
145
|
+
if (prev) prev.end = Math.max(prev.end, w.end);
|
|
146
|
+
else if (next) next.start = Math.min(next.start, w.start);
|
|
147
|
+
words.splice(i, 1);
|
|
148
|
+
}
|
|
39
149
|
// Whisper occasionally emits zero-length or inverted stamps; repair minimally.
|
|
40
150
|
for (let i = 0; i < words.length; i++) {
|
|
41
151
|
const w = words[i]!;
|
|
@@ -51,17 +161,36 @@ export interface WhisperOptions {
|
|
|
51
161
|
modelPath: string;
|
|
52
162
|
/** Output base path; whisper writes `${outBase}.json`. */
|
|
53
163
|
outBase: string;
|
|
164
|
+
/**
|
|
165
|
+
* Language code passed as `-l` (e.g. "ur", "de", "auto"). whisper.cpp
|
|
166
|
+
* defaults to English when the flag is absent, which decodes garbage out of
|
|
167
|
+
* a non-English fine-tune (Urdu field test 2026-08-05: ggml-medium-urdu
|
|
168
|
+
* needed `-l ur` to emit Urdu script at all). Left unset, the spawned args
|
|
169
|
+
* stay byte-identical to what English-suffixed models always got.
|
|
170
|
+
*/
|
|
171
|
+
language?: string;
|
|
54
172
|
}
|
|
55
173
|
|
|
56
|
-
|
|
57
|
-
|
|
174
|
+
/**
|
|
175
|
+
* Pure arg construction, split from the spawn the same way openCommand() is
|
|
176
|
+
* split from openInBrowser(): the `-l` conditional is exactly the kind of
|
|
177
|
+
* branch that must be testable without a whisper binary on the box.
|
|
178
|
+
*/
|
|
179
|
+
export function whisperArgs(opts: WhisperOptions, wavPath: string): string[] {
|
|
180
|
+
const args = [
|
|
58
181
|
"-m", opts.modelPath,
|
|
59
182
|
"-f", wavPath,
|
|
60
183
|
"-oj",
|
|
61
184
|
"-of", opts.outBase,
|
|
62
185
|
"-ml", "1",
|
|
63
186
|
"--no-prints",
|
|
64
|
-
]
|
|
65
|
-
|
|
66
|
-
return
|
|
187
|
+
];
|
|
188
|
+
if (opts.language !== undefined) args.push("-l", opts.language);
|
|
189
|
+
return args;
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
export async function runWhisper(opts: WhisperOptions, wavPath: string): Promise<Transcript> {
|
|
193
|
+
await run(opts.whisperPath, whisperArgs(opts, wavPath));
|
|
194
|
+
// Bytes, not "utf8": the utf8 read is where the §130 split characters died.
|
|
195
|
+
return parseWhisperOutput(await readFile(`${opts.outBase}.json`));
|
|
67
196
|
}
|