@ossclip/core 0.1.11 → 0.1.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ossclip/core",
3
- "version": "0.1.11",
3
+ "version": "0.1.13",
4
4
  "description": "ossclip's framework-free pipeline: schema, transcription, analysis, cutlist, captions, framing, and the LLM producer",
5
5
  "type": "module",
6
6
  "license": "MIT",
package/src/browser.ts CHANGED
@@ -19,6 +19,9 @@ export {
19
19
  type ContentRect,
20
20
  type ContentRectSegment,
21
21
  } from "./content-rect";
22
- export type { CaptionLine, CaptionWord } from "./captions";
22
+ // lineDirection is a VALUE export but stays browser-safe: captions.ts
23
+ // imports types only. CaptionTrack needs it at render time (Urdu field test
24
+ // 2026-08-05 — RTL lines were laying out LTR).
25
+ export { lineDirection, type CaptionLine, type CaptionWord } from "./captions";
23
26
  export type { KeptSpan } from "./timemap";
24
27
  export type { Probe, Production, RenderSettings, Segment, Transcript, Word } from "./schema";
package/src/captions.ts CHANGED
@@ -14,6 +14,26 @@ export interface CaptionLine {
14
14
  end: number;
15
15
  }
16
16
 
17
+ /**
18
+ * Per-LINE direction from the text itself — the first-strong-character
19
+ * heuristic (Unicode UAX #9 rules P2/P3), not the transcript's language
20
+ * code. The Urdu field transcript (2026-08-05) code-switches: lines opening
21
+ * with a Latin loanword ("Fulfillment …") exist alongside pure Urdu lines,
22
+ * and first-strong is the standard resolution for exactly that — the line
23
+ * lays out the way its own leading text reads. Digits and punctuation are
24
+ * bidi-weak/neutral and skipped, so "2026 میں …" still resolves RTL.
25
+ */
26
+ const STRONG_RTL = /[\p{Script=Arabic}\p{Script=Hebrew}\p{Script=Syriac}\p{Script=Thaana}\p{Script=Nko}]/u;
27
+ const STRONG_LTR = /\p{L}/u; // checked AFTER the RTL scripts, which are also \p{L}
28
+
29
+ export function lineDirection(text: string): "rtl" | "ltr" {
30
+ for (const ch of text) {
31
+ if (STRONG_RTL.test(ch)) return "rtl";
32
+ if (STRONG_LTR.test(ch)) return "ltr";
33
+ }
34
+ return "ltr";
35
+ }
36
+
17
37
  export interface CaptionOptions {
18
38
  maxWordsPerLine?: number;
19
39
  maxLineDuration?: number;
package/src/config.ts CHANGED
@@ -30,6 +30,14 @@ export interface OssclipConfig {
30
30
  * user picks one of its "stop asking" answers.
31
31
  */
32
32
  openEditorAfterProduce?: OpenEditorPref;
33
+ /**
34
+ * Opt-in "made with ossclip" wordmark on every produce run, so voluntary
35
+ * attribution is a one-time config write instead of a flag remembered per
36
+ * run. DEFAULT OFF for everyone — a forced watermark on an open-source
37
+ * tool reads as a free-tier limitation, and this is a credit, not one.
38
+ * `--watermark` / `--no-watermark` win over this per run.
39
+ */
40
+ watermark?: boolean;
33
41
  browserExecutable?: string;
34
42
  /**
35
43
  * USD per million tokens, keyed by model id or family substring — overrides
@@ -108,6 +116,12 @@ export function loadConfig(): OssclipConfig {
108
116
  openEditorAfterProduce: (process.env.OSSCLIP_OPEN_EDITOR ??
109
117
  fileCfg.openEditorAfterProduce) as OpenEditorPref | undefined,
110
118
  browserExecutable: process.env.OSSCLIP_BROWSER ?? fileCfg.browserExecutable,
119
+ // File-only, like `pricing`: an env spelling would arrive as a string,
120
+ // and "false" is truthy — parse-don't-coerce says no such trap. The
121
+ // strict `=== true` check lives at the consumer (produce's
122
+ // resolveWatermark), so a hand-edited non-boolean stays OFF, the safe
123
+ // default for a credit.
124
+ watermark: fileCfg.watermark,
111
125
  pricing: fileCfg.pricing,
112
126
  };
113
127
  }
package/src/cover.ts CHANGED
@@ -59,6 +59,27 @@ export function coverHeadline(text: string, maxWords = COVER_MAX_WORDS): string
59
59
  return out.join(" ").replace(/[,;:—–-]+$/, "");
60
60
  }
61
61
 
62
+ /**
63
+ * What the cover step should emit.
64
+ *
65
+ * "textless" exists because of the Urdu field run 2026-08-05: a run without
66
+ * `--produce` has no LLM hook text, and the old behavior skipped the cover
67
+ * entirely — but frame selection needs no text at all, and the
68
+ * sharpness-scored face frame is a clean thumbnail on its own. Only the
69
+ * banner needs a headline; the pick does not, so a missing headline demotes
70
+ * the cover to a bare frame instead of erasing it.
71
+ */
72
+ export type CoverDecision = "banner" | "textless" | "none";
73
+
74
+ /**
75
+ * Pure so the three-way outcome is testable without a video, ffmpeg, or a
76
+ * beat sheet on disk — the same I/O split as `openCommand()`.
77
+ */
78
+ export function coverDecision(coverEnabled: boolean, headline: string): CoverDecision {
79
+ if (!coverEnabled) return "none";
80
+ return headline.trim() ? "banner" : "textless";
81
+ }
82
+
62
83
  /** Where the face sits in the COVER frame, as fractions of it. */
63
84
  export interface CoverFace {
64
85
  centerXFrac: number;
package/src/transcribe.ts CHANGED
@@ -13,6 +13,96 @@ export interface WhisperJson {
13
13
 
14
14
  const NOISE_TOKEN = /^[[(].*[\])]$/; // [BLANK_AUDIO], (buzzing), [MUSIC] …
15
15
 
16
+ /**
17
+ * How many bytes at the END of `bytes` form the start of a multi-byte UTF-8
18
+ * character whose continuation bytes are missing (§130: whisper.cpp `-ml 1`
19
+ * splits byte-level BPE tokens mid-character, so a segment's text can end on
20
+ * a bare lead byte — the field file ended one segment `0x20 0xD9` and started
21
+ * the next `0xB9 …`, the two halves of ٹ). 0 when the tail is complete.
22
+ */
23
+ function utf8DanglingTailLen(bytes: Buffer): number {
24
+ let i = bytes.length - 1;
25
+ let cont = 0;
26
+ while (i >= 0 && cont < 3 && (bytes[i]! & 0xc0) === 0x80) {
27
+ i--;
28
+ cont++;
29
+ }
30
+ // All continuation bytes: a HEAD fragment, someone else's tail — not ours.
31
+ if (i < 0) return 0;
32
+ const lead = bytes[i]!;
33
+ const need = lead >= 0xf0 ? 4 : lead >= 0xe0 ? 3 : lead >= 0xc0 ? 2 : 1;
34
+ // ASCII (or a stray byte no neighbor could complete) — nothing dangling.
35
+ if (need === 1) return 0;
36
+ return cont < need - 1 ? bytes.length - i : 0;
37
+ }
38
+
39
+ /**
40
+ * Repair segments whose text was split MID-CHARACTER by byte-level BPE
41
+ * (§130). Operates in byte space: `text` values here are latin1-decoded, one
42
+ * code point per original byte, because the fix must see the real bytes —
43
+ * reading the file as utf8 first would already have destroyed them (Node
44
+ * substitutes U+FFFD, and the two halves of the character are gone for good).
45
+ * A segment ending on an incomplete sequence merges with a following segment
46
+ * that begins with continuation bytes, spanning both segments' offsets; the
47
+ * loop re-checks the merged tail so a 3–4 byte character split across three
48
+ * segments still heals. Only that exact shape merges — a dangling tail whose
49
+ * neighbor does NOT continue it is unrecoverable and left to decode to
50
+ * U+FFFD, which parseWhisperJson then folds into a neighboring word.
51
+ */
52
+ function repairSplitSegments(json: WhisperJson): WhisperJson {
53
+ const out: WhisperJson["transcription"] = [];
54
+ for (const seg of json.transcription ?? []) {
55
+ const prev = out[out.length - 1];
56
+ const bytes = Buffer.from(seg.text, "latin1");
57
+ if (
58
+ prev &&
59
+ bytes.length > 0 &&
60
+ (bytes[0]! & 0xc0) === 0x80 &&
61
+ utf8DanglingTailLen(Buffer.from(prev.text, "latin1")) > 0
62
+ ) {
63
+ prev.text += seg.text;
64
+ prev.offsets = {
65
+ from: Math.min(prev.offsets.from, seg.offsets.from),
66
+ to: Math.max(prev.offsets.to, seg.offsets.to),
67
+ };
68
+ } else {
69
+ out.push({ ...seg, offsets: { ...seg.offsets } });
70
+ }
71
+ }
72
+ return {
73
+ ...json,
74
+ transcription: out.map((s) => ({
75
+ ...s,
76
+ text: Buffer.from(s.text, "latin1").toString("utf8"),
77
+ })),
78
+ };
79
+ }
80
+
81
+ const STRICT_UTF8 = new TextDecoder("utf-8", { fatal: true });
82
+
83
+ /**
84
+ * Parse whisper.cpp's `-oj` output from its raw BYTES. The file is not
85
+ * guaranteed to be valid UTF-8 (§130): with `-ml 1` a multi-byte character
86
+ * can be split across two segments' text fields at the byte level, and a
87
+ * plain `readFile(path, "utf8")` silently replaces both halves with U+FFFD —
88
+ * which is exactly the `��اپک` that shipped into the Urdu field captions
89
+ * (Urdu field test 2026-08-05). Valid files take the strict-decode path and
90
+ * behave byte-identically to before; invalid ones round-trip through latin1
91
+ * (byte-transparent, and UTF-8 continuation bytes are ≥0x80 so JSON's ASCII
92
+ * structure is untouched) so the split can be healed with the bytes intact.
93
+ */
94
+ export function parseWhisperOutput(raw: Buffer): Transcript {
95
+ let text: string | null = null;
96
+ try {
97
+ text = STRICT_UTF8.decode(raw);
98
+ } catch {
99
+ // Invalid UTF-8: the token-split shape. Fall through to byte repair —
100
+ // but only for DECODE failures; JSON syntax errors below still throw.
101
+ }
102
+ if (text !== null) return parseWhisperJson(JSON.parse(text) as WhisperJson);
103
+ return parseWhisperJson(repairSplitSegments(JSON.parse(raw.toString("latin1")) as WhisperJson));
104
+ }
105
+
16
106
  /**
17
107
  * Convert whisper.cpp `-ml 1` segments (≈ one token each) into words.
18
108
  * Tokens beginning with whitespace start a new word; bare continuations
@@ -36,6 +126,26 @@ export function parseWhisperJson(json: WhisperJson): Transcript {
36
126
  words.push({ text, start, end });
37
127
  }
38
128
  }
129
+ // U+FFFD is never displayable speech — it is a byte the split repair could
130
+ // not heal (§130: the field file also holds a lone lead byte whose
131
+ // continuation whisper never emitted at all, mid-word between "ی" and "ج"),
132
+ // and shipping it paints a literal � caption. Strip it from mixed words; a
133
+ // word left EMPTY by the strip folds its time span into a neighbor instead
134
+ // of vanishing — the span still belongs to speech.
135
+ for (let i = words.length - 1; i >= 0; i--) {
136
+ const w = words[i]!;
137
+ if (!w.text.includes("�")) continue;
138
+ const cleaned = w.text.replaceAll("�", "");
139
+ if (cleaned) {
140
+ w.text = cleaned;
141
+ continue;
142
+ }
143
+ const prev = words[i - 1];
144
+ const next = words[i + 1];
145
+ if (prev) prev.end = Math.max(prev.end, w.end);
146
+ else if (next) next.start = Math.min(next.start, w.start);
147
+ words.splice(i, 1);
148
+ }
39
149
  // Whisper occasionally emits zero-length or inverted stamps; repair minimally.
40
150
  for (let i = 0; i < words.length; i++) {
41
151
  const w = words[i]!;
@@ -51,17 +161,36 @@ export interface WhisperOptions {
51
161
  modelPath: string;
52
162
  /** Output base path; whisper writes `${outBase}.json`. */
53
163
  outBase: string;
164
+ /**
165
+ * Language code passed as `-l` (e.g. "ur", "de", "auto"). whisper.cpp
166
+ * defaults to English when the flag is absent, which decodes garbage out of
167
+ * a non-English fine-tune (Urdu field test 2026-08-05: ggml-medium-urdu
168
+ * needed `-l ur` to emit Urdu script at all). Left unset, the spawned args
169
+ * stay byte-identical to what English-suffixed models always got.
170
+ */
171
+ language?: string;
54
172
  }
55
173
 
56
- export async function runWhisper(opts: WhisperOptions, wavPath: string): Promise<Transcript> {
57
- await run(opts.whisperPath, [
174
+ /**
175
+ * Pure arg construction, split from the spawn the same way openCommand() is
176
+ * split from openInBrowser(): the `-l` conditional is exactly the kind of
177
+ * branch that must be testable without a whisper binary on the box.
178
+ */
179
+ export function whisperArgs(opts: WhisperOptions, wavPath: string): string[] {
180
+ const args = [
58
181
  "-m", opts.modelPath,
59
182
  "-f", wavPath,
60
183
  "-oj",
61
184
  "-of", opts.outBase,
62
185
  "-ml", "1",
63
186
  "--no-prints",
64
- ]);
65
- const json = JSON.parse(await readFile(`${opts.outBase}.json`, "utf8")) as WhisperJson;
66
- return parseWhisperJson(json);
187
+ ];
188
+ if (opts.language !== undefined) args.push("-l", opts.language);
189
+ return args;
190
+ }
191
+
192
+ export async function runWhisper(opts: WhisperOptions, wavPath: string): Promise<Transcript> {
193
+ await run(opts.whisperPath, whisperArgs(opts, wavPath));
194
+ // Bytes, not "utf8": the utf8 read is where the §130 split characters died.
195
+ return parseWhisperOutput(await readFile(`${opts.outBase}.json`));
67
196
  }