@ossclip/core 0.1.35 → 0.1.37
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/browser.ts +15 -0
- package/src/color-grade.ts +562 -0
- package/src/config.ts +55 -0
- package/src/index.ts +2 -0
- package/src/ingest.ts +87 -8
- package/src/lut-library.ts +81 -0
- package/src/overrides.ts +32 -0
- package/src/recut.ts +32 -3
- package/src/scene-schema.ts +3 -0
- package/src/transcribe/index.ts +3 -0
- package/src/transcribe/openai-compatible.ts +199 -0
- package/src/transcribe/provider.ts +101 -0
- package/src/{transcribe.ts → transcribe/whisper-cli.ts} +8 -61
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { readFile } from "node:fs/promises";
|
|
2
|
-
import { run } from "
|
|
3
|
-
import type { Transcript, Word } from "
|
|
2
|
+
import { run } from "../exec";
|
|
3
|
+
import type { Transcript, Word } from "../schema";
|
|
4
|
+
import { NOISE_TOKEN, normalizeWords } from "./provider";
|
|
4
5
|
|
|
5
6
|
/** Shape of whisper.cpp's `-oj` JSON output (the fields we consume). */
|
|
6
7
|
export interface WhisperJson {
|
|
@@ -11,8 +12,6 @@ export interface WhisperJson {
|
|
|
11
12
|
}>;
|
|
12
13
|
}
|
|
13
14
|
|
|
14
|
-
const NOISE_TOKEN = /^[[(].*[\])]$/; // [BLANK_AUDIO], (buzzing), [MUSIC] …
|
|
15
|
-
|
|
16
15
|
/**
|
|
17
16
|
* How many bytes at the END of `bytes` form the start of a multi-byte UTF-8
|
|
18
17
|
* character whose continuation bytes are missing (§130: whisper.cpp `-ml 1`
|
|
@@ -78,51 +77,6 @@ function repairSplitSegments(json: WhisperJson): WhisperJson {
|
|
|
78
77
|
};
|
|
79
78
|
}
|
|
80
79
|
|
|
81
|
-
/**
|
|
82
|
-
* Run length at which a stack of zero-length words at ONE instant stops being
|
|
83
|
-
* a rounding artifact and becomes a repetition-loop hallucination. Real speech
|
|
84
|
-
* never emits 8 tokens at a single instant; the field case emitted 118.
|
|
85
|
-
*/
|
|
86
|
-
export const REPETITION_BURST_MIN = 8;
|
|
87
|
-
|
|
88
|
-
/**
|
|
89
|
-
* Drop whisper repetition-loop bursts (field case 2026-08-18): an Urdu take
|
|
90
|
-
* re-decoded a whole phrase as 118 CONSECUTIVE tokens all stamped
|
|
91
|
-
* `from === to === 31040` — zero length, at one instant. The stamp repair
|
|
92
|
-
* below then fans such a burst out into 118 fabricated 50ms words marching
|
|
93
|
-
* forward from 31.04s, so the phrase ships TWICE in the captions (31.04s and
|
|
94
|
-
* 33.54s) and a fifth of the transcript carries the tell-tale exactly-0.05s
|
|
95
|
-
* duration. `-mc 0` in whisperArgs is the decoder-side mitigation for the same
|
|
96
|
-
* failure; it did not prevent this occurrence, and it can never repair an
|
|
97
|
-
* already-cached transcript.json — hence a parse-side guard too.
|
|
98
|
-
*
|
|
99
|
-
* A burst is a MAXIMAL run of consecutive zero-length/inverted words sharing
|
|
100
|
-
* one `start`. Equality is exact, not epsilon: these stamps are integer
|
|
101
|
-
* milliseconds divided by 1000, so members of one burst are the same double
|
|
102
|
-
* bit-for-bit, and a tolerance would only start swallowing real neighbors.
|
|
103
|
-
* Runs shorter than REPETITION_BURST_MIN fall through untouched — a lone
|
|
104
|
-
* zero-length stamp is a rounding artifact, not a hallucination. The drop is
|
|
105
|
-
* silent by design: this function is pure and total, and there is no logging
|
|
106
|
-
* channel in the parse path to warn on.
|
|
107
|
-
*/
|
|
108
|
-
export function dropRepetitionBursts(words: readonly Word[]): Word[] {
|
|
109
|
-
const out: Word[] = [];
|
|
110
|
-
let i = 0;
|
|
111
|
-
while (i < words.length) {
|
|
112
|
-
const w = words[i]!;
|
|
113
|
-
if (w.end > w.start) {
|
|
114
|
-
out.push(w);
|
|
115
|
-
i++;
|
|
116
|
-
continue;
|
|
117
|
-
}
|
|
118
|
-
let j = i + 1;
|
|
119
|
-
while (j < words.length && words[j]!.end <= words[j]!.start && words[j]!.start === w.start) j++;
|
|
120
|
-
if (j - i < REPETITION_BURST_MIN) for (let k = i; k < j; k++) out.push(words[k]!);
|
|
121
|
-
i = j;
|
|
122
|
-
}
|
|
123
|
-
return out;
|
|
124
|
-
}
|
|
125
|
-
|
|
126
80
|
const STRICT_UTF8 = new TextDecoder("utf-8", { fatal: true });
|
|
127
81
|
|
|
128
82
|
/**
|
|
@@ -199,18 +153,11 @@ export function parseWhisperJson(json: WhisperJson): Transcript {
|
|
|
199
153
|
else if (next) next.start = Math.min(next.start, w.start);
|
|
200
154
|
words.splice(i, 1);
|
|
201
155
|
}
|
|
202
|
-
//
|
|
203
|
-
//
|
|
204
|
-
//
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
for (let i = 0; i < kept.length; i++) {
|
|
208
|
-
const w = kept[i]!;
|
|
209
|
-
if (w.end <= w.start) w.end = w.start + 0.05;
|
|
210
|
-
const next = kept[i + 1];
|
|
211
|
-
if (next && next.start < w.end) next.start = w.end;
|
|
212
|
-
}
|
|
213
|
-
return { language: json.result?.language ?? "en", words: kept };
|
|
156
|
+
// Burst drop + stamp repair now live in `normalizeWords` (provider.ts,
|
|
157
|
+
// 2026-09-01): the remote backend needs exactly the same two passes in
|
|
158
|
+
// exactly the same order, and duplicating them is how the two paths would
|
|
159
|
+
// drift. Behavior here is unchanged — the parser matrix pins it.
|
|
160
|
+
return { language: json.result?.language ?? "en", words: normalizeWords(words) };
|
|
214
161
|
}
|
|
215
162
|
|
|
216
163
|
export interface WhisperOptions {
|