@ossclip/core 0.1.36 → 0.1.37
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json
CHANGED
package/src/config.ts
CHANGED
|
@@ -199,6 +199,30 @@ export interface OssclipConfig {
|
|
|
199
199
|
* (`publishConfigured` in the CLI), never coerced.
|
|
200
200
|
*/
|
|
201
201
|
postizUrl?: string;
|
|
202
|
+
/**
|
|
203
|
+
* Base URL of an OpenAI-compatible transcription server, ending in `/v1`
|
|
204
|
+
* (Groq: "https://api.groq.com/openai/v1"; a self-hosted speaches:
|
|
205
|
+
* "http://localhost:8000/v1"). Set, transcription runs REMOTELY instead of
|
|
206
|
+
* on this machine's CPU — the 2026-09-01 field report from an i3 2nd gen,
|
|
207
|
+
* where whisper is the dominant cost of a run. Unset, local whisper-cli
|
|
208
|
+
* stays the default, and `--whisper-backend local` overrides per run.
|
|
209
|
+
*
|
|
210
|
+
* Non-secret, the `postizUrl` posture, so it may live here; the API key is
|
|
211
|
+
* `OSSCLIP_WHISPER_API_KEY` in the ENVIRONMENT only (env.ts's documented
|
|
212
|
+
* rule) — and OPTIONAL, because self-hosted servers run keyless. Validated
|
|
213
|
+
* at the consumer (`resolveWhisperBackend` in the CLI), never coerced.
|
|
214
|
+
*/
|
|
215
|
+
whisperUrl?: string;
|
|
216
|
+
/**
|
|
217
|
+
* Model name sent to that server ("whisper-large-v3-turbo" on Groq,
|
|
218
|
+
* "Systran/faster-whisper-large-v3" on a speaches box). Deliberately NOT
|
|
219
|
+
* `model`, which names the local ggml file — a remote run must not be able
|
|
220
|
+
* to send "small.en" to a server that has never heard of it. The default
|
|
221
|
+
* lives at the consumer (`resolveWhisperBackend`), not in DEFAULTS: an
|
|
222
|
+
* unset key here must stay unset so nothing writes a remote model name into
|
|
223
|
+
* a local-only config.
|
|
224
|
+
*/
|
|
225
|
+
whisperRemoteModel?: string;
|
|
202
226
|
/**
|
|
203
227
|
* `--resolution`'s default for this machine: "auto" (keep what the source
|
|
204
228
|
* has, capped at 2160), "1080" (the built-in default), "1440" or "2160".
|
|
@@ -367,6 +391,15 @@ export function resolveConfig(
|
|
|
367
391
|
// deliberately lives in the environment (publish.ts's
|
|
368
392
|
// `publishConfigured`), so this is only the instance URL.
|
|
369
393
|
postizUrl: fileCfg.postizUrl,
|
|
394
|
+
// Env spellings ON PURPOSE, unlike postizUrl: the Groq quickstart is
|
|
395
|
+
// "export two vars and run", and a user trying a free tier for the first
|
|
396
|
+
// time should not have to hand-edit config.json to do it. Env beats file
|
|
397
|
+
// so a one-off `OSSCLIP_WHISPER_URL=… ossclip produce` works against a
|
|
398
|
+
// machine whose config says otherwise. Both are strings straight through
|
|
399
|
+
// — validated at the consumer (`resolveWhisperBackend`), never coerced,
|
|
400
|
+
// and the postizUrl lesson above is why they are copied here at all.
|
|
401
|
+
whisperUrl: env.OSSCLIP_WHISPER_URL ?? fileCfg.whisperUrl,
|
|
402
|
+
whisperRemoteModel: env.OSSCLIP_WHISPER_REMOTE_MODEL ?? fileCfg.whisperRemoteModel,
|
|
370
403
|
resolution: fileCfg.resolution,
|
|
371
404
|
};
|
|
372
405
|
}
|
package/src/ingest.ts
CHANGED
|
@@ -85,6 +85,42 @@ export async function extractAudio(tools: IngestTools, src: string, outWav: stri
|
|
|
85
85
|
]);
|
|
86
86
|
}
|
|
87
87
|
|
|
88
|
+
/**
|
|
89
|
+
* Pure: ffmpeg args for the compressed remote-upload sidecar (2026-09-01,
|
|
90
|
+
* the remote-transcription backend). Split from the spawn the same way
|
|
91
|
+
* `openCommand()` is split from `openInBrowser()` — the codec choice is the
|
|
92
|
+
* whole decision here, so it must be assertable without an ffmpeg on the box.
|
|
93
|
+
*
|
|
94
|
+
* Opus 32 kbps 16 kHz mono: ASR-transparent for speech and ~240 KB/min, so
|
|
95
|
+
* ~100 minutes of audio fit under Groq's free-tier 25 MB per-file cap. The
|
|
96
|
+
* PCM wav `extractAudio` writes is 1.92 MB/min and hits that cap at ~13
|
|
97
|
+
* minutes — unusable for a normal recording session. FLAC was rejected:
|
|
98
|
+
* lossless buys ASR nothing and only reaches ~25 minutes.
|
|
99
|
+
*/
|
|
100
|
+
export function uploadAudioArgs(wavPath: string, outOgg: string): string[] {
|
|
101
|
+
return ["-y", "-i", wavPath, "-vn", "-c:a", "libopus", "-b:a", "32k", "-ar", "16000", "-ac", "1", outOgg];
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Encode the remote-upload sidecar. libopus is in the bundled static ffmpeg;
|
|
106
|
+
* a user's own minimal build may lack it, in which case `run` surfaces
|
|
107
|
+
* ffmpeg's own "Unknown encoder" verbatim — better than a guess at the cause.
|
|
108
|
+
*/
|
|
109
|
+
export async function encodeUploadAudio(tools: IngestTools, wav: string, outOgg: string): Promise<void> {
|
|
110
|
+
await run(tools.ffmpegPath, uploadAudioArgs(wav, outOgg));
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* 24 million bytes: Groq's free tier caps a file at "25MB" without saying
|
|
115
|
+
* whether that is decimal or MiB — 24_000_000 sits under BOTH readings
|
|
116
|
+
* (25,000,000 and 26,214,336), so a boundary file fails HERE with our message
|
|
117
|
+
* (naming the ~100-minute ceiling and the local-backend escape hatch) instead
|
|
118
|
+
* of coming back as somebody else's 413 after the whole upload was paid for.
|
|
119
|
+
* (24 MiB = 25,165,824 would EXCEED a decimal cap — caught 2026-09-01.)
|
|
120
|
+
* Chunking is out of scope for v1.
|
|
121
|
+
*/
|
|
122
|
+
export const REMOTE_UPLOAD_MAX_BYTES = 24_000_000;
|
|
123
|
+
|
|
88
124
|
/**
|
|
89
125
|
* Extract ONE span of an existing wav, same 16 kHz mono PCM shape
|
|
90
126
|
* (2026-08-26, the caption re-alignment pass).
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
import { basename } from "node:path";
|
|
2
|
+
import { z } from "zod/v4";
|
|
3
|
+
import { TranscriptSchema, type Transcript, type Word } from "../schema";
|
|
4
|
+
import { NOISE_TOKEN, normalizeWords, type TranscribeProvider, type TranscribeRequest } from "./provider";
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Transcription against any OpenAI-compatible `/v1/audio/transcriptions`
|
|
8
|
+
* server — Groq's free tier (8h audio/day, `whisper-large-v3-turbo`), a
|
|
9
|
+
* self-hosted speaches, or anything else speaking that shape.
|
|
10
|
+
*
|
|
11
|
+
* Why (2026-09-01 field report): on a weak CPU — an i3 2nd gen — whisper is
|
|
12
|
+
* the dominant cost of a produce run, minutes of decode per minute of video.
|
|
13
|
+
* A remote call makes it seconds. Local whisper-cli stays the DEFAULT; this
|
|
14
|
+
* is opt-in via `whisperUrl` / `OSSCLIP_WHISPER_URL`, and the API key is
|
|
15
|
+
* optional because self-hosted servers run keyless (unlike publish, where the
|
|
16
|
+
* key is required).
|
|
17
|
+
*
|
|
18
|
+
* Error posture is postiz.ts's, for the same reason: a transcription is the
|
|
19
|
+
* user's explicit action, so every non-2xx throws with the status and a body
|
|
20
|
+
* snippet, and there are NO retries — a retry against a metered free tier
|
|
21
|
+
* silently doubles the quota burn for a failure the user is about to see
|
|
22
|
+
* anyway.
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* `whisperUrl` → the endpoint: trailing slashes dropped,
|
|
27
|
+
* `/audio/transcriptions` appended unless the user already wrote it
|
|
28
|
+
* (`postizApiBase`'s mould). The user configures the OpenAI-compatible BASE
|
|
29
|
+
* ("https://api.groq.com/openai/v1"), which is the URL every provider's
|
|
30
|
+
* quickstart prints — but someone who pastes the full endpoint must not end
|
|
31
|
+
* up posting to `/v1/audio/transcriptions/audio/transcriptions`.
|
|
32
|
+
*/
|
|
33
|
+
export function openaiTranscriptionsUrl(baseUrl: string): string {
|
|
34
|
+
const trimmed = baseUrl.replace(/\/+$/, "");
|
|
35
|
+
return trimmed.endsWith("/audio/transcriptions") ? trimmed : `${trimmed}/audio/transcriptions`;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export class RemoteTranscribeHttpError extends Error {
|
|
39
|
+
constructor(
|
|
40
|
+
readonly url: string,
|
|
41
|
+
readonly status: number,
|
|
42
|
+
bodySnippet: string,
|
|
43
|
+
) {
|
|
44
|
+
// Per-status hints, PostizHttpError's mould: the raw status tells a user
|
|
45
|
+
// nothing about which of the two env vars (or which URL spelling) is
|
|
46
|
+
// wrong, and this path is reached by people who just pasted a quickstart.
|
|
47
|
+
const hint =
|
|
48
|
+
status === 401 || status === 403
|
|
49
|
+
? " — the server rejected OSSCLIP_WHISPER_API_KEY (or none was sent — set it in the environment or ~/.ossclip/.env)"
|
|
50
|
+
: status === 404
|
|
51
|
+
? " — no /audio/transcriptions here — whisperUrl should be the OpenAI-compatible base ending in /v1 (e.g. https://api.groq.com/openai/v1)"
|
|
52
|
+
: status === 413
|
|
53
|
+
? " — audio too large for this server — free Groq caps uploads at 25MB; use --whisper-backend local or the dev tier"
|
|
54
|
+
: status === 429
|
|
55
|
+
? " — rate limited (Groq free tier: 8h audio/day)"
|
|
56
|
+
: "";
|
|
57
|
+
super(`remote transcription POST ${url} failed: ${status}${hint}${bodySnippet ? `\n${bodySnippet}` : ""}`);
|
|
58
|
+
this.name = "RemoteTranscribeHttpError";
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* `verbose_json` as we consume it. LOOSE on purpose: servers add fields
|
|
64
|
+
* freely (Groq sends `task`, `duration`, `segments`, `x_groq`), and a strict
|
|
65
|
+
* object would turn a perfectly good transcription into a parse error the
|
|
66
|
+
* next time one of them ships a field.
|
|
67
|
+
*/
|
|
68
|
+
const RemoteWordSchema = z.looseObject({
|
|
69
|
+
word: z.string(),
|
|
70
|
+
start: z.number(),
|
|
71
|
+
end: z.number(),
|
|
72
|
+
});
|
|
73
|
+
const VerboseJsonSchema = z.looseObject({
|
|
74
|
+
language: z.string().optional(),
|
|
75
|
+
words: z.array(RemoteWordSchema).optional(),
|
|
76
|
+
});
|
|
77
|
+
|
|
78
|
+
export interface OpenAiCompatibleOptions {
|
|
79
|
+
/** OpenAI-compatible base, e.g. "https://api.groq.com/openai/v1". */
|
|
80
|
+
baseUrl: string;
|
|
81
|
+
model: string;
|
|
82
|
+
/** Optional: self-hosted servers (speaches, whisper.cpp server) run keyless. */
|
|
83
|
+
apiKey?: string;
|
|
84
|
+
/** The postiz test seam — the whole HTTP surface is testable without a network. */
|
|
85
|
+
fetchImpl?: typeof fetch;
|
|
86
|
+
/** Per-request cap; an hour of audio takes a while to upload and decode. */
|
|
87
|
+
timeoutMs?: number;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
const DEFAULT_TIMEOUT_MS = 10 * 60 * 1000;
|
|
91
|
+
const BODY_SNIPPET_CHARS = 300;
|
|
92
|
+
|
|
93
|
+
export function createOpenAiCompatibleProvider(opts: OpenAiCompatibleOptions): TranscribeProvider {
|
|
94
|
+
const url = openaiTranscriptionsUrl(opts.baseUrl);
|
|
95
|
+
const fetchImpl = opts.fetchImpl ?? fetch;
|
|
96
|
+
const timeoutMs = opts.timeoutMs ?? DEFAULT_TIMEOUT_MS;
|
|
97
|
+
|
|
98
|
+
return {
|
|
99
|
+
name: "openai-compatible",
|
|
100
|
+
async transcribe(audioPath: string, req: TranscribeRequest): Promise<Transcript> {
|
|
101
|
+
// Belt and braces — the CLI refuses this combination earlier, with the
|
|
102
|
+
// fix named. `/audio/translations` is a different endpoint AND a
|
|
103
|
+
// different default model upstream, so silently swapping both behind
|
|
104
|
+
// `--whisper-translate` would be a surprise, not a convenience.
|
|
105
|
+
if (req.translate === true) {
|
|
106
|
+
throw new Error(
|
|
107
|
+
"--whisper-translate needs the local backend (the OpenAI-compatible API translates on a different endpoint and model) — use --whisper-backend local, or drop the flag.",
|
|
108
|
+
);
|
|
109
|
+
}
|
|
110
|
+
const { openAsBlob } = await import("node:fs");
|
|
111
|
+
// Streams the file into multipart form-data instead of holding it in
|
|
112
|
+
// memory (postiz.ts's rationale): the upload sidecar is small, but a
|
|
113
|
+
// span wav or an uncompressed hour is not.
|
|
114
|
+
const blob = await openAsBlob(audioPath, {
|
|
115
|
+
type: audioPath.endsWith(".ogg") ? "audio/ogg" : "audio/wav",
|
|
116
|
+
});
|
|
117
|
+
const form = new FormData();
|
|
118
|
+
form.append("file", blob, basename(audioPath));
|
|
119
|
+
form.append("model", opts.model);
|
|
120
|
+
form.append("response_format", "verbose_json");
|
|
121
|
+
// The literal bracketed field name is the wire spelling OpenAI and Groq
|
|
122
|
+
// accept — it is an array parameter in a multipart body, not a typo.
|
|
123
|
+
form.append("timestamp_granularities[]", "word");
|
|
124
|
+
// "auto" is whisper.cpp's vocabulary, not an ISO code: sending it makes
|
|
125
|
+
// the server reject the request, while OMITTING the field is exactly
|
|
126
|
+
// what asks for auto-detection.
|
|
127
|
+
if (req.language !== undefined && req.language !== "auto") form.append("language", req.language);
|
|
128
|
+
if (req.prompt !== undefined) form.append("prompt", req.prompt);
|
|
129
|
+
|
|
130
|
+
const ac = new AbortController();
|
|
131
|
+
const timer = setTimeout(() => ac.abort(), timeoutMs);
|
|
132
|
+
let res: Response;
|
|
133
|
+
try {
|
|
134
|
+
res = await fetchImpl(url, {
|
|
135
|
+
method: "POST",
|
|
136
|
+
headers: opts.apiKey ? { Authorization: `Bearer ${opts.apiKey}` } : {},
|
|
137
|
+
body: form,
|
|
138
|
+
signal: ac.signal,
|
|
139
|
+
});
|
|
140
|
+
} catch (err) {
|
|
141
|
+
throw new Error(
|
|
142
|
+
`remote transcription unreachable at ${url}: ${err instanceof Error ? err.message : String(err)}`,
|
|
143
|
+
);
|
|
144
|
+
} finally {
|
|
145
|
+
clearTimeout(timer);
|
|
146
|
+
}
|
|
147
|
+
const text = await res.text();
|
|
148
|
+
if (!res.ok) throw new RemoteTranscribeHttpError(url, res.status, text.slice(0, BODY_SNIPPET_CHARS));
|
|
149
|
+
let json: unknown;
|
|
150
|
+
try {
|
|
151
|
+
json = JSON.parse(text);
|
|
152
|
+
} catch {
|
|
153
|
+
throw new Error(
|
|
154
|
+
`remote transcription answered non-JSON from ${url}: ${text.slice(0, BODY_SNIPPET_CHARS)}`,
|
|
155
|
+
);
|
|
156
|
+
}
|
|
157
|
+
const parsed = VerboseJsonSchema.parse(json);
|
|
158
|
+
if (!parsed.words || parsed.words.length === 0) {
|
|
159
|
+
// Two causes, both worth naming: a server that transcribed fine but
|
|
160
|
+
// has no word-timestamp support (a plain whisper.cpp server), and
|
|
161
|
+
// near-silent audio, which Groq's turbo model answers wordlessly.
|
|
162
|
+
// Everything downstream (cuts, captions, zoom) is word-stamp driven,
|
|
163
|
+
// so a text-only answer is unusable, not a degraded success.
|
|
164
|
+
throw new Error(
|
|
165
|
+
`the server answered without word timestamps — it must support response_format=verbose_json with timestamp_granularities[]=word (Groq and speaches do; a plain whisper.cpp server may not), or the audio contained no speech (${url})`,
|
|
166
|
+
);
|
|
167
|
+
}
|
|
168
|
+
// No token merging and no §130 byte repair here: those heal whisper.cpp
|
|
169
|
+
// `-ml 1` artifacts (one BPE token per segment, split mid-character).
|
|
170
|
+
// The HTTP API returns whole words with punctuation attached — already
|
|
171
|
+
// the shape parseWhisperJson works to produce.
|
|
172
|
+
const words: Word[] = [];
|
|
173
|
+
for (const w of parsed.words) {
|
|
174
|
+
const wordText = w.word.trim();
|
|
175
|
+
if (!wordText || NOISE_TOKEN.test(wordText)) continue;
|
|
176
|
+
words.push({
|
|
177
|
+
text: wordText,
|
|
178
|
+
// Clamped: a server answering -0.01 for the first word would trip
|
|
179
|
+
// WordSchema's nonnegative and fail the whole run over a rounding
|
|
180
|
+
// artifact at the very start of the audio.
|
|
181
|
+
start: Math.max(0, w.start),
|
|
182
|
+
end: w.end,
|
|
183
|
+
});
|
|
184
|
+
}
|
|
185
|
+
return TranscriptSchema.parse({
|
|
186
|
+
// The requested code wins — it is what the cache key and the caption
|
|
187
|
+
// pipeline were told the audio is. Otherwise the server's own answer,
|
|
188
|
+
// lowercased. NOTE: some servers answer with a full NAME ("english")
|
|
189
|
+
// rather than a code; no code-sensitive consumer exists today
|
|
190
|
+
// (captions' RTL check is a Unicode heuristic), so no name→code table.
|
|
191
|
+
language:
|
|
192
|
+
req.language !== undefined && req.language !== "auto"
|
|
193
|
+
? req.language
|
|
194
|
+
: parsed.language?.toLowerCase(),
|
|
195
|
+
words: normalizeWords(words),
|
|
196
|
+
});
|
|
197
|
+
},
|
|
198
|
+
};
|
|
199
|
+
}
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
import type { Transcript, Word } from "../schema";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* The seam between ossclip and any transcription backend (2026-09-01, the
|
|
5
|
+
* weak-CPU field report: whisper is the dominant cost on an i3 2nd gen, and a
|
|
6
|
+
* free remote tier makes it disappear). Two implementations today —
|
|
7
|
+
* whisper.cpp on the box (`whisper-cli.ts`) and any OpenAI-compatible
|
|
8
|
+
* `/v1/audio/transcriptions` server (`openai-compatible.ts`).
|
|
9
|
+
*
|
|
10
|
+
* The local path deliberately keeps calling `runWhisper` directly rather than
|
|
11
|
+
* going through this interface: produce.ts and the edit server inject
|
|
12
|
+
* `runWhisper` itself as a test seam, and forcing that through a provider
|
|
13
|
+
* object would churn every stub for no behavior change.
|
|
14
|
+
*/
|
|
15
|
+
export interface TranscribeRequest {
|
|
16
|
+
/** Resolved language code; "auto" is handled per provider (whisper.cpp
|
|
17
|
+
* takes it literally, the HTTP API wants the field OMITTED). */
|
|
18
|
+
language?: string;
|
|
19
|
+
/** `whisperPromptFor()` output — the user dictionary as decoder bias. */
|
|
20
|
+
prompt?: string;
|
|
21
|
+
/** whisper-cli's TRANSLATE task; the OpenAI-compatible provider rejects it
|
|
22
|
+
* (a different endpoint AND a different default model upstream). */
|
|
23
|
+
translate?: boolean;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export interface TranscribeProvider {
|
|
27
|
+
name: string;
|
|
28
|
+
transcribe(audioPath: string, req: TranscribeRequest): Promise<Transcript>;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** [BLANK_AUDIO], (buzzing), [MUSIC] … — noise markers, not speech. Shared:
|
|
32
|
+
* remote servers emit the same bracketed markers whisper.cpp does. */
|
|
33
|
+
export const NOISE_TOKEN = /^[[(].*[\])]$/;
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Run length at which a stack of zero-length words at ONE instant stops being
|
|
37
|
+
* a rounding artifact and becomes a repetition-loop hallucination. Real speech
|
|
38
|
+
* never emits 8 tokens at a single instant; the field case emitted 118.
|
|
39
|
+
*/
|
|
40
|
+
export const REPETITION_BURST_MIN = 8;
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Drop whisper repetition-loop bursts (field case 2026-08-18): an Urdu take
|
|
44
|
+
* re-decoded a whole phrase as 118 CONSECUTIVE tokens all stamped
|
|
45
|
+
* `from === to === 31040` — zero length, at one instant. The stamp repair
|
|
46
|
+
* below then fans such a burst out into 118 fabricated 50ms words marching
|
|
47
|
+
* forward from 31.04s, so the phrase ships TWICE in the captions (31.04s and
|
|
48
|
+
* 33.54s) and a fifth of the transcript carries the tell-tale exactly-0.05s
|
|
49
|
+
* duration. `-mc 0` in whisperArgs is the decoder-side mitigation for the same
|
|
50
|
+
* failure; it did not prevent this occurrence, and it can never repair an
|
|
51
|
+
* already-cached transcript.json — hence a parse-side guard too.
|
|
52
|
+
*
|
|
53
|
+
* A burst is a MAXIMAL run of consecutive zero-length/inverted words sharing
|
|
54
|
+
* one `start`. Equality is exact, not epsilon: these stamps are integer
|
|
55
|
+
* milliseconds divided by 1000, so members of one burst are the same double
|
|
56
|
+
* bit-for-bit, and a tolerance would only start swallowing real neighbors.
|
|
57
|
+
* Runs shorter than REPETITION_BURST_MIN fall through untouched — a lone
|
|
58
|
+
* zero-length stamp is a rounding artifact, not a hallucination. The drop is
|
|
59
|
+
* silent by design: this function is pure and total, and there is no logging
|
|
60
|
+
* channel in the parse path to warn on.
|
|
61
|
+
*/
|
|
62
|
+
export function dropRepetitionBursts(words: readonly Word[]): Word[] {
|
|
63
|
+
const out: Word[] = [];
|
|
64
|
+
let i = 0;
|
|
65
|
+
while (i < words.length) {
|
|
66
|
+
const w = words[i]!;
|
|
67
|
+
if (w.end > w.start) {
|
|
68
|
+
out.push(w);
|
|
69
|
+
i++;
|
|
70
|
+
continue;
|
|
71
|
+
}
|
|
72
|
+
let j = i + 1;
|
|
73
|
+
while (j < words.length && words[j]!.end <= words[j]!.start && words[j]!.start === w.start) j++;
|
|
74
|
+
if (j - i < REPETITION_BURST_MIN) for (let k = i; k < j; k++) out.push(words[k]!);
|
|
75
|
+
i = j;
|
|
76
|
+
}
|
|
77
|
+
return out;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* The word-stamp hygiene EVERY backend's output goes through, extracted from
|
|
82
|
+
* the tail of `parseWhisperJson` so the remote provider cannot drift from it
|
|
83
|
+
* (2026-09-01). Burst drop runs BEFORE the repair, never after: the repair
|
|
84
|
+
* rewrites every burst member into a distinct monotone stamp, so once it has
|
|
85
|
+
* run the shared timestamp — the only evidence a burst existed — is gone.
|
|
86
|
+
*
|
|
87
|
+
* Copies before mutating, so a caller's array survives the call unchanged;
|
|
88
|
+
* `parseWhisperJson` builds its words fresh, so this is byte-identical to the
|
|
89
|
+
* in-place loop it replaces.
|
|
90
|
+
*/
|
|
91
|
+
export function normalizeWords(words: readonly Word[]): Word[] {
|
|
92
|
+
const kept = dropRepetitionBursts(words).map((w) => ({ ...w }));
|
|
93
|
+
// Whisper occasionally emits zero-length or inverted stamps; repair minimally.
|
|
94
|
+
for (let i = 0; i < kept.length; i++) {
|
|
95
|
+
const w = kept[i]!;
|
|
96
|
+
if (w.end <= w.start) w.end = w.start + 0.05;
|
|
97
|
+
const next = kept[i + 1];
|
|
98
|
+
if (next && next.start < w.end) next.start = w.end;
|
|
99
|
+
}
|
|
100
|
+
return kept;
|
|
101
|
+
}
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { readFile } from "node:fs/promises";
|
|
2
|
-
import { run } from "
|
|
3
|
-
import type { Transcript, Word } from "
|
|
2
|
+
import { run } from "../exec";
|
|
3
|
+
import type { Transcript, Word } from "../schema";
|
|
4
|
+
import { NOISE_TOKEN, normalizeWords } from "./provider";
|
|
4
5
|
|
|
5
6
|
/** Shape of whisper.cpp's `-oj` JSON output (the fields we consume). */
|
|
6
7
|
export interface WhisperJson {
|
|
@@ -11,8 +12,6 @@ export interface WhisperJson {
|
|
|
11
12
|
}>;
|
|
12
13
|
}
|
|
13
14
|
|
|
14
|
-
const NOISE_TOKEN = /^[[(].*[\])]$/; // [BLANK_AUDIO], (buzzing), [MUSIC] …
|
|
15
|
-
|
|
16
15
|
/**
|
|
17
16
|
* How many bytes at the END of `bytes` form the start of a multi-byte UTF-8
|
|
18
17
|
* character whose continuation bytes are missing (§130: whisper.cpp `-ml 1`
|
|
@@ -78,51 +77,6 @@ function repairSplitSegments(json: WhisperJson): WhisperJson {
|
|
|
78
77
|
};
|
|
79
78
|
}
|
|
80
79
|
|
|
81
|
-
/**
|
|
82
|
-
* Run length at which a stack of zero-length words at ONE instant stops being
|
|
83
|
-
* a rounding artifact and becomes a repetition-loop hallucination. Real speech
|
|
84
|
-
* never emits 8 tokens at a single instant; the field case emitted 118.
|
|
85
|
-
*/
|
|
86
|
-
export const REPETITION_BURST_MIN = 8;
|
|
87
|
-
|
|
88
|
-
/**
|
|
89
|
-
* Drop whisper repetition-loop bursts (field case 2026-08-18): an Urdu take
|
|
90
|
-
* re-decoded a whole phrase as 118 CONSECUTIVE tokens all stamped
|
|
91
|
-
* `from === to === 31040` — zero length, at one instant. The stamp repair
|
|
92
|
-
* below then fans such a burst out into 118 fabricated 50ms words marching
|
|
93
|
-
* forward from 31.04s, so the phrase ships TWICE in the captions (31.04s and
|
|
94
|
-
* 33.54s) and a fifth of the transcript carries the tell-tale exactly-0.05s
|
|
95
|
-
* duration. `-mc 0` in whisperArgs is the decoder-side mitigation for the same
|
|
96
|
-
* failure; it did not prevent this occurrence, and it can never repair an
|
|
97
|
-
* already-cached transcript.json — hence a parse-side guard too.
|
|
98
|
-
*
|
|
99
|
-
* A burst is a MAXIMAL run of consecutive zero-length/inverted words sharing
|
|
100
|
-
* one `start`. Equality is exact, not epsilon: these stamps are integer
|
|
101
|
-
* milliseconds divided by 1000, so members of one burst are the same double
|
|
102
|
-
* bit-for-bit, and a tolerance would only start swallowing real neighbors.
|
|
103
|
-
* Runs shorter than REPETITION_BURST_MIN fall through untouched — a lone
|
|
104
|
-
* zero-length stamp is a rounding artifact, not a hallucination. The drop is
|
|
105
|
-
* silent by design: this function is pure and total, and there is no logging
|
|
106
|
-
* channel in the parse path to warn on.
|
|
107
|
-
*/
|
|
108
|
-
export function dropRepetitionBursts(words: readonly Word[]): Word[] {
|
|
109
|
-
const out: Word[] = [];
|
|
110
|
-
let i = 0;
|
|
111
|
-
while (i < words.length) {
|
|
112
|
-
const w = words[i]!;
|
|
113
|
-
if (w.end > w.start) {
|
|
114
|
-
out.push(w);
|
|
115
|
-
i++;
|
|
116
|
-
continue;
|
|
117
|
-
}
|
|
118
|
-
let j = i + 1;
|
|
119
|
-
while (j < words.length && words[j]!.end <= words[j]!.start && words[j]!.start === w.start) j++;
|
|
120
|
-
if (j - i < REPETITION_BURST_MIN) for (let k = i; k < j; k++) out.push(words[k]!);
|
|
121
|
-
i = j;
|
|
122
|
-
}
|
|
123
|
-
return out;
|
|
124
|
-
}
|
|
125
|
-
|
|
126
80
|
const STRICT_UTF8 = new TextDecoder("utf-8", { fatal: true });
|
|
127
81
|
|
|
128
82
|
/**
|
|
@@ -199,18 +153,11 @@ export function parseWhisperJson(json: WhisperJson): Transcript {
|
|
|
199
153
|
else if (next) next.start = Math.min(next.start, w.start);
|
|
200
154
|
words.splice(i, 1);
|
|
201
155
|
}
|
|
202
|
-
//
|
|
203
|
-
//
|
|
204
|
-
//
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
for (let i = 0; i < kept.length; i++) {
|
|
208
|
-
const w = kept[i]!;
|
|
209
|
-
if (w.end <= w.start) w.end = w.start + 0.05;
|
|
210
|
-
const next = kept[i + 1];
|
|
211
|
-
if (next && next.start < w.end) next.start = w.end;
|
|
212
|
-
}
|
|
213
|
-
return { language: json.result?.language ?? "en", words: kept };
|
|
156
|
+
// Burst drop + stamp repair now live in `normalizeWords` (provider.ts,
|
|
157
|
+
// 2026-09-01): the remote backend needs exactly the same two passes in
|
|
158
|
+
// exactly the same order, and duplicating them is how the two paths would
|
|
159
|
+
// drift. Behavior here is unchanged — the parser matrix pins it.
|
|
160
|
+
return { language: json.result?.language ?? "en", words: normalizeWords(words) };
|
|
214
161
|
}
|
|
215
162
|
|
|
216
163
|
export interface WhisperOptions {
|