@ossclip/core 0.1.36 → 0.1.38

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ossclip/core",
3
- "version": "0.1.36",
3
+ "version": "0.1.38",
4
4
  "description": "ossclip's framework-free pipeline: schema, transcription, analysis, cutlist, captions, framing, and the LLM producer",
5
5
  "type": "module",
6
6
  "license": "MIT",
package/src/assemble.ts CHANGED
@@ -136,6 +136,13 @@ export interface SfxCue {
136
136
  atSec: number;
137
137
  /** The sound's own gain times the placement's, resolved once here. */
138
138
  gain: number;
139
+ /**
140
+ * The sound's own length, when its pack declared one. Carried into
141
+ * render-props so the renderer can BOUND the cue's Sequence — an unbounded
142
+ * one kept every started effect's `<Audio>` mounted and exhausted the
143
+ * editor Player's shared audio tags (scenes' `SFX_FALLBACK_DURATION_SEC`).
144
+ */
145
+ durationSec?: number;
139
146
  }
140
147
 
141
148
  /** Where staged sounds live inside the render's public dir. */
@@ -310,6 +317,11 @@ export function resolveSfxCues(
310
317
  // mixing arithmetic of its own, so what the editor shows as a gain and
311
318
  // what the render plays can never disagree.
312
319
  gain: sound.gain * (p.gain ?? 1),
320
+ // Optional all the way down: a pack that declares no length gets the
321
+ // renderer's fallback bound, which is why an OLD render-props.json
322
+ // (every one written before 2026-09-18) needs no re-produce to stop
323
+ // taking the preview down.
324
+ ...(sound.durationSec === undefined ? {} : { durationSec: sound.durationSec }),
313
325
  });
314
326
  }
315
327
 
package/src/config.ts CHANGED
@@ -199,6 +199,30 @@ export interface OssclipConfig {
199
199
  * (`publishConfigured` in the CLI), never coerced.
200
200
  */
201
201
  postizUrl?: string;
202
+ /**
203
+ * Base URL of an OpenAI-compatible transcription server, ending in `/v1`
204
+ * (Groq: "https://api.groq.com/openai/v1"; a self-hosted speaches:
205
+ * "http://localhost:8000/v1"). Set, transcription runs REMOTELY instead of
206
+ * on this machine's CPU — the 2026-09-01 field report from an i3 2nd gen,
207
+ * where whisper is the dominant cost of a run. Unset, local whisper-cli
208
+ * stays the default, and `--whisper-backend local` overrides per run.
209
+ *
210
+ * Non-secret, the `postizUrl` posture, so it may live here; the API key is
211
+ * `OSSCLIP_WHISPER_API_KEY` in the ENVIRONMENT only (env.ts's documented
212
+ * rule) — and OPTIONAL, because self-hosted servers run keyless. Validated
213
+ * at the consumer (`resolveWhisperBackend` in the CLI), never coerced.
214
+ */
215
+ whisperUrl?: string;
216
+ /**
217
+ * Model name sent to that server ("whisper-large-v3-turbo" on Groq,
218
+ * "Systran/faster-whisper-large-v3" on a speaches box). Deliberately NOT
219
+ * `model`, which names the local ggml file — a remote run must not be able
220
+ * to send "small.en" to a server that has never heard of it. The default
221
+ * lives at the consumer (`resolveWhisperBackend`), not in DEFAULTS: an
222
+ * unset key here must stay unset so nothing writes a remote model name into
223
+ * a local-only config.
224
+ */
225
+ whisperRemoteModel?: string;
202
226
  /**
203
227
  * `--resolution`'s default for this machine: "auto" (keep what the source
204
228
  * has, capped at 2160), "1080" (the built-in default), "1440" or "2160".
@@ -367,6 +391,15 @@ export function resolveConfig(
367
391
  // deliberately lives in the environment (publish.ts's
368
392
  // `publishConfigured`), so this is only the instance URL.
369
393
  postizUrl: fileCfg.postizUrl,
394
+ // Env spellings ON PURPOSE, unlike postizUrl: the Groq quickstart is
395
+ // "export two vars and run", and a user trying a free tier for the first
396
+ // time should not have to hand-edit config.json to do it. Env beats file
397
+ // so a one-off `OSSCLIP_WHISPER_URL=… ossclip produce` works against a
398
+ // machine whose config says otherwise. Both are strings straight through
399
+ // — validated at the consumer (`resolveWhisperBackend`), never coerced,
400
+ // and the postizUrl lesson above is why they are copied here at all.
401
+ whisperUrl: env.OSSCLIP_WHISPER_URL ?? fileCfg.whisperUrl,
402
+ whisperRemoteModel: env.OSSCLIP_WHISPER_REMOTE_MODEL ?? fileCfg.whisperRemoteModel,
370
403
  resolution: fileCfg.resolution,
371
404
  };
372
405
  }
package/src/ingest.ts CHANGED
@@ -85,6 +85,42 @@ export async function extractAudio(tools: IngestTools, src: string, outWav: stri
85
85
  ]);
86
86
  }
87
87
 
88
+ /**
89
+ * Pure: ffmpeg args for the compressed remote-upload sidecar (2026-09-01,
90
+ * the remote-transcription backend). Split from the spawn the same way
91
+ * `openCommand()` is split from `openInBrowser()` — the codec choice is the
92
+ * whole decision here, so it must be assertable without an ffmpeg on the box.
93
+ *
94
+ * Opus 32 kbps 16 kHz mono: ASR-transparent for speech and ~240 KB/min, so
95
+ * ~100 minutes of audio fit under Groq's free-tier 25 MB per-file cap. The
96
+ * PCM wav `extractAudio` writes is 1.92 MB/min and hits that cap at ~13
97
+ * minutes — unusable for a normal recording session. FLAC was rejected:
98
+ * lossless buys ASR nothing and only reaches ~25 minutes.
99
+ */
100
+ export function uploadAudioArgs(wavPath: string, outOgg: string): string[] {
101
+ return ["-y", "-i", wavPath, "-vn", "-c:a", "libopus", "-b:a", "32k", "-ar", "16000", "-ac", "1", outOgg];
102
+ }
103
+
104
+ /**
105
+ * Encode the remote-upload sidecar. libopus is in the bundled static ffmpeg;
106
+ * a user's own minimal build may lack it, in which case `run` surfaces
107
+ * ffmpeg's own "Unknown encoder" verbatim — better than a guess at the cause.
108
+ */
109
+ export async function encodeUploadAudio(tools: IngestTools, wav: string, outOgg: string): Promise<void> {
110
+ await run(tools.ffmpegPath, uploadAudioArgs(wav, outOgg));
111
+ }
112
+
113
+ /**
114
+ * 24 million bytes: Groq's free tier caps a file at "25MB" without saying
115
+ * whether that is decimal or MiB — 24_000_000 sits under BOTH readings
116
+ * (25,000,000 and 26,214,336), so a boundary file fails HERE with our message
117
+ * (naming the ~100-minute ceiling and the local-backend escape hatch) instead
118
+ * of coming back as somebody else's 413 after the whole upload was paid for.
119
+ * (24 MiB = 25,165,824 would EXCEED a decimal cap — caught 2026-09-01.)
120
+ * Chunking is out of scope for v1.
121
+ */
122
+ export const REMOTE_UPLOAD_MAX_BYTES = 24_000_000;
123
+
88
124
  /**
89
125
  * Extract ONE span of an existing wav, same 16 kHz mono PCM shape
90
126
  * (2026-08-26, the caption re-alignment pass).
@@ -0,0 +1,3 @@
1
+ export * from "./whisper-cli";
2
+ export * from "./provider";
3
+ export * from "./openai-compatible";
@@ -0,0 +1,199 @@
1
+ import { basename } from "node:path";
2
+ import { z } from "zod/v4";
3
+ import { TranscriptSchema, type Transcript, type Word } from "../schema";
4
+ import { NOISE_TOKEN, normalizeWords, type TranscribeProvider, type TranscribeRequest } from "./provider";
5
+
6
+ /**
7
+ * Transcription against any OpenAI-compatible `/v1/audio/transcriptions`
8
+ * server — Groq's free tier (8h audio/day, `whisper-large-v3-turbo`), a
9
+ * self-hosted speaches, or anything else speaking that shape.
10
+ *
11
+ * Why (2026-09-01 field report): on a weak CPU — an i3 2nd gen — whisper is
12
+ * the dominant cost of a produce run, minutes of decode per minute of video.
13
+ * A remote call makes it seconds. Local whisper-cli stays the DEFAULT; this
14
+ * is opt-in via `whisperUrl` / `OSSCLIP_WHISPER_URL`, and the API key is
15
+ * optional because self-hosted servers run keyless (unlike publish, where the
16
+ * key is required).
17
+ *
18
+ * Error posture is postiz.ts's, for the same reason: a transcription is the
19
+ * user's explicit action, so every non-2xx throws with the status and a body
20
+ * snippet, and there are NO retries — a retry against a metered free tier
21
+ * silently doubles the quota burn for a failure the user is about to see
22
+ * anyway.
23
+ */
24
+
25
+ /**
26
+ * `whisperUrl` → the endpoint: trailing slashes dropped,
27
+ * `/audio/transcriptions` appended unless the user already wrote it
28
+ * (`postizApiBase`'s mould). The user configures the OpenAI-compatible BASE
29
+ * ("https://api.groq.com/openai/v1"), which is the URL every provider's
30
+ * quickstart prints — but someone who pastes the full endpoint must not end
31
+ * up posting to `/v1/audio/transcriptions/audio/transcriptions`.
32
+ */
33
+ export function openaiTranscriptionsUrl(baseUrl: string): string {
34
+ const trimmed = baseUrl.replace(/\/+$/, "");
35
+ return trimmed.endsWith("/audio/transcriptions") ? trimmed : `${trimmed}/audio/transcriptions`;
36
+ }
37
+
38
+ export class RemoteTranscribeHttpError extends Error {
39
+ constructor(
40
+ readonly url: string,
41
+ readonly status: number,
42
+ bodySnippet: string,
43
+ ) {
44
+ // Per-status hints, PostizHttpError's mould: the raw status tells a user
45
+ // nothing about which of the two env vars (or which URL spelling) is
46
+ // wrong, and this path is reached by people who just pasted a quickstart.
47
+ const hint =
48
+ status === 401 || status === 403
49
+ ? " — the server rejected OSSCLIP_WHISPER_API_KEY (or none was sent — set it in the environment or ~/.ossclip/.env)"
50
+ : status === 404
51
+ ? " — no /audio/transcriptions here — whisperUrl should be the OpenAI-compatible base ending in /v1 (e.g. https://api.groq.com/openai/v1)"
52
+ : status === 413
53
+ ? " — audio too large for this server — free Groq caps uploads at 25MB; use --whisper-backend local or the dev tier"
54
+ : status === 429
55
+ ? " — rate limited (Groq free tier: 8h audio/day)"
56
+ : "";
57
+ super(`remote transcription POST ${url} failed: ${status}${hint}${bodySnippet ? `\n${bodySnippet}` : ""}`);
58
+ this.name = "RemoteTranscribeHttpError";
59
+ }
60
+ }
61
+
62
+ /**
63
+ * `verbose_json` as we consume it. LOOSE on purpose: servers add fields
64
+ * freely (Groq sends `task`, `duration`, `segments`, `x_groq`), and a strict
65
+ * object would turn a perfectly good transcription into a parse error the
66
+ * next time one of them ships a field.
67
+ */
68
+ const RemoteWordSchema = z.looseObject({
69
+ word: z.string(),
70
+ start: z.number(),
71
+ end: z.number(),
72
+ });
73
+ const VerboseJsonSchema = z.looseObject({
74
+ language: z.string().optional(),
75
+ words: z.array(RemoteWordSchema).optional(),
76
+ });
77
+
78
+ export interface OpenAiCompatibleOptions {
79
+ /** OpenAI-compatible base, e.g. "https://api.groq.com/openai/v1". */
80
+ baseUrl: string;
81
+ model: string;
82
+ /** Optional: self-hosted servers (speaches, whisper.cpp server) run keyless. */
83
+ apiKey?: string;
84
+ /** The postiz test seam — the whole HTTP surface is testable without a network. */
85
+ fetchImpl?: typeof fetch;
86
+ /** Per-request cap; an hour of audio takes a while to upload and decode. */
87
+ timeoutMs?: number;
88
+ }
89
+
90
+ const DEFAULT_TIMEOUT_MS = 10 * 60 * 1000;
91
+ const BODY_SNIPPET_CHARS = 300;
92
+
93
+ export function createOpenAiCompatibleProvider(opts: OpenAiCompatibleOptions): TranscribeProvider {
94
+ const url = openaiTranscriptionsUrl(opts.baseUrl);
95
+ const fetchImpl = opts.fetchImpl ?? fetch;
96
+ const timeoutMs = opts.timeoutMs ?? DEFAULT_TIMEOUT_MS;
97
+
98
+ return {
99
+ name: "openai-compatible",
100
+ async transcribe(audioPath: string, req: TranscribeRequest): Promise<Transcript> {
101
+ // Belt and braces — the CLI refuses this combination earlier, with the
102
+ // fix named. `/audio/translations` is a different endpoint AND a
103
+ // different default model upstream, so silently swapping both behind
104
+ // `--whisper-translate` would be a surprise, not a convenience.
105
+ if (req.translate === true) {
106
+ throw new Error(
107
+ "--whisper-translate needs the local backend (the OpenAI-compatible API translates on a different endpoint and model) — use --whisper-backend local, or drop the flag.",
108
+ );
109
+ }
110
+ const { openAsBlob } = await import("node:fs");
111
+ // Streams the file into multipart form-data instead of holding it in
112
+ // memory (postiz.ts's rationale): the upload sidecar is small, but a
113
+ // span wav or an uncompressed hour is not.
114
+ const blob = await openAsBlob(audioPath, {
115
+ type: audioPath.endsWith(".ogg") ? "audio/ogg" : "audio/wav",
116
+ });
117
+ const form = new FormData();
118
+ form.append("file", blob, basename(audioPath));
119
+ form.append("model", opts.model);
120
+ form.append("response_format", "verbose_json");
121
+ // The literal bracketed field name is the wire spelling OpenAI and Groq
122
+ // accept — it is an array parameter in a multipart body, not a typo.
123
+ form.append("timestamp_granularities[]", "word");
124
+ // "auto" is whisper.cpp's vocabulary, not an ISO code: sending it makes
125
+ // the server reject the request, while OMITTING the field is exactly
126
+ // what asks for auto-detection.
127
+ if (req.language !== undefined && req.language !== "auto") form.append("language", req.language);
128
+ if (req.prompt !== undefined) form.append("prompt", req.prompt);
129
+
130
+ const ac = new AbortController();
131
+ const timer = setTimeout(() => ac.abort(), timeoutMs);
132
+ let res: Response;
133
+ try {
134
+ res = await fetchImpl(url, {
135
+ method: "POST",
136
+ headers: opts.apiKey ? { Authorization: `Bearer ${opts.apiKey}` } : {},
137
+ body: form,
138
+ signal: ac.signal,
139
+ });
140
+ } catch (err) {
141
+ throw new Error(
142
+ `remote transcription unreachable at ${url}: ${err instanceof Error ? err.message : String(err)}`,
143
+ );
144
+ } finally {
145
+ clearTimeout(timer);
146
+ }
147
+ const text = await res.text();
148
+ if (!res.ok) throw new RemoteTranscribeHttpError(url, res.status, text.slice(0, BODY_SNIPPET_CHARS));
149
+ let json: unknown;
150
+ try {
151
+ json = JSON.parse(text);
152
+ } catch {
153
+ throw new Error(
154
+ `remote transcription answered non-JSON from ${url}: ${text.slice(0, BODY_SNIPPET_CHARS)}`,
155
+ );
156
+ }
157
+ const parsed = VerboseJsonSchema.parse(json);
158
+ if (!parsed.words || parsed.words.length === 0) {
159
+ // Two causes, both worth naming: a server that transcribed fine but
160
+ // has no word-timestamp support (a plain whisper.cpp server), and
161
+ // near-silent audio, which Groq's turbo model answers wordlessly.
162
+ // Everything downstream (cuts, captions, zoom) is word-stamp driven,
163
+ // so a text-only answer is unusable, not a degraded success.
164
+ throw new Error(
165
+ `the server answered without word timestamps — it must support response_format=verbose_json with timestamp_granularities[]=word (Groq and speaches do; a plain whisper.cpp server may not), or the audio contained no speech (${url})`,
166
+ );
167
+ }
168
+ // No token merging and no §130 byte repair here: those heal whisper.cpp
169
+ // `-ml 1` artifacts (one BPE token per segment, split mid-character).
170
+ // The HTTP API returns whole words with punctuation attached — already
171
+ // the shape parseWhisperJson works to produce.
172
+ const words: Word[] = [];
173
+ for (const w of parsed.words) {
174
+ const wordText = w.word.trim();
175
+ if (!wordText || NOISE_TOKEN.test(wordText)) continue;
176
+ words.push({
177
+ text: wordText,
178
+ // Clamped: a server answering -0.01 for the first word would trip
179
+ // WordSchema's nonnegative and fail the whole run over a rounding
180
+ // artifact at the very start of the audio.
181
+ start: Math.max(0, w.start),
182
+ end: w.end,
183
+ });
184
+ }
185
+ return TranscriptSchema.parse({
186
+ // The requested code wins — it is what the cache key and the caption
187
+ // pipeline were told the audio is. Otherwise the server's own answer,
188
+ // lowercased. NOTE: some servers answer with a full NAME ("english")
189
+ // rather than a code; no code-sensitive consumer exists today
190
+ // (captions' RTL check is a Unicode heuristic), so no name→code table.
191
+ language:
192
+ req.language !== undefined && req.language !== "auto"
193
+ ? req.language
194
+ : parsed.language?.toLowerCase(),
195
+ words: normalizeWords(words),
196
+ });
197
+ },
198
+ };
199
+ }
@@ -0,0 +1,101 @@
1
+ import type { Transcript, Word } from "../schema";
2
+
3
+ /**
4
+ * The seam between ossclip and any transcription backend (2026-09-01, the
5
+ * weak-CPU field report: whisper is the dominant cost on an i3 2nd gen, and a
6
+ * free remote tier makes it disappear). Two implementations today —
7
+ * whisper.cpp on the box (`whisper-cli.ts`) and any OpenAI-compatible
8
+ * `/v1/audio/transcriptions` server (`openai-compatible.ts`).
9
+ *
10
+ * The local path deliberately keeps calling `runWhisper` directly rather than
11
+ * going through this interface: produce.ts and the edit server inject
12
+ * `runWhisper` itself as a test seam, and forcing that through a provider
13
+ * object would churn every stub for no behavior change.
14
+ */
15
+ export interface TranscribeRequest {
16
+ /** Resolved language code; "auto" is handled per provider (whisper.cpp
17
+ * takes it literally, the HTTP API wants the field OMITTED). */
18
+ language?: string;
19
+ /** `whisperPromptFor()` output — the user dictionary as decoder bias. */
20
+ prompt?: string;
21
+ /** whisper-cli's TRANSLATE task; the OpenAI-compatible provider rejects it
22
+ * (a different endpoint AND a different default model upstream). */
23
+ translate?: boolean;
24
+ }
25
+
26
+ export interface TranscribeProvider {
27
+ name: string;
28
+ transcribe(audioPath: string, req: TranscribeRequest): Promise<Transcript>;
29
+ }
30
+
31
+ /** [BLANK_AUDIO], (buzzing), [MUSIC] … — noise markers, not speech. Shared:
32
+ * remote servers emit the same bracketed markers whisper.cpp does. */
33
+ export const NOISE_TOKEN = /^[[(].*[\])]$/;
34
+
35
+ /**
36
+ * Run length at which a stack of zero-length words at ONE instant stops being
37
+ * a rounding artifact and becomes a repetition-loop hallucination. Real speech
38
+ * never emits 8 tokens at a single instant; the field case emitted 118.
39
+ */
40
+ export const REPETITION_BURST_MIN = 8;
41
+
42
+ /**
43
+ * Drop whisper repetition-loop bursts (field case 2026-08-18): an Urdu take
44
+ * re-decoded a whole phrase as 118 CONSECUTIVE tokens all stamped
45
+ * `from === to === 31040` — zero length, at one instant. The stamp repair
46
+ * below then fans such a burst out into 118 fabricated 50ms words marching
47
+ * forward from 31.04s, so the phrase ships TWICE in the captions (31.04s and
48
+ * 33.54s) and a fifth of the transcript carries the tell-tale exactly-0.05s
49
+ * duration. `-mc 0` in whisperArgs is the decoder-side mitigation for the same
50
+ * failure; it did not prevent this occurrence, and it can never repair an
51
+ * already-cached transcript.json — hence a parse-side guard too.
52
+ *
53
+ * A burst is a MAXIMAL run of consecutive zero-length/inverted words sharing
54
+ * one `start`. Equality is exact, not epsilon: these stamps are integer
55
+ * milliseconds divided by 1000, so members of one burst are the same double
56
+ * bit-for-bit, and a tolerance would only start swallowing real neighbors.
57
+ * Runs shorter than REPETITION_BURST_MIN fall through untouched — a lone
58
+ * zero-length stamp is a rounding artifact, not a hallucination. The drop is
59
+ * silent by design: this function is pure and total, and there is no logging
60
+ * channel in the parse path to warn on.
61
+ */
62
+ export function dropRepetitionBursts(words: readonly Word[]): Word[] {
63
+ const out: Word[] = [];
64
+ let i = 0;
65
+ while (i < words.length) {
66
+ const w = words[i]!;
67
+ if (w.end > w.start) {
68
+ out.push(w);
69
+ i++;
70
+ continue;
71
+ }
72
+ let j = i + 1;
73
+ while (j < words.length && words[j]!.end <= words[j]!.start && words[j]!.start === w.start) j++;
74
+ if (j - i < REPETITION_BURST_MIN) for (let k = i; k < j; k++) out.push(words[k]!);
75
+ i = j;
76
+ }
77
+ return out;
78
+ }
79
+
80
+ /**
81
+ * The word-stamp hygiene EVERY backend's output goes through, extracted from
82
+ * the tail of `parseWhisperJson` so the remote provider cannot drift from it
83
+ * (2026-09-01). Burst drop runs BEFORE the repair, never after: the repair
84
+ * rewrites every burst member into a distinct monotone stamp, so once it has
85
+ * run the shared timestamp — the only evidence a burst existed — is gone.
86
+ *
87
+ * Copies before mutating, so a caller's array survives the call unchanged;
88
+ * `parseWhisperJson` builds its words fresh, so this is byte-identical to the
89
+ * in-place loop it replaces.
90
+ */
91
+ export function normalizeWords(words: readonly Word[]): Word[] {
92
+ const kept = dropRepetitionBursts(words).map((w) => ({ ...w }));
93
+ // Whisper occasionally emits zero-length or inverted stamps; repair minimally.
94
+ for (let i = 0; i < kept.length; i++) {
95
+ const w = kept[i]!;
96
+ if (w.end <= w.start) w.end = w.start + 0.05;
97
+ const next = kept[i + 1];
98
+ if (next && next.start < w.end) next.start = w.end;
99
+ }
100
+ return kept;
101
+ }
@@ -1,6 +1,7 @@
1
1
  import { readFile } from "node:fs/promises";
2
- import { run } from "./exec";
3
- import type { Transcript, Word } from "./schema";
2
+ import { run } from "../exec";
3
+ import type { Transcript, Word } from "../schema";
4
+ import { NOISE_TOKEN, normalizeWords } from "./provider";
4
5
 
5
6
  /** Shape of whisper.cpp's `-oj` JSON output (the fields we consume). */
6
7
  export interface WhisperJson {
@@ -11,8 +12,6 @@ export interface WhisperJson {
11
12
  }>;
12
13
  }
13
14
 
14
- const NOISE_TOKEN = /^[[(].*[\])]$/; // [BLANK_AUDIO], (buzzing), [MUSIC] …
15
-
16
15
  /**
17
16
  * How many bytes at the END of `bytes` form the start of a multi-byte UTF-8
18
17
  * character whose continuation bytes are missing (§130: whisper.cpp `-ml 1`
@@ -78,51 +77,6 @@ function repairSplitSegments(json: WhisperJson): WhisperJson {
78
77
  };
79
78
  }
80
79
 
81
- /**
82
- * Run length at which a stack of zero-length words at ONE instant stops being
83
- * a rounding artifact and becomes a repetition-loop hallucination. Real speech
84
- * never emits 8 tokens at a single instant; the field case emitted 118.
85
- */
86
- export const REPETITION_BURST_MIN = 8;
87
-
88
- /**
89
- * Drop whisper repetition-loop bursts (field case 2026-08-18): an Urdu take
90
- * re-decoded a whole phrase as 118 CONSECUTIVE tokens all stamped
91
- * `from === to === 31040` — zero length, at one instant. The stamp repair
92
- * below then fans such a burst out into 118 fabricated 50ms words marching
93
- * forward from 31.04s, so the phrase ships TWICE in the captions (31.04s and
94
- * 33.54s) and a fifth of the transcript carries the tell-tale exactly-0.05s
95
- * duration. `-mc 0` in whisperArgs is the decoder-side mitigation for the same
96
- * failure; it did not prevent this occurrence, and it can never repair an
97
- * already-cached transcript.json — hence a parse-side guard too.
98
- *
99
- * A burst is a MAXIMAL run of consecutive zero-length/inverted words sharing
100
- * one `start`. Equality is exact, not epsilon: these stamps are integer
101
- * milliseconds divided by 1000, so members of one burst are the same double
102
- * bit-for-bit, and a tolerance would only start swallowing real neighbors.
103
- * Runs shorter than REPETITION_BURST_MIN fall through untouched — a lone
104
- * zero-length stamp is a rounding artifact, not a hallucination. The drop is
105
- * silent by design: this function is pure and total, and there is no logging
106
- * channel in the parse path to warn on.
107
- */
108
- export function dropRepetitionBursts(words: readonly Word[]): Word[] {
109
- const out: Word[] = [];
110
- let i = 0;
111
- while (i < words.length) {
112
- const w = words[i]!;
113
- if (w.end > w.start) {
114
- out.push(w);
115
- i++;
116
- continue;
117
- }
118
- let j = i + 1;
119
- while (j < words.length && words[j]!.end <= words[j]!.start && words[j]!.start === w.start) j++;
120
- if (j - i < REPETITION_BURST_MIN) for (let k = i; k < j; k++) out.push(words[k]!);
121
- i = j;
122
- }
123
- return out;
124
- }
125
-
126
80
  const STRICT_UTF8 = new TextDecoder("utf-8", { fatal: true });
127
81
 
128
82
  /**
@@ -199,18 +153,11 @@ export function parseWhisperJson(json: WhisperJson): Transcript {
199
153
  else if (next) next.start = Math.min(next.start, w.start);
200
154
  words.splice(i, 1);
201
155
  }
202
- // BEFORE the repair loop, never after: the repair rewrites every burst
203
- // member into a distinct monotone stamp, so once it has run the shared
204
- // timestamp — the only evidence a burst existed — is gone.
205
- const kept = dropRepetitionBursts(words);
206
- // Whisper occasionally emits zero-length or inverted stamps; repair minimally.
207
- for (let i = 0; i < kept.length; i++) {
208
- const w = kept[i]!;
209
- if (w.end <= w.start) w.end = w.start + 0.05;
210
- const next = kept[i + 1];
211
- if (next && next.start < w.end) next.start = w.end;
212
- }
213
- return { language: json.result?.language ?? "en", words: kept };
156
+ // Burst drop + stamp repair now live in `normalizeWords` (provider.ts,
157
+ // 2026-09-01): the remote backend needs exactly the same two passes in
158
+ // exactly the same order, and duplicating them is how the two paths would
159
+ // drift. Behavior here is unchanged — the parser matrix pins it.
160
+ return { language: json.result?.language ?? "en", words: normalizeWords(words) };
214
161
  }
215
162
 
216
163
  export interface WhisperOptions {