@ossclip/core 0.1.35 → 0.1.37
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/browser.ts +15 -0
- package/src/color-grade.ts +562 -0
- package/src/config.ts +55 -0
- package/src/index.ts +2 -0
- package/src/ingest.ts +87 -8
- package/src/lut-library.ts +81 -0
- package/src/overrides.ts +32 -0
- package/src/recut.ts +32 -3
- package/src/scene-schema.ts +3 -0
- package/src/transcribe/index.ts +3 -0
- package/src/transcribe/openai-compatible.ts +199 -0
- package/src/transcribe/provider.ts +101 -0
- package/src/{transcribe.ts → transcribe/whisper-cli.ts} +8 -61
package/src/ingest.ts
CHANGED
|
@@ -85,6 +85,42 @@ export async function extractAudio(tools: IngestTools, src: string, outWav: stri
|
|
|
85
85
|
]);
|
|
86
86
|
}
|
|
87
87
|
|
|
88
|
+
/**
|
|
89
|
+
* Pure: ffmpeg args for the compressed remote-upload sidecar (2026-09-01,
|
|
90
|
+
* the remote-transcription backend). Split from the spawn the same way
|
|
91
|
+
* `openCommand()` is split from `openInBrowser()` — the codec choice is the
|
|
92
|
+
* whole decision here, so it must be assertable without an ffmpeg on the box.
|
|
93
|
+
*
|
|
94
|
+
* Opus 32 kbps 16 kHz mono: ASR-transparent for speech and ~240 KB/min, so
|
|
95
|
+
* ~100 minutes of audio fit under Groq's free-tier 25 MB per-file cap. The
|
|
96
|
+
* PCM wav `extractAudio` writes is 1.92 MB/min and hits that cap at ~13
|
|
97
|
+
* minutes — unusable for a normal recording session. FLAC was rejected:
|
|
98
|
+
* lossless buys ASR nothing and only reaches ~25 minutes.
|
|
99
|
+
*/
|
|
100
|
+
export function uploadAudioArgs(wavPath: string, outOgg: string): string[] {
|
|
101
|
+
return ["-y", "-i", wavPath, "-vn", "-c:a", "libopus", "-b:a", "32k", "-ar", "16000", "-ac", "1", outOgg];
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Encode the remote-upload sidecar. libopus is in the bundled static ffmpeg;
|
|
106
|
+
* a user's own minimal build may lack it, in which case `run` surfaces
|
|
107
|
+
* ffmpeg's own "Unknown encoder" verbatim — better than a guess at the cause.
|
|
108
|
+
*/
|
|
109
|
+
export async function encodeUploadAudio(tools: IngestTools, wav: string, outOgg: string): Promise<void> {
|
|
110
|
+
await run(tools.ffmpegPath, uploadAudioArgs(wav, outOgg));
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* 24 million bytes: Groq's free tier caps a file at "25MB" without saying
|
|
115
|
+
* whether that is decimal or MiB — 24_000_000 sits under BOTH readings
|
|
116
|
+
* (25,000,000 and 26,214,336), so a boundary file fails HERE with our message
|
|
117
|
+
* (naming the ~100-minute ceiling and the local-backend escape hatch) instead
|
|
118
|
+
* of coming back as somebody else's 413 after the whole upload was paid for.
|
|
119
|
+
* (24 MiB = 25,165,824 would EXCEED a decimal cap — caught 2026-09-01.)
|
|
120
|
+
* Chunking is out of scope for v1.
|
|
121
|
+
*/
|
|
122
|
+
export const REMOTE_UPLOAD_MAX_BYTES = 24_000_000;
|
|
123
|
+
|
|
88
124
|
/**
|
|
89
125
|
* Extract ONE span of an existing wav, same 16 kHz mono PCM shape
|
|
90
126
|
* (2026-08-26, the caption re-alignment pass).
|
|
@@ -190,11 +226,56 @@ export function mezzanineScale(
|
|
|
190
226
|
* the legacy names so existing workdir caches stay valid; a scaled run
|
|
191
227
|
* rebuilds once under its own name and old workdirs' render-props keep
|
|
192
228
|
* referencing (and rendering from) the file they were emitted against.
|
|
229
|
+
*
|
|
230
|
+
* The LUT hash is in the name for the same reason: grading is baked into the
|
|
231
|
+
* mezzanine at build time, so a warm workdir keyed only on crop/scale would
|
|
232
|
+
* satisfy a graded run with UNGRADED frames (or a re-graded run with the old
|
|
233
|
+
* look). No LUT keeps today's names byte-for-byte, so existing warm workdirs
|
|
234
|
+
* stay valid.
|
|
193
235
|
*/
|
|
194
|
-
export function mezzanineFileName(cropped: boolean, scale: MezzanineScale | null): string {
|
|
236
|
+
export function mezzanineFileName(cropped: boolean, scale: MezzanineScale | null, lutHash?: string): string {
|
|
195
237
|
const base = cropped ? "mezzanine-content" : "mezzanine";
|
|
196
|
-
|
|
197
|
-
|
|
238
|
+
const scaleSeg = scale ? `-${scale.width}x${scale.height}@${Math.round(scale.fps)}` : "";
|
|
239
|
+
const lutSeg = lutHash ? `-lut${lutHash}` : "";
|
|
240
|
+
return `${base}${scaleSeg}${lutSeg}.mp4`;
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
/** A 3D LUT to bake into the mezzanine; `hash` keys the cache (see `mezzanineFileName`). */
|
|
244
|
+
export interface MezzanineLut {
|
|
245
|
+
path: string;
|
|
246
|
+
hash: string;
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
/**
|
|
250
|
+
* Escape a filesystem path for use as an ffmpeg filter option value.
|
|
251
|
+
*
|
|
252
|
+
* A `-vf` string is parsed twice: once as a filtergraph (where `\` `'` `[`
|
|
253
|
+
* `]` `,` `;` are special) and once as the filter's option value (where `:`
|
|
254
|
+
* `\` `'` are special — `:` is the option separator, so an unescaped drive
|
|
255
|
+
* letter like `C:` truncates the path there). Each level strips one layer of
|
|
256
|
+
* backslashes, so the option-level escapes must themselves be escaped for
|
|
257
|
+
* the graph level: `:` → `\\:`, `'` → `\\\'`, `\` → `\\\\`. Spaces need
|
|
258
|
+
* nothing — the argv goes straight to ffmpeg, no shell in between.
|
|
259
|
+
*/
|
|
260
|
+
export function escapeFilterPath(p: string): string {
|
|
261
|
+
// Level 1: filter option value — `:` `\` `'` are special.
|
|
262
|
+
const option = p.replace(/[\\':]/g, (c) => `\\${c}`);
|
|
263
|
+
// Level 2: filtergraph — escape again so level-1 backslashes survive.
|
|
264
|
+
return option.replace(/[\\'[\],;]/g, (c) => `\\${c}`);
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
/**
|
|
268
|
+
* The mezzanine's `-vf` chain, pure so the ordering contract is testable:
|
|
269
|
+
* LUT strictly AFTER crop/scale — grading the letterbox bars would be
|
|
270
|
+
* wasted math, and grading pre-scale pixels the render never sees changes
|
|
271
|
+
* nothing but costs full-res per-pixel lookups.
|
|
272
|
+
*/
|
|
273
|
+
export function mezzanineVf(opts: { cropVf?: string; scale?: MezzanineScale; lut?: MezzanineLut }): string {
|
|
274
|
+
return [
|
|
275
|
+
...(opts.cropVf ? [opts.cropVf] : []),
|
|
276
|
+
...(opts.scale ? [`scale=${opts.scale.width}:${opts.scale.height}`] : []),
|
|
277
|
+
...(opts.lut ? [`lut3d=file=${escapeFilterPath(opts.lut.path)}:interp=tetrahedral`] : []),
|
|
278
|
+
].join(",");
|
|
198
279
|
}
|
|
199
280
|
|
|
200
281
|
/**
|
|
@@ -206,17 +287,15 @@ export function mezzanineFileName(cropped: boolean, scale: MezzanineScale | null
|
|
|
206
287
|
*
|
|
207
288
|
* `scale` (from `mezzanineScale`) downsizes to display size in the SAME
|
|
208
289
|
* pass, crop first — the scale dims are computed on the post-crop picture.
|
|
290
|
+
* `lut` bakes a 3D grade in last, on exactly the pixels the render will see.
|
|
209
291
|
*/
|
|
210
292
|
export async function makeMezzanine(
|
|
211
293
|
tools: IngestTools,
|
|
212
294
|
src: string,
|
|
213
295
|
out: string,
|
|
214
|
-
opts: { cropVf?: string; scale?: MezzanineScale } = {},
|
|
296
|
+
opts: { cropVf?: string; scale?: MezzanineScale; lut?: MezzanineLut } = {},
|
|
215
297
|
): Promise<void> {
|
|
216
|
-
const vf =
|
|
217
|
-
...(opts.cropVf ? [opts.cropVf] : []),
|
|
218
|
-
...(opts.scale ? [`scale=${opts.scale.width}:${opts.scale.height}`] : []),
|
|
219
|
-
].join(",");
|
|
298
|
+
const vf = mezzanineVf(opts);
|
|
220
299
|
await run(tools.ffmpegPath, [
|
|
221
300
|
"-y", "-i", src,
|
|
222
301
|
...(vf ? ["-vf", vf] : []),
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
import { readFileSync, readdirSync } from "node:fs";
|
|
2
|
+
import { basename, extname, join } from "node:path";
|
|
3
|
+
import { CONFIG_DIR } from "./config";
|
|
4
|
+
import { parseCubeLut } from "./color-grade";
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* The user's .cube LUT directory, discovered — the `sfx-pack.ts` shape for
|
|
8
|
+
* color grades. `parseCubeLut` stays pure; this module is the thin fs layer
|
|
9
|
+
* that walks `~/.ossclip/luts` and reports, per file, either a usable LUT or
|
|
10
|
+
* the reason it is not one. Nothing here throws: a hand-dropped .cube is user
|
|
11
|
+
* input, and a broken one must cost that one menu entry, not the editor.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
/** Where user LUTs live: `~/.ossclip/luts/<name>.cube`. */
|
|
15
|
+
export function userLutDir(): string {
|
|
16
|
+
return join(CONFIG_DIR, "luts");
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/** One LUT the menu can offer. `path` stays server-side, like SFX `absPath`. */
|
|
20
|
+
export interface LutLibraryItem {
|
|
21
|
+
/** The filename stem — what `ColorGrade.lut` (basename) resolves against. */
|
|
22
|
+
id: string;
|
|
23
|
+
/** The .cube's own TITLE when it has one, else the stem. */
|
|
24
|
+
title: string;
|
|
25
|
+
/** Absolute path — the caller's I/O concern, never sent to a client. */
|
|
26
|
+
path: string;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** Why one file is not in the library — `SfxPackIssue`'s shape, per file. */
|
|
30
|
+
export interface LutLibraryIssue {
|
|
31
|
+
/** The .cube filename (basename) that failed. */
|
|
32
|
+
file: string;
|
|
33
|
+
message: string;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export interface LutLibrary {
|
|
37
|
+
items: LutLibraryItem[];
|
|
38
|
+
issues: LutLibraryIssue[];
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Every parseable `.cube` under `dir`, plus an issue per file that is not one.
|
|
43
|
+
* Each file is fully parsed here — not just listed — because the menu is the
|
|
44
|
+
* ONLY surface where a broken LUT can be reported before a render silently
|
|
45
|
+
* drops it: `parseCubeLut` is strict about content on purpose, and offering a
|
|
46
|
+
* file the bake will refuse is the exact mismatch the SFX library gate exists
|
|
47
|
+
* to avoid.
|
|
48
|
+
*
|
|
49
|
+
* A missing directory is the normal case (most users never drop a LUT), not
|
|
50
|
+
* an issue. `dir` is a parameter with a default rather than a `homedir()`
|
|
51
|
+
* read inside, so tests point it at a tmp dir and never touch a real home
|
|
52
|
+
* (`loadSfxLibrary`'s rule).
|
|
53
|
+
*/
|
|
54
|
+
export function loadLutLibrary(dir: string = userLutDir()): LutLibrary {
|
|
55
|
+
let names: string[] = [];
|
|
56
|
+
try {
|
|
57
|
+
names = readdirSync(dir, { withFileTypes: true })
|
|
58
|
+
.filter((e) => e.isFile() && e.name.toLowerCase().endsWith(".cube"))
|
|
59
|
+
.map((e) => e.name)
|
|
60
|
+
// Sorted so the menu reads the same on every machine — readdir order is
|
|
61
|
+
// not a promise (loadSfxLibrary's merge-order rule).
|
|
62
|
+
.sort();
|
|
63
|
+
} catch {
|
|
64
|
+
return { items: [], issues: [] };
|
|
65
|
+
}
|
|
66
|
+
const items: LutLibraryItem[] = [];
|
|
67
|
+
const issues: LutLibraryIssue[] = [];
|
|
68
|
+
for (const name of names) {
|
|
69
|
+
const path = join(dir, name);
|
|
70
|
+
const stem = basename(name, extname(name));
|
|
71
|
+
try {
|
|
72
|
+
const lut = parseCubeLut(readFileSync(path, "utf8"));
|
|
73
|
+
// TITLE when the exporter wrote one — a human-readable label the stem
|
|
74
|
+
// (often `Vendor_Look_33pt_v2`) cannot match. Empty titles fall back.
|
|
75
|
+
items.push({ id: stem, title: lut.title?.trim() || stem, path });
|
|
76
|
+
} catch (e) {
|
|
77
|
+
issues.push({ file: name, message: e instanceof Error ? e.message : String(e) });
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
return { items, issues };
|
|
81
|
+
}
|
package/src/overrides.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { z } from "zod/v4";
|
|
2
|
+
import { ColorGradeSchema } from "./color-grade";
|
|
2
3
|
import {
|
|
3
4
|
LayoutSchema,
|
|
4
5
|
SceneAnchorSchema,
|
|
@@ -142,6 +143,19 @@ export const SceneOverrideSchema = z.object({
|
|
|
142
143
|
dx: z.number().optional(),
|
|
143
144
|
/** `false` switches the automatic idle-zoom layer off for this scene. */
|
|
144
145
|
autoZoom: z.boolean().optional(),
|
|
146
|
+
/**
|
|
147
|
+
* Audio gain for this window's footage, 1 = as recorded (field report
|
|
148
|
+
* 2026-08-31: one concatenated clip was recorded quieter than the
|
|
149
|
+
* rest). Lives inside `video` on purpose — it is a property of this
|
|
150
|
+
* window's playback, and the key already merges per scene, inherits
|
|
151
|
+
* across split halves, and has patch/clear plumbing. 0 mutes. Above 1
|
|
152
|
+
* amplifies — in the render via allowAmplificationDuringRender, and in
|
|
153
|
+
* the preview via the Player's own WebAudio gain (Remotion ≥4.0.5xx,
|
|
154
|
+
* use-amplification). Max 4: a field clip arrived quiet enough that 2x
|
|
155
|
+
* did not reach its neighbours (2026-08-31); past 4x you are boosting
|
|
156
|
+
* noise floor, not speech.
|
|
157
|
+
*/
|
|
158
|
+
volume: z.number().min(0).max(4).optional(),
|
|
145
159
|
})
|
|
146
160
|
.optional(),
|
|
147
161
|
/**
|
|
@@ -527,6 +541,24 @@ export type SfxAddedPlacement = z.infer<typeof SfxAddedPlacementSchema>;
|
|
|
527
541
|
export const OverrideDocSchema = z.object({
|
|
528
542
|
/** Global style tokens — the look is a system, so these are not per-element. */
|
|
529
543
|
theme: ThemeSchema.partial().default({}),
|
|
544
|
+
/**
|
|
545
|
+
* Doc-global color grade — the `ColorGradeSchema` shape, or `false` for
|
|
546
|
+
* "explicitly no grade on this project". Doc-global like `theme`: a grade
|
|
547
|
+
* is one decision about the whole output, not a per-scene key. Optional
|
|
548
|
+
* with NO default (the `captionsHidden` rule) so every overrides.json
|
|
549
|
+
* written before the key existed parses byte-identically — but UNLIKE
|
|
550
|
+
* `captionsHidden`, an explicit `false` is meaningful and kept: it
|
|
551
|
+
* disables a config-level default grade for this one project, which
|
|
552
|
+
* deleting the key cannot express (absent means "let the flag, then the
|
|
553
|
+
* config, decide" — `resolveProductionColorGrade` in produce.ts owns that
|
|
554
|
+
* precedence). Schema-valid is not yet USABLE: an unknown preset id passes
|
|
555
|
+
* here (the schema cannot list what exists without going stale) and is
|
|
556
|
+
* caught by `resolveColorGrade` at the consumer, where it warns and falls
|
|
557
|
+
* through to the next layer instead of failing the whole doc. Timeless —
|
|
558
|
+
* no seconds anywhere — so `remapOverridesThroughRecut`'s `...doc` spreads
|
|
559
|
+
* carry it through a recut untouched.
|
|
560
|
+
*/
|
|
561
|
+
colorGrade: z.union([ColorGradeSchema, z.literal(false)]).optional(),
|
|
530
562
|
/**
|
|
531
563
|
* Captions OFF for the whole video. Doc-global like `theme`, deliberately
|
|
532
564
|
* NOT a per-scene key: visibility is one decision about the output —
|
package/src/recut.ts
CHANGED
|
@@ -269,6 +269,11 @@ export function resolveCutSourceRanges(
|
|
|
269
269
|
* user cut exactly like an automatic one, with no separate report path
|
|
270
270
|
* needed for "what got removed."
|
|
271
271
|
*/
|
|
272
|
+
/** See the sliver comment inside `subtractRangesFromCutlist` — a keep this
|
|
273
|
+
* short only ever comes from rounded cut edges meeting full-precision span
|
|
274
|
+
* floats, and it crashes the Remotion player if it survives. */
|
|
275
|
+
const SLIVER_EPS = 0.002;
|
|
276
|
+
|
|
272
277
|
export function subtractRangesFromCutlist(
|
|
273
278
|
cutlist: readonly Segment[],
|
|
274
279
|
ranges: readonly { start: number; end: number }[],
|
|
@@ -292,11 +297,35 @@ export function subtractRangesFromCutlist(
|
|
|
292
297
|
const overlapStart = Math.max(cursor, r.start);
|
|
293
298
|
const overlapEnd = Math.min(seg.srcOut, r.end);
|
|
294
299
|
if (overlapStart >= overlapEnd) continue;
|
|
295
|
-
|
|
296
|
-
|
|
300
|
+
// A keep sliver shorter than SLIVER_EPS joins the removal instead of
|
|
301
|
+
// surviving as its own segment (2026-08-31, the ADK crash): the cut
|
|
302
|
+
// writers round `src` to 3 decimals while the cutlist keeps full float
|
|
303
|
+
// precision, so the honest set difference can leave a keep of ~100µs.
|
|
304
|
+
// EdlVideo rounds both ends of such a span to the SAME frame and
|
|
305
|
+
// Remotion throws ("trimAfter must be greater than trimBefore"),
|
|
306
|
+
// blanking the whole player — in the preview AND the next render.
|
|
307
|
+
// 2ms covers the worst rounding drift (0.5ms per edge) and is far
|
|
308
|
+
// under one frame at any real fps, so nothing watchable is lost.
|
|
309
|
+
if (overlapStart - cursor >= SLIVER_EPS) {
|
|
310
|
+
out.push({ srcIn: cursor, srcOut: overlapStart, kind: "keep" });
|
|
311
|
+
out.push({ srcIn: overlapStart, srcOut: overlapEnd, kind: "remove", reason: "user", confidence: 1 });
|
|
312
|
+
} else {
|
|
313
|
+
out.push({ srcIn: cursor, srcOut: overlapEnd, kind: "remove", reason: "user", confidence: 1 });
|
|
314
|
+
}
|
|
297
315
|
cursor = overlapEnd;
|
|
298
316
|
}
|
|
299
|
-
if (
|
|
317
|
+
if (seg.srcOut - cursor >= SLIVER_EPS) {
|
|
318
|
+
out.push({ srcIn: cursor, srcOut: seg.srcOut, kind: "keep" });
|
|
319
|
+
} else if (cursor < seg.srcOut) {
|
|
320
|
+
// Tail sliver: extend the removal that ends at `cursor` (there always
|
|
321
|
+
// is one — cursor only advances past a pushed removal).
|
|
322
|
+
const last = out[out.length - 1];
|
|
323
|
+
if (last && last.kind === "remove" && last.srcOut === cursor) {
|
|
324
|
+
out[out.length - 1] = { ...last, srcOut: seg.srcOut };
|
|
325
|
+
} else {
|
|
326
|
+
out.push({ srcIn: cursor, srcOut: seg.srcOut, kind: "remove", reason: "user", confidence: 1 });
|
|
327
|
+
}
|
|
328
|
+
}
|
|
300
329
|
}
|
|
301
330
|
return out;
|
|
302
331
|
}
|
package/src/scene-schema.ts
CHANGED
|
@@ -146,6 +146,9 @@ export const SceneCueSchema = z
|
|
|
146
146
|
dx: z.number().optional(),
|
|
147
147
|
/** `false` switches the automatic idle-zoom layer off for this scene. */
|
|
148
148
|
autoZoom: z.boolean().optional(),
|
|
149
|
+
/** Audio gain for this window, 1 = as recorded — mirror of
|
|
150
|
+
* `SceneOverrideSchema.video.volume`, which owns the argument. */
|
|
151
|
+
volume: z.number().min(0).max(4).optional(),
|
|
149
152
|
})
|
|
150
153
|
.optional(),
|
|
151
154
|
/**
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
import { basename } from "node:path";
|
|
2
|
+
import { z } from "zod/v4";
|
|
3
|
+
import { TranscriptSchema, type Transcript, type Word } from "../schema";
|
|
4
|
+
import { NOISE_TOKEN, normalizeWords, type TranscribeProvider, type TranscribeRequest } from "./provider";
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Transcription against any OpenAI-compatible `/v1/audio/transcriptions`
|
|
8
|
+
* server — Groq's free tier (8h audio/day, `whisper-large-v3-turbo`), a
|
|
9
|
+
* self-hosted speaches, or anything else speaking that shape.
|
|
10
|
+
*
|
|
11
|
+
* Why (2026-09-01 field report): on a weak CPU — an i3 2nd gen — whisper is
|
|
12
|
+
* the dominant cost of a produce run, minutes of decode per minute of video.
|
|
13
|
+
* A remote call makes it seconds. Local whisper-cli stays the DEFAULT; this
|
|
14
|
+
* is opt-in via `whisperUrl` / `OSSCLIP_WHISPER_URL`, and the API key is
|
|
15
|
+
* optional because self-hosted servers run keyless (unlike publish, where the
|
|
16
|
+
* key is required).
|
|
17
|
+
*
|
|
18
|
+
* Error posture is postiz.ts's, for the same reason: a transcription is the
|
|
19
|
+
* user's explicit action, so every non-2xx throws with the status and a body
|
|
20
|
+
* snippet, and there are NO retries — a retry against a metered free tier
|
|
21
|
+
* silently doubles the quota burn for a failure the user is about to see
|
|
22
|
+
* anyway.
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* `whisperUrl` → the endpoint: trailing slashes dropped,
|
|
27
|
+
* `/audio/transcriptions` appended unless the user already wrote it
|
|
28
|
+
* (`postizApiBase`'s mould). The user configures the OpenAI-compatible BASE
|
|
29
|
+
* ("https://api.groq.com/openai/v1"), which is the URL every provider's
|
|
30
|
+
* quickstart prints — but someone who pastes the full endpoint must not end
|
|
31
|
+
* up posting to `/v1/audio/transcriptions/audio/transcriptions`.
|
|
32
|
+
*/
|
|
33
|
+
export function openaiTranscriptionsUrl(baseUrl: string): string {
|
|
34
|
+
const trimmed = baseUrl.replace(/\/+$/, "");
|
|
35
|
+
return trimmed.endsWith("/audio/transcriptions") ? trimmed : `${trimmed}/audio/transcriptions`;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export class RemoteTranscribeHttpError extends Error {
|
|
39
|
+
constructor(
|
|
40
|
+
readonly url: string,
|
|
41
|
+
readonly status: number,
|
|
42
|
+
bodySnippet: string,
|
|
43
|
+
) {
|
|
44
|
+
// Per-status hints, PostizHttpError's mould: the raw status tells a user
|
|
45
|
+
// nothing about which of the two env vars (or which URL spelling) is
|
|
46
|
+
// wrong, and this path is reached by people who just pasted a quickstart.
|
|
47
|
+
const hint =
|
|
48
|
+
status === 401 || status === 403
|
|
49
|
+
? " — the server rejected OSSCLIP_WHISPER_API_KEY (or none was sent — set it in the environment or ~/.ossclip/.env)"
|
|
50
|
+
: status === 404
|
|
51
|
+
? " — no /audio/transcriptions here — whisperUrl should be the OpenAI-compatible base ending in /v1 (e.g. https://api.groq.com/openai/v1)"
|
|
52
|
+
: status === 413
|
|
53
|
+
? " — audio too large for this server — free Groq caps uploads at 25MB; use --whisper-backend local or the dev tier"
|
|
54
|
+
: status === 429
|
|
55
|
+
? " — rate limited (Groq free tier: 8h audio/day)"
|
|
56
|
+
: "";
|
|
57
|
+
super(`remote transcription POST ${url} failed: ${status}${hint}${bodySnippet ? `\n${bodySnippet}` : ""}`);
|
|
58
|
+
this.name = "RemoteTranscribeHttpError";
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* `verbose_json` as we consume it. LOOSE on purpose: servers add fields
|
|
64
|
+
* freely (Groq sends `task`, `duration`, `segments`, `x_groq`), and a strict
|
|
65
|
+
* object would turn a perfectly good transcription into a parse error the
|
|
66
|
+
* next time one of them ships a field.
|
|
67
|
+
*/
|
|
68
|
+
const RemoteWordSchema = z.looseObject({
|
|
69
|
+
word: z.string(),
|
|
70
|
+
start: z.number(),
|
|
71
|
+
end: z.number(),
|
|
72
|
+
});
|
|
73
|
+
const VerboseJsonSchema = z.looseObject({
|
|
74
|
+
language: z.string().optional(),
|
|
75
|
+
words: z.array(RemoteWordSchema).optional(),
|
|
76
|
+
});
|
|
77
|
+
|
|
78
|
+
export interface OpenAiCompatibleOptions {
|
|
79
|
+
/** OpenAI-compatible base, e.g. "https://api.groq.com/openai/v1". */
|
|
80
|
+
baseUrl: string;
|
|
81
|
+
model: string;
|
|
82
|
+
/** Optional: self-hosted servers (speaches, whisper.cpp server) run keyless. */
|
|
83
|
+
apiKey?: string;
|
|
84
|
+
/** The postiz test seam — the whole HTTP surface is testable without a network. */
|
|
85
|
+
fetchImpl?: typeof fetch;
|
|
86
|
+
/** Per-request cap; an hour of audio takes a while to upload and decode. */
|
|
87
|
+
timeoutMs?: number;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
const DEFAULT_TIMEOUT_MS = 10 * 60 * 1000;
|
|
91
|
+
const BODY_SNIPPET_CHARS = 300;
|
|
92
|
+
|
|
93
|
+
export function createOpenAiCompatibleProvider(opts: OpenAiCompatibleOptions): TranscribeProvider {
|
|
94
|
+
const url = openaiTranscriptionsUrl(opts.baseUrl);
|
|
95
|
+
const fetchImpl = opts.fetchImpl ?? fetch;
|
|
96
|
+
const timeoutMs = opts.timeoutMs ?? DEFAULT_TIMEOUT_MS;
|
|
97
|
+
|
|
98
|
+
return {
|
|
99
|
+
name: "openai-compatible",
|
|
100
|
+
async transcribe(audioPath: string, req: TranscribeRequest): Promise<Transcript> {
|
|
101
|
+
// Belt and braces — the CLI refuses this combination earlier, with the
|
|
102
|
+
// fix named. `/audio/translations` is a different endpoint AND a
|
|
103
|
+
// different default model upstream, so silently swapping both behind
|
|
104
|
+
// `--whisper-translate` would be a surprise, not a convenience.
|
|
105
|
+
if (req.translate === true) {
|
|
106
|
+
throw new Error(
|
|
107
|
+
"--whisper-translate needs the local backend (the OpenAI-compatible API translates on a different endpoint and model) — use --whisper-backend local, or drop the flag.",
|
|
108
|
+
);
|
|
109
|
+
}
|
|
110
|
+
const { openAsBlob } = await import("node:fs");
|
|
111
|
+
// Streams the file into multipart form-data instead of holding it in
|
|
112
|
+
// memory (postiz.ts's rationale): the upload sidecar is small, but a
|
|
113
|
+
// span wav or an uncompressed hour is not.
|
|
114
|
+
const blob = await openAsBlob(audioPath, {
|
|
115
|
+
type: audioPath.endsWith(".ogg") ? "audio/ogg" : "audio/wav",
|
|
116
|
+
});
|
|
117
|
+
const form = new FormData();
|
|
118
|
+
form.append("file", blob, basename(audioPath));
|
|
119
|
+
form.append("model", opts.model);
|
|
120
|
+
form.append("response_format", "verbose_json");
|
|
121
|
+
// The literal bracketed field name is the wire spelling OpenAI and Groq
|
|
122
|
+
// accept — it is an array parameter in a multipart body, not a typo.
|
|
123
|
+
form.append("timestamp_granularities[]", "word");
|
|
124
|
+
// "auto" is whisper.cpp's vocabulary, not an ISO code: sending it makes
|
|
125
|
+
// the server reject the request, while OMITTING the field is exactly
|
|
126
|
+
// what asks for auto-detection.
|
|
127
|
+
if (req.language !== undefined && req.language !== "auto") form.append("language", req.language);
|
|
128
|
+
if (req.prompt !== undefined) form.append("prompt", req.prompt);
|
|
129
|
+
|
|
130
|
+
const ac = new AbortController();
|
|
131
|
+
const timer = setTimeout(() => ac.abort(), timeoutMs);
|
|
132
|
+
let res: Response;
|
|
133
|
+
try {
|
|
134
|
+
res = await fetchImpl(url, {
|
|
135
|
+
method: "POST",
|
|
136
|
+
headers: opts.apiKey ? { Authorization: `Bearer ${opts.apiKey}` } : {},
|
|
137
|
+
body: form,
|
|
138
|
+
signal: ac.signal,
|
|
139
|
+
});
|
|
140
|
+
} catch (err) {
|
|
141
|
+
throw new Error(
|
|
142
|
+
`remote transcription unreachable at ${url}: ${err instanceof Error ? err.message : String(err)}`,
|
|
143
|
+
);
|
|
144
|
+
} finally {
|
|
145
|
+
clearTimeout(timer);
|
|
146
|
+
}
|
|
147
|
+
const text = await res.text();
|
|
148
|
+
if (!res.ok) throw new RemoteTranscribeHttpError(url, res.status, text.slice(0, BODY_SNIPPET_CHARS));
|
|
149
|
+
let json: unknown;
|
|
150
|
+
try {
|
|
151
|
+
json = JSON.parse(text);
|
|
152
|
+
} catch {
|
|
153
|
+
throw new Error(
|
|
154
|
+
`remote transcription answered non-JSON from ${url}: ${text.slice(0, BODY_SNIPPET_CHARS)}`,
|
|
155
|
+
);
|
|
156
|
+
}
|
|
157
|
+
const parsed = VerboseJsonSchema.parse(json);
|
|
158
|
+
if (!parsed.words || parsed.words.length === 0) {
|
|
159
|
+
// Two causes, both worth naming: a server that transcribed fine but
|
|
160
|
+
// has no word-timestamp support (a plain whisper.cpp server), and
|
|
161
|
+
// near-silent audio, which Groq's turbo model answers wordlessly.
|
|
162
|
+
// Everything downstream (cuts, captions, zoom) is word-stamp driven,
|
|
163
|
+
// so a text-only answer is unusable, not a degraded success.
|
|
164
|
+
throw new Error(
|
|
165
|
+
`the server answered without word timestamps — it must support response_format=verbose_json with timestamp_granularities[]=word (Groq and speaches do; a plain whisper.cpp server may not), or the audio contained no speech (${url})`,
|
|
166
|
+
);
|
|
167
|
+
}
|
|
168
|
+
// No token merging and no §130 byte repair here: those heal whisper.cpp
|
|
169
|
+
// `-ml 1` artifacts (one BPE token per segment, split mid-character).
|
|
170
|
+
// The HTTP API returns whole words with punctuation attached — already
|
|
171
|
+
// the shape parseWhisperJson works to produce.
|
|
172
|
+
const words: Word[] = [];
|
|
173
|
+
for (const w of parsed.words) {
|
|
174
|
+
const wordText = w.word.trim();
|
|
175
|
+
if (!wordText || NOISE_TOKEN.test(wordText)) continue;
|
|
176
|
+
words.push({
|
|
177
|
+
text: wordText,
|
|
178
|
+
// Clamped: a server answering -0.01 for the first word would trip
|
|
179
|
+
// WordSchema's nonnegative and fail the whole run over a rounding
|
|
180
|
+
// artifact at the very start of the audio.
|
|
181
|
+
start: Math.max(0, w.start),
|
|
182
|
+
end: w.end,
|
|
183
|
+
});
|
|
184
|
+
}
|
|
185
|
+
return TranscriptSchema.parse({
|
|
186
|
+
// The requested code wins — it is what the cache key and the caption
|
|
187
|
+
// pipeline were told the audio is. Otherwise the server's own answer,
|
|
188
|
+
// lowercased. NOTE: some servers answer with a full NAME ("english")
|
|
189
|
+
// rather than a code; no code-sensitive consumer exists today
|
|
190
|
+
// (captions' RTL check is a Unicode heuristic), so no name→code table.
|
|
191
|
+
language:
|
|
192
|
+
req.language !== undefined && req.language !== "auto"
|
|
193
|
+
? req.language
|
|
194
|
+
: parsed.language?.toLowerCase(),
|
|
195
|
+
words: normalizeWords(words),
|
|
196
|
+
});
|
|
197
|
+
},
|
|
198
|
+
};
|
|
199
|
+
}
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
import type { Transcript, Word } from "../schema";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* The seam between ossclip and any transcription backend (2026-09-01, the
|
|
5
|
+
* weak-CPU field report: whisper is the dominant cost on an i3 2nd gen, and a
|
|
6
|
+
* free remote tier makes it disappear). Two implementations today —
|
|
7
|
+
* whisper.cpp on the box (`whisper-cli.ts`) and any OpenAI-compatible
|
|
8
|
+
* `/v1/audio/transcriptions` server (`openai-compatible.ts`).
|
|
9
|
+
*
|
|
10
|
+
* The local path deliberately keeps calling `runWhisper` directly rather than
|
|
11
|
+
* going through this interface: produce.ts and the edit server inject
|
|
12
|
+
* `runWhisper` itself as a test seam, and forcing that through a provider
|
|
13
|
+
* object would churn every stub for no behavior change.
|
|
14
|
+
*/
|
|
15
|
+
export interface TranscribeRequest {
|
|
16
|
+
/** Resolved language code; "auto" is handled per provider (whisper.cpp
|
|
17
|
+
* takes it literally, the HTTP API wants the field OMITTED). */
|
|
18
|
+
language?: string;
|
|
19
|
+
/** `whisperPromptFor()` output — the user dictionary as decoder bias. */
|
|
20
|
+
prompt?: string;
|
|
21
|
+
/** whisper-cli's TRANSLATE task; the OpenAI-compatible provider rejects it
|
|
22
|
+
* (a different endpoint AND a different default model upstream). */
|
|
23
|
+
translate?: boolean;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export interface TranscribeProvider {
|
|
27
|
+
name: string;
|
|
28
|
+
transcribe(audioPath: string, req: TranscribeRequest): Promise<Transcript>;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** [BLANK_AUDIO], (buzzing), [MUSIC] … — noise markers, not speech. Shared:
|
|
32
|
+
* remote servers emit the same bracketed markers whisper.cpp does. */
|
|
33
|
+
export const NOISE_TOKEN = /^[[(].*[\])]$/;
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Run length at which a stack of zero-length words at ONE instant stops being
|
|
37
|
+
* a rounding artifact and becomes a repetition-loop hallucination. Real speech
|
|
38
|
+
* never emits 8 tokens at a single instant; the field case emitted 118.
|
|
39
|
+
*/
|
|
40
|
+
export const REPETITION_BURST_MIN = 8;
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Drop whisper repetition-loop bursts (field case 2026-08-18): an Urdu take
|
|
44
|
+
* re-decoded a whole phrase as 118 CONSECUTIVE tokens all stamped
|
|
45
|
+
* `from === to === 31040` — zero length, at one instant. The stamp repair
|
|
46
|
+
* below then fans such a burst out into 118 fabricated 50ms words marching
|
|
47
|
+
* forward from 31.04s, so the phrase ships TWICE in the captions (31.04s and
|
|
48
|
+
* 33.54s) and a fifth of the transcript carries the tell-tale exactly-0.05s
|
|
49
|
+
* duration. `-mc 0` in whisperArgs is the decoder-side mitigation for the same
|
|
50
|
+
* failure; it did not prevent this occurrence, and it can never repair an
|
|
51
|
+
* already-cached transcript.json — hence a parse-side guard too.
|
|
52
|
+
*
|
|
53
|
+
* A burst is a MAXIMAL run of consecutive zero-length/inverted words sharing
|
|
54
|
+
* one `start`. Equality is exact, not epsilon: these stamps are integer
|
|
55
|
+
* milliseconds divided by 1000, so members of one burst are the same double
|
|
56
|
+
* bit-for-bit, and a tolerance would only start swallowing real neighbors.
|
|
57
|
+
* Runs shorter than REPETITION_BURST_MIN fall through untouched — a lone
|
|
58
|
+
* zero-length stamp is a rounding artifact, not a hallucination. The drop is
|
|
59
|
+
* silent by design: this function is pure and total, and there is no logging
|
|
60
|
+
* channel in the parse path to warn on.
|
|
61
|
+
*/
|
|
62
|
+
export function dropRepetitionBursts(words: readonly Word[]): Word[] {
|
|
63
|
+
const out: Word[] = [];
|
|
64
|
+
let i = 0;
|
|
65
|
+
while (i < words.length) {
|
|
66
|
+
const w = words[i]!;
|
|
67
|
+
if (w.end > w.start) {
|
|
68
|
+
out.push(w);
|
|
69
|
+
i++;
|
|
70
|
+
continue;
|
|
71
|
+
}
|
|
72
|
+
let j = i + 1;
|
|
73
|
+
while (j < words.length && words[j]!.end <= words[j]!.start && words[j]!.start === w.start) j++;
|
|
74
|
+
if (j - i < REPETITION_BURST_MIN) for (let k = i; k < j; k++) out.push(words[k]!);
|
|
75
|
+
i = j;
|
|
76
|
+
}
|
|
77
|
+
return out;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* The word-stamp hygiene EVERY backend's output goes through, extracted from
|
|
82
|
+
* the tail of `parseWhisperJson` so the remote provider cannot drift from it
|
|
83
|
+
* (2026-09-01). Burst drop runs BEFORE the repair, never after: the repair
|
|
84
|
+
* rewrites every burst member into a distinct monotone stamp, so once it has
|
|
85
|
+
* run the shared timestamp — the only evidence a burst existed — is gone.
|
|
86
|
+
*
|
|
87
|
+
* Copies before mutating, so a caller's array survives the call unchanged;
|
|
88
|
+
* `parseWhisperJson` builds its words fresh, so this is byte-identical to the
|
|
89
|
+
* in-place loop it replaces.
|
|
90
|
+
*/
|
|
91
|
+
export function normalizeWords(words: readonly Word[]): Word[] {
|
|
92
|
+
const kept = dropRepetitionBursts(words).map((w) => ({ ...w }));
|
|
93
|
+
// Whisper occasionally emits zero-length or inverted stamps; repair minimally.
|
|
94
|
+
for (let i = 0; i < kept.length; i++) {
|
|
95
|
+
const w = kept[i]!;
|
|
96
|
+
if (w.end <= w.start) w.end = w.start + 0.05;
|
|
97
|
+
const next = kept[i + 1];
|
|
98
|
+
if (next && next.start < w.end) next.start = w.end;
|
|
99
|
+
}
|
|
100
|
+
return kept;
|
|
101
|
+
}
|