ossclip 0.1.36 → 0.1.38
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -1
- package/editor-dist/assets/{index-pWbFr8vc.js → index-DpySfDtS.js} +37 -37
- package/editor-dist/index.html +1 -1
- package/package.json +4 -4
- package/src/analyze.ts +4 -0
- package/src/doctor.ts +38 -6
- package/src/edit.ts +137 -32
- package/src/interactive/produce-wizard.ts +7 -2
- package/src/produce.ts +145 -44
- package/src/program.ts +47 -0
- package/src/setup/plan.ts +18 -0
- package/src/whisper-backend.ts +103 -0
package/editor-dist/index.html
CHANGED
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
<meta charset="UTF-8" />
|
|
5
5
|
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
|
6
6
|
<title>ossclip editor</title>
|
|
7
|
-
<script type="module" crossorigin src="/assets/index-
|
|
7
|
+
<script type="module" crossorigin src="/assets/index-DpySfDtS.js"></script>
|
|
8
8
|
<link rel="stylesheet" crossorigin href="/assets/index-Bx2VQLP8.css">
|
|
9
9
|
</head>
|
|
10
10
|
<body>
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ossclip",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.38",
|
|
4
4
|
"description": "Local-first CLI video producer: cuts silence and fillers, word-timed captions, face-aware framing, and LLM-planned code-rendered graphics — transcription and rendering never leave your machine",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -36,9 +36,9 @@
|
|
|
36
36
|
"commander": "^12.1.0",
|
|
37
37
|
"tsx": "^4.19.0",
|
|
38
38
|
"zod": "^3.25.76",
|
|
39
|
-
"@ossclip/core": "0.1.
|
|
40
|
-
"@ossclip/
|
|
41
|
-
"@ossclip/
|
|
39
|
+
"@ossclip/core": "0.1.38",
|
|
40
|
+
"@ossclip/scenes": "0.1.38",
|
|
41
|
+
"@ossclip/renderer": "0.1.38"
|
|
42
42
|
},
|
|
43
43
|
"homepage": "https://github.com/AhsanAyaz/ossclip#readme",
|
|
44
44
|
"bugs": {
|
package/src/analyze.ts
CHANGED
|
@@ -117,6 +117,9 @@ export interface AnalyzeOptions {
|
|
|
117
117
|
noiseDb?: number;
|
|
118
118
|
whisperModel?: string;
|
|
119
119
|
whisperLanguage?: string;
|
|
120
|
+
/** `--whisper-backend`, already zod-parsed by program.ts; undefined lets a
|
|
121
|
+
* configured `whisperUrl` decide (resolveWhisperBackend in produce). */
|
|
122
|
+
whisperBackend?: "local" | "remote";
|
|
120
123
|
blooperMarker?: string;
|
|
121
124
|
collapseRetakes?: boolean;
|
|
122
125
|
sort?: "name" | "mtime";
|
|
@@ -157,6 +160,7 @@ export async function runAnalyze(
|
|
|
157
160
|
noiseDb: opts.noiseDb,
|
|
158
161
|
whisperModel: opts.whisperModel,
|
|
159
162
|
whisperLanguage: opts.whisperLanguage,
|
|
163
|
+
whisperBackend: opts.whisperBackend,
|
|
160
164
|
blooperMarker: opts.blooperMarker,
|
|
161
165
|
collapseRetakes: opts.collapseRetakes,
|
|
162
166
|
sort: opts.sort,
|
package/src/doctor.ts
CHANGED
|
@@ -2,6 +2,7 @@ import { spawn } from "node:child_process";
|
|
|
2
2
|
import { existsSync } from "node:fs";
|
|
3
3
|
import type { OssclipConfig } from "@ossclip/core";
|
|
4
4
|
import { modelUrl, validModelSources, whisperModelPath } from "./setup/manifest";
|
|
5
|
+
import { WHISPER_API_KEY_ENV, resolveWhisperBackend } from "./whisper-backend";
|
|
5
6
|
|
|
6
7
|
/**
|
|
7
8
|
* `ossclip doctor` (R18 §90a): check every prerequisite and print the exact
|
|
@@ -106,12 +107,23 @@ export async function runDoctor(cfg: OssclipConfig, p: DoctorProbes): Promise<Do
|
|
|
106
107
|
}),
|
|
107
108
|
});
|
|
108
109
|
|
|
110
|
+
// Remote transcription (2026-09-01 weak-CPU field report) makes the next
|
|
111
|
+
// two checks OPTIONAL rather than blocking: a machine that transcribes on
|
|
112
|
+
// Groq has no reason to own whisper.cpp or a 1.5 GB model, and doctor
|
|
113
|
+
// reporting two red lines on a working install is how a user concludes the
|
|
114
|
+
// tool is broken. The LLM-provider posture, one level up: pass with a
|
|
115
|
+
// detail that says why nothing is needed.
|
|
116
|
+
const remote = resolveWhisperBackend(undefined, cfg, p.env);
|
|
117
|
+
const remoteBackend = remote.ok && remote.backend.kind === "remote" ? remote.backend : null;
|
|
118
|
+
const notNeeded = (found: string): string =>
|
|
119
|
+
`${found} not found — not needed: remote transcription configured`;
|
|
120
|
+
|
|
109
121
|
const whisperOk = await p.binRuns(cfg.whisperPath, "--help");
|
|
110
122
|
checks.push({
|
|
111
123
|
name: "whisper-cli",
|
|
112
|
-
ok: whisperOk,
|
|
113
|
-
detail: cfg.whisperPath,
|
|
114
|
-
...(whisperOk
|
|
124
|
+
ok: whisperOk || remoteBackend !== null,
|
|
125
|
+
detail: whisperOk ? cfg.whisperPath : remoteBackend !== null ? notNeeded(cfg.whisperPath) : cfg.whisperPath,
|
|
126
|
+
...(whisperOk || remoteBackend !== null
|
|
115
127
|
? {}
|
|
116
128
|
: {
|
|
117
129
|
fix: viaSetup(
|
|
@@ -134,9 +146,9 @@ export async function runDoctor(cfg: OssclipConfig, p: DoctorProbes): Promise<Do
|
|
|
134
146
|
const modelOk = p.exists(modelPath);
|
|
135
147
|
checks.push({
|
|
136
148
|
name: `whisper model (${cfg.model})`,
|
|
137
|
-
ok: modelOk,
|
|
138
|
-
detail: modelPath,
|
|
139
|
-
...(modelOk
|
|
149
|
+
ok: modelOk || remoteBackend !== null,
|
|
150
|
+
detail: modelOk ? modelPath : remoteBackend !== null ? notNeeded(modelPath) : modelPath,
|
|
151
|
+
...(modelOk || remoteBackend !== null
|
|
140
152
|
? {}
|
|
141
153
|
: {
|
|
142
154
|
fix: viaSetup(
|
|
@@ -146,6 +158,26 @@ export async function runDoctor(cfg: OssclipConfig, p: DoctorProbes): Promise<Do
|
|
|
146
158
|
}),
|
|
147
159
|
});
|
|
148
160
|
|
|
161
|
+
// NO network call, unlike every other backend doctor could probe: a
|
|
162
|
+
// transcription request costs the user's metered free tier, and `doctor` is
|
|
163
|
+
// run repeatedly while fixing something else. This line reports the
|
|
164
|
+
// CONFIGURATION — the three things a 401 or a 404 would be about — and the
|
|
165
|
+
// provider's own status hints name the rest when a real run happens.
|
|
166
|
+
// Omitted entirely when remote is not configured: the local install is the
|
|
167
|
+
// default, and an extra "not configured" line for an opt-in feature is
|
|
168
|
+
// noise on every other machine.
|
|
169
|
+
if (remoteBackend !== null) {
|
|
170
|
+
checks.push({
|
|
171
|
+
name: "remote transcription",
|
|
172
|
+
ok: true,
|
|
173
|
+
detail:
|
|
174
|
+
`${remoteBackend.baseUrl} · model ${remoteBackend.model} · ` +
|
|
175
|
+
(remoteBackend.apiKey !== undefined
|
|
176
|
+
? `${WHISPER_API_KEY_ENV} set`
|
|
177
|
+
: `no API key (fine for self-hosted; Groq needs ${WHISPER_API_KEY_ENV})`),
|
|
178
|
+
});
|
|
179
|
+
}
|
|
180
|
+
|
|
149
181
|
// Provider, in the same order auto-detection uses (agy → claude CLI →
|
|
150
182
|
// gemini key → anthropic key): subscription CLIs beat ambient env keys
|
|
151
183
|
// since 2026-08 — a logged-in CLI is an explicit, already-paid choice
|
package/src/edit.ts
CHANGED
|
@@ -70,6 +70,11 @@ import {
|
|
|
70
70
|
portraitMimeType,
|
|
71
71
|
readCoverProvenance,
|
|
72
72
|
runWhisper,
|
|
73
|
+
// The remote transcription backend (2026-09-01) — the span re-decode goes
|
|
74
|
+
// through it whenever `whisperUrl` is configured.
|
|
75
|
+
createOpenAiCompatibleProvider,
|
|
76
|
+
type TranscribeRequest,
|
|
77
|
+
type Transcript,
|
|
73
78
|
SegmentSchema,
|
|
74
79
|
spliceTranscript,
|
|
75
80
|
TranscriptSchema,
|
|
@@ -116,6 +121,7 @@ import { expandHome } from "./paths";
|
|
|
116
121
|
// produce all resolve through — a second copy here would send a user to a
|
|
117
122
|
// model file the rest of the tool never looks for.
|
|
118
123
|
import { modelImpliedLanguage, whisperModelPath } from "./setup/manifest";
|
|
124
|
+
import { resolveWhisperBackend } from "./whisper-backend";
|
|
119
125
|
import {
|
|
120
126
|
PORTRAIT_OVERRIDE_BASENAME,
|
|
121
127
|
portraitExtensionForMime,
|
|
@@ -198,6 +204,37 @@ export interface RetranscribeConfig {
|
|
|
198
204
|
modelDir?: string;
|
|
199
205
|
language?: unknown;
|
|
200
206
|
dictionary?: unknown;
|
|
207
|
+
/** The remote-backend pair (2026-09-01): a configured URL means this span
|
|
208
|
+
* is decoded by an OpenAI-compatible server instead of whisper.cpp, so the
|
|
209
|
+
* server must see the same two keys `resolveWhisperBackend` reads. */
|
|
210
|
+
whisperUrl?: string;
|
|
211
|
+
whisperRemoteModel?: string;
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/**
|
|
215
|
+
* The half of a range re-decode that is the same on BOTH backends: the
|
|
216
|
+
* decoder bias. Extracted (2026-09-01) so the remote path shares these rules
|
|
217
|
+
* verbatim instead of growing a second copy that drifts — the validation is
|
|
218
|
+
* the load-bearing part, and it is unchanged from the local-only version.
|
|
219
|
+
*/
|
|
220
|
+
function retranscribeBias(cfg: RetranscribeConfig): { language?: string; prompt?: string } {
|
|
221
|
+
const dict = Array.isArray(cfg.dictionary)
|
|
222
|
+
&& cfg.dictionary.length > 0
|
|
223
|
+
&& cfg.dictionary.every((t) => typeof t === "string" && t.trim().length > 0)
|
|
224
|
+
? (cfg.dictionary as string[]).map((t) => t.trim())
|
|
225
|
+
: [];
|
|
226
|
+
// `cfg.model` can be absent on the remote backend (a machine that never
|
|
227
|
+
// installed a local model), and no model name implies no language.
|
|
228
|
+
const language = typeof cfg.language === "string" && cfg.language.trim().length > 0
|
|
229
|
+
? cfg.language.trim()
|
|
230
|
+
: cfg.model !== undefined
|
|
231
|
+
? modelImpliedLanguage(cfg.model)
|
|
232
|
+
: undefined;
|
|
233
|
+
const prompt = whisperPromptFor(dict);
|
|
234
|
+
return {
|
|
235
|
+
...(language !== undefined ? { language } : {}),
|
|
236
|
+
...(prompt !== undefined ? { prompt } : {}),
|
|
237
|
+
};
|
|
201
238
|
}
|
|
202
239
|
|
|
203
240
|
/**
|
|
@@ -238,21 +275,38 @@ export function retranscribeSettings(
|
|
|
238
275
|
"then `ossclip setup` to install it.",
|
|
239
276
|
};
|
|
240
277
|
}
|
|
241
|
-
const dict = Array.isArray(cfg.dictionary)
|
|
242
|
-
&& cfg.dictionary.length > 0
|
|
243
|
-
&& cfg.dictionary.every((t) => typeof t === "string" && t.trim().length > 0)
|
|
244
|
-
? (cfg.dictionary as string[]).map((t) => t.trim())
|
|
245
|
-
: [];
|
|
246
|
-
const language = typeof cfg.language === "string" && cfg.language.trim().length > 0
|
|
247
|
-
? cfg.language.trim()
|
|
248
|
-
: modelImpliedLanguage(cfg.model);
|
|
249
|
-
const prompt = whisperPromptFor(dict);
|
|
250
278
|
return {
|
|
251
279
|
tools: { ffmpegPath: cfg.ffmpegPath, ffprobePath: cfg.ffprobePath },
|
|
252
280
|
whisperPath: cfg.whisperPath,
|
|
253
281
|
modelPath: whisperModelPath(cfg.model, cfg.modelDir),
|
|
254
|
-
...(
|
|
255
|
-
|
|
282
|
+
...retranscribeBias(cfg),
|
|
283
|
+
};
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
/**
|
|
287
|
+
* The same thing for the REMOTE backend (2026-09-01): no whisper binary and
|
|
288
|
+
* no model file, because the point of remote is that neither is installed —
|
|
289
|
+
* but ffmpeg still is, since the span is sliced out of `audio.wav` locally
|
|
290
|
+
* before it is uploaded.
|
|
291
|
+
*
|
|
292
|
+
* Pure, like its local twin: the whole "remote configured but ffmpeg isn't"
|
|
293
|
+
* corner is testable without a network.
|
|
294
|
+
*/
|
|
295
|
+
export function retranscribeRemoteSettings(
|
|
296
|
+
cfg: RetranscribeConfig,
|
|
297
|
+
):
|
|
298
|
+
| { tools: { ffmpegPath: string; ffprobePath: string }; language?: string; prompt?: string }
|
|
299
|
+
| { error: string } {
|
|
300
|
+
if (!cfg.ffmpegPath || !cfg.ffprobePath) {
|
|
301
|
+
return {
|
|
302
|
+
error:
|
|
303
|
+
"ffmpeg is not configured — run `ossclip doctor` to see what is missing, " +
|
|
304
|
+
"then `ossclip setup` to install it. (Remote transcription still slices the span locally.)",
|
|
305
|
+
};
|
|
306
|
+
}
|
|
307
|
+
return {
|
|
308
|
+
tools: { ffmpegPath: cfg.ffmpegPath, ffprobePath: cfg.ffprobePath },
|
|
309
|
+
...retranscribeBias(cfg),
|
|
256
310
|
};
|
|
257
311
|
}
|
|
258
312
|
|
|
@@ -484,6 +538,13 @@ export async function startEditServer(
|
|
|
484
538
|
*/
|
|
485
539
|
sliceAudio?: typeof extractAudioSpan;
|
|
486
540
|
runWhisper?: typeof runWhisper;
|
|
541
|
+
/**
|
|
542
|
+
* The remote backend's half of the `runWhisper` seam (2026-09-01): with a
|
|
543
|
+
* `whisperUrl` configured the span goes to an OpenAI-compatible server
|
|
544
|
+
* instead of whisper.cpp, and a test must be able to observe that —
|
|
545
|
+
* including the failure sentence — without a network or an API key.
|
|
546
|
+
*/
|
|
547
|
+
transcribeRemote?: (wavPath: string, req: TranscribeRequest) => Promise<Transcript>;
|
|
487
548
|
/**
|
|
488
549
|
* The sound library the SFX routes serve (`loadCfg`'s rule applied to the
|
|
489
550
|
* pack loader): tests inject a hand-written library over a tmp dir, so the
|
|
@@ -955,17 +1016,67 @@ export async function startEditServer(
|
|
|
955
1016
|
error: "this workdir has no transcript.json to re-stamp — re-run `ossclip produce`.",
|
|
956
1017
|
});
|
|
957
1018
|
}
|
|
958
|
-
const
|
|
959
|
-
|
|
960
|
-
|
|
961
|
-
|
|
962
|
-
|
|
963
|
-
|
|
964
|
-
|
|
965
|
-
|
|
966
|
-
|
|
967
|
-
|
|
1019
|
+
const cfg = (opts.loadCfg ?? loadConfig)();
|
|
1020
|
+
// Which engine re-decodes the span, resolved HERE rather than in
|
|
1021
|
+
// retranscribeSettings (2026-09-01): the flag is a CLI thing and
|
|
1022
|
+
// there is no CLI in this loop, so a configured `whisperUrl` is
|
|
1023
|
+
// the whole switch. `undefined` as the flag can only answer ok —
|
|
1024
|
+
// only an explicit `--whisper-backend remote` with nothing
|
|
1025
|
+
// configured fails — so this is a narrowing, not a live branch.
|
|
1026
|
+
const backendPick = resolveWhisperBackend(undefined, cfg, process.env);
|
|
1027
|
+
if (!backendPick.ok) return send(200, { ok: false, error: backendPick.message });
|
|
1028
|
+
const backend = backendPick.backend;
|
|
1029
|
+
// One plan, two shapes: the local one needs a binary and a model
|
|
1030
|
+
// file on disk, the remote one needs neither. Both need ffmpeg —
|
|
1031
|
+
// the span is always sliced here.
|
|
1032
|
+
let plan: {
|
|
1033
|
+
tools: { ffmpegPath: string; ffprobePath: string };
|
|
1034
|
+
language?: string;
|
|
1035
|
+
prompt?: string;
|
|
1036
|
+
decode: (wavPath: string, req: TranscribeRequest) => Promise<Transcript>;
|
|
1037
|
+
};
|
|
1038
|
+
if (backend.kind === "remote") {
|
|
1039
|
+
const settings = retranscribeRemoteSettings(cfg);
|
|
1040
|
+
if ("error" in settings) return send(200, { ok: false, error: settings.error });
|
|
1041
|
+
const provider = createOpenAiCompatibleProvider({
|
|
1042
|
+
baseUrl: backend.baseUrl,
|
|
1043
|
+
model: backend.model,
|
|
1044
|
+
...(backend.apiKey !== undefined ? { apiKey: backend.apiKey } : {}),
|
|
968
1045
|
});
|
|
1046
|
+
plan = {
|
|
1047
|
+
...settings,
|
|
1048
|
+
// Uploaded AS-IS, no opus sidecar (produce's rule does not
|
|
1049
|
+
// apply): a span is seconds long, so its wav is far under any
|
|
1050
|
+
// upload cap and the encode would cost more than it saves.
|
|
1051
|
+
decode: opts.transcribeRemote ?? ((wavPath, r) => provider.transcribe(wavPath, r)),
|
|
1052
|
+
};
|
|
1053
|
+
} else {
|
|
1054
|
+
const settings = retranscribeSettings(cfg);
|
|
1055
|
+
if ("error" in settings) return send(200, { ok: false, error: settings.error });
|
|
1056
|
+
if (!existsSync(settings.modelPath)) {
|
|
1057
|
+
// The `--transcript`-only install: whisper was never needed to
|
|
1058
|
+
// make this project, so say what to run rather than 500ing.
|
|
1059
|
+
return send(200, {
|
|
1060
|
+
ok: false,
|
|
1061
|
+
error:
|
|
1062
|
+
`whisper model not found at ${settings.modelPath} — run \`ossclip setup\` ` +
|
|
1063
|
+
`to download it.`,
|
|
1064
|
+
});
|
|
1065
|
+
}
|
|
1066
|
+
plan = {
|
|
1067
|
+
...settings,
|
|
1068
|
+
decode: (wavPath, r) =>
|
|
1069
|
+
(opts.runWhisper ?? runWhisper)(
|
|
1070
|
+
{
|
|
1071
|
+
whisperPath: settings.whisperPath,
|
|
1072
|
+
modelPath: settings.modelPath,
|
|
1073
|
+
outBase,
|
|
1074
|
+
...(r.language !== undefined ? { language: r.language } : {}),
|
|
1075
|
+
...(r.prompt !== undefined ? { prompt: r.prompt } : {}),
|
|
1076
|
+
},
|
|
1077
|
+
wavPath,
|
|
1078
|
+
),
|
|
1079
|
+
};
|
|
969
1080
|
}
|
|
970
1081
|
// Parsed, not cast: this file is about to be rewritten, and a
|
|
971
1082
|
// truncated one must fail loudly here rather than become the new
|
|
@@ -985,22 +1096,16 @@ export async function startEditServer(
|
|
|
985
1096
|
});
|
|
986
1097
|
}
|
|
987
1098
|
await (opts.sliceAudio ?? extractAudioSpan)(
|
|
988
|
-
|
|
1099
|
+
plan.tools,
|
|
989
1100
|
audio,
|
|
990
1101
|
tmpWav,
|
|
991
1102
|
srcIn,
|
|
992
1103
|
srcOut - srcIn,
|
|
993
1104
|
);
|
|
994
|
-
const fresh = await (
|
|
995
|
-
{
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
outBase,
|
|
999
|
-
...(settings.language !== undefined ? { language: settings.language } : {}),
|
|
1000
|
-
...(settings.prompt !== undefined ? { prompt: settings.prompt } : {}),
|
|
1001
|
-
},
|
|
1002
|
-
tmpWav,
|
|
1003
|
-
);
|
|
1105
|
+
const fresh = await plan.decode(tmpWav, {
|
|
1106
|
+
...(plan.language !== undefined ? { language: plan.language } : {}),
|
|
1107
|
+
...(plan.prompt !== undefined ? { prompt: plan.prompt } : {}),
|
|
1108
|
+
});
|
|
1004
1109
|
const restamped = alignRestamp(
|
|
1005
1110
|
transcript.words.slice(range.from, range.to),
|
|
1006
1111
|
fresh.words,
|
|
@@ -10,7 +10,7 @@ import { produceArgv, type ProduceAnswers, type ProduceExtras } from "./produce-
|
|
|
10
10
|
import { assertInteractive, confirm, intro, multiselect, select, text, unwrap } from "./prompts";
|
|
11
11
|
|
|
12
12
|
/**
|
|
13
|
-
* The produce wizard. Forty-
|
|
13
|
+
* The produce wizard. Forty-four flags (plus the positional input path)
|
|
14
14
|
* sorted into three tiers: six prompts asked directly — the input path, plus
|
|
15
15
|
* five flags (--out, --cleanup, --aspect, --produce, --intent) — twelve
|
|
16
16
|
* behind one "anything else?" multiselect (--sfx being the twelfth, with
|
|
@@ -28,7 +28,12 @@ import { assertInteractive, confirm, intro, multiselect, select, text, unwrap }
|
|
|
28
28
|
* multiselect entry is the OFF switch and the positive flag exists only for
|
|
29
29
|
* replay pinning), --add-jump-cuts (same mirror: auto already punches, the
|
|
30
30
|
* multiselect entry is the OFF switch, and the force flag exists to beat a
|
|
31
|
-
* future config-off),
|
|
31
|
+
* future config-off),
|
|
32
|
+
* --whisper-backend (2026-09-01, the --color-grade shape: it selects machine
|
|
33
|
+
* INFRASTRUCTURE — which transcription engine this box owns, alongside
|
|
34
|
+
* whisperPath and modelDir — not a per-run editorial choice, and the durable
|
|
35
|
+
* spelling is `whisperUrl` in ~/.ossclip/config.json; an honest prompt would
|
|
36
|
+
* also have to explain a base URL and an API key at a menu), or
|
|
32
37
|
* (final-review fix wave, Finding 1) --sort. A folder's clip order only means anything once the
|
|
33
38
|
* folder has been enumerated, and that enumeration happens inside
|
|
34
39
|
* `produce()` — after the wizard has already returned argv — so there is
|
package/src/produce.ts
CHANGED
|
@@ -66,6 +66,10 @@ import {
|
|
|
66
66
|
splitThenDropHidden,
|
|
67
67
|
emptyOverrideDoc,
|
|
68
68
|
extractAudio,
|
|
69
|
+
encodeUploadAudio,
|
|
70
|
+
REMOTE_UPLOAD_MAX_BYTES,
|
|
71
|
+
createOpenAiCompatibleProvider,
|
|
72
|
+
openaiTranscriptionsUrl,
|
|
69
73
|
fillPlainCues,
|
|
70
74
|
splitCues,
|
|
71
75
|
landscapeLayout,
|
|
@@ -249,6 +253,7 @@ import { RenderTimelineHUD, StageAnimator, printProductionCompleteBanner } from
|
|
|
249
253
|
import { reconcileCaptionEdits } from "./caption-report";
|
|
250
254
|
import { overridesWriteLine, writeOverrideDoc } from "./overrides-write";
|
|
251
255
|
import { recordedProduceArgs } from "./replay-argv";
|
|
256
|
+
import { remoteWhisperHost, resolveWhisperBackend } from "./whisper-backend";
|
|
252
257
|
import { makeCancelSignal, renderCover, renderProduction } from "@ossclip/renderer";
|
|
253
258
|
import type { RenderPhase } from "@ossclip/renderer";
|
|
254
259
|
import {
|
|
@@ -311,6 +316,17 @@ export const TranscriptKeySchema = z.object({
|
|
|
311
316
|
* files) means "no translation", the `dictionary` contract.
|
|
312
317
|
*/
|
|
313
318
|
translate: z.boolean().optional(),
|
|
319
|
+
/**
|
|
320
|
+
* Which BACKEND decoded it (2026-09-01): `remote:<normalized endpoint>`, or
|
|
321
|
+
* absent for local whisper.cpp — the `dictionary`/`translate` contract, so
|
|
322
|
+
* every key file written before remote existed still reads as local. Two
|
|
323
|
+
* engines on the same audio produce different words, so without this a warm
|
|
324
|
+
* workdir serves the local transcript to a remote run (and vice versa) —
|
|
325
|
+
* the exact staleness `language` and `translate` were added for. The
|
|
326
|
+
* remote MODEL name rides in `model` above, so switching Groq models
|
|
327
|
+
* re-keys through the existing field.
|
|
328
|
+
*/
|
|
329
|
+
backend: z.string().optional(),
|
|
314
330
|
});
|
|
315
331
|
export type TranscriptKey = z.infer<typeof TranscriptKeySchema>;
|
|
316
332
|
|
|
@@ -338,6 +354,10 @@ export function transcriptCacheReusable(
|
|
|
338
354
|
// Absent and false are the same "no translation", so pre-flag key
|
|
339
355
|
// files reuse under a non-translate request.
|
|
340
356
|
(effective.translate ?? false) === (requested.translate ?? false) &&
|
|
357
|
+
// Absent means LOCAL on both sides, so every pre-2026-09-01 key file
|
|
358
|
+
// still reuses under a local request — and a remote request against
|
|
359
|
+
// one of them re-transcribes, which is the point.
|
|
360
|
+
(effective.backend ?? "") === (requested.backend ?? "") &&
|
|
341
361
|
// ORDER-SENSITIVE by choice: the dictionary becomes whisper's --prompt
|
|
342
362
|
// text verbatim, so a reordered list genuinely is a different decoder
|
|
343
363
|
// input — treating it as equal would serve a transcript biased by a
|
|
@@ -634,6 +654,13 @@ export interface ProduceOptions {
|
|
|
634
654
|
* together (whisper decodes better knowing the source language).
|
|
635
655
|
*/
|
|
636
656
|
whisperTranslate?: boolean;
|
|
657
|
+
/**
|
|
658
|
+
* `--whisper-backend`, already zod-parsed to the union by program.ts
|
|
659
|
+
* (2026-09-01 weak-CPU field report). Undefined means "not typed", which
|
|
660
|
+
* is what lets a configured `whisperUrl` select remote — the flag is
|
|
661
|
+
* mainly `local`, the per-run opt-out.
|
|
662
|
+
*/
|
|
663
|
+
whisperBackend?: "local" | "remote";
|
|
637
664
|
/**
|
|
638
665
|
* Vocabulary terms for this run (`--dictionary`, F4 2026-08-16), already
|
|
639
666
|
* split/trimmed by the action. Wholesale beats the config's `dictionary`
|
|
@@ -2664,9 +2691,34 @@ export async function produce(inputArg: string, opts: ProduceOptions): Promise<P
|
|
|
2664
2691
|
`--whisper-language overrides)`,
|
|
2665
2692
|
);
|
|
2666
2693
|
}
|
|
2694
|
+
// Local whisper.cpp or an OpenAI-compatible server (2026-09-01 weak-CPU
|
|
2695
|
+
// field report). Resolved BEFORE the key, like the language, so what
|
|
2696
|
+
// actually decodes is what the cache is keyed on.
|
|
2697
|
+
const backendPick = resolveWhisperBackend(opts.whisperBackend, cfg, process.env);
|
|
2698
|
+
if (!backendPick.ok) throw new Error(backendPick.message);
|
|
2699
|
+
const backend = backendPick.backend;
|
|
2700
|
+
// BEFORE the key is built, so a translate request can never cross the
|
|
2701
|
+
// cache with a remote one: the OpenAI-compatible API translates on a
|
|
2702
|
+
// DIFFERENT endpoint with a DIFFERENT default model, and swapping both
|
|
2703
|
+
// behind one flag would be a surprise rather than a convenience.
|
|
2704
|
+
if (backend.kind === "remote" && opts.whisperTranslate === true) {
|
|
2705
|
+
throw new Error(
|
|
2706
|
+
"--whisper-translate needs the local backend (the OpenAI-compatible API translates on a " +
|
|
2707
|
+
"different endpoint and model) — use --whisper-backend local, or drop the flag.",
|
|
2708
|
+
);
|
|
2709
|
+
}
|
|
2667
2710
|
const requestedKey: TranscriptKey = {
|
|
2668
|
-
model
|
|
2711
|
+
// The REMOTE model name when remote — one field, both engines, so an
|
|
2712
|
+
// A/B between two Groq models re-keys the cache exactly like a local one.
|
|
2713
|
+
model: backend.kind === "remote" ? backend.model : requestedModel,
|
|
2669
2714
|
...(whisperLang.language !== undefined ? { language: whisperLang.language } : {}),
|
|
2715
|
+
// Spread-omitted on local so local key files stay byte-identical to
|
|
2716
|
+
// every one written before remote existed (the translate posture). The
|
|
2717
|
+
// URL goes through openaiTranscriptionsUrl so ".../v1" and ".../v1/"
|
|
2718
|
+
// key identically — a trailing slash is not a different server.
|
|
2719
|
+
...(backend.kind === "remote"
|
|
2720
|
+
? { backend: `remote:${openaiTranscriptionsUrl(backend.baseUrl)}` }
|
|
2721
|
+
: {}),
|
|
2670
2722
|
// Omitted when off, so a non-translate run's key stays byte-identical to
|
|
2671
2723
|
// every pre-flag key file (the dictionary posture).
|
|
2672
2724
|
...(opts.whisperTranslate === true ? { translate: true } : {}),
|
|
@@ -2704,51 +2756,100 @@ export async function produce(inputArg: string, opts: ProduceOptions): Promise<P
|
|
|
2704
2756
|
`re-transcribing with ${fmt(requestedKey)}`,
|
|
2705
2757
|
);
|
|
2706
2758
|
}
|
|
2707
|
-
|
|
2708
|
-
|
|
2709
|
-
|
|
2710
|
-
|
|
2711
|
-
|
|
2712
|
-
|
|
2713
|
-
|
|
2714
|
-
|
|
2715
|
-
|
|
2716
|
-
|
|
2717
|
-
|
|
2718
|
-
|
|
2719
|
-
|
|
2720
|
-
|
|
2721
|
-
|
|
2759
|
+
if (backend.kind === "remote") {
|
|
2760
|
+
// No whisper binary and no model file on this branch — the whole point
|
|
2761
|
+
// of remote is that neither is installed (2026-09-01 field report).
|
|
2762
|
+
// ffmpeg still is: the upload sidecar is an encode.
|
|
2763
|
+
const host = remoteWhisperHost(backend.baseUrl);
|
|
2764
|
+
const uploadPath = join(work, "audio-upload.ogg");
|
|
2765
|
+
transcript = await phases.time("transcribe", async () => {
|
|
2766
|
+
await encodeUploadAudio(tools, audioPath, uploadPath);
|
|
2767
|
+
const bytes = statSync(uploadPath).size;
|
|
2768
|
+
if (bytes > REMOTE_UPLOAD_MAX_BYTES) {
|
|
2769
|
+
// Named here rather than paid for as somebody else's 413 after the
|
|
2770
|
+
// whole upload: chunking is out of scope for v1, so the error has
|
|
2771
|
+
// to carry both escape hatches itself.
|
|
2772
|
+
throw new Error(
|
|
2773
|
+
`the compressed audio is ${(bytes / 1_000_000).toFixed(1)} MB, over the ` +
|
|
2774
|
+
`${(REMOTE_UPLOAD_MAX_BYTES / 1_000_000).toFixed(0)} MB single-file limit for remote ` +
|
|
2775
|
+
`transcription (about 100 minutes of speech at this bitrate).\n` +
|
|
2776
|
+
`Transcribe locally with --whisper-backend local, split the take, or point ` +
|
|
2777
|
+
`OSSCLIP_WHISPER_URL at a server with a larger cap (Groq's dev tier allows 100 MB).`,
|
|
2778
|
+
);
|
|
2779
|
+
}
|
|
2780
|
+
const anim = isInteractive()
|
|
2781
|
+
? new StageAnimator(
|
|
2782
|
+
"REMOTE ASR",
|
|
2783
|
+
`Transcribing via ${host} (${backend.model})...`,
|
|
2784
|
+
"whisper",
|
|
2785
|
+
).start()
|
|
2786
|
+
: null;
|
|
2787
|
+
if (!anim) console.log(`▸ transcribing remotely (${host}, ${backend.model})…`);
|
|
2788
|
+
try {
|
|
2789
|
+
return await createOpenAiCompatibleProvider({
|
|
2790
|
+
baseUrl: backend.baseUrl,
|
|
2791
|
+
model: backend.model,
|
|
2792
|
+
...(backend.apiKey !== undefined ? { apiKey: backend.apiKey } : {}),
|
|
2793
|
+
}).transcribe(uploadPath, {
|
|
2794
|
+
// From the KEY, like the local branch: whatever re-keys the cache
|
|
2795
|
+
// is what actually decoded, so the two can never disagree.
|
|
2796
|
+
language: requestedKey.language,
|
|
2797
|
+
prompt: whisperPromptFor(dictionary),
|
|
2798
|
+
});
|
|
2799
|
+
} finally {
|
|
2800
|
+
// In a finally, unlike the local branch's trailing stop(): an HTTP
|
|
2801
|
+
// failure here is EXPECTED (a wrong key, a rate limit), and a
|
|
2802
|
+
// spinner still animating would overwrite the hint the user needs.
|
|
2803
|
+
anim?.stop();
|
|
2804
|
+
}
|
|
2805
|
+
});
|
|
2806
|
+
} else {
|
|
2807
|
+
await preflight(
|
|
2808
|
+
cfg.whisperPath,
|
|
2809
|
+
"Run `ossclip setup`, install whisper.cpp yourself (https://github.com/ggml-org/whisper.cpp), or set OSSCLIP_WHISPER.",
|
|
2810
|
+
);
|
|
2811
|
+
const model = requestedKey.model;
|
|
2812
|
+
// whisperModelPath/modelUrl are THE resolution and URL sources (shared
|
|
2813
|
+
// with doctor and setup) — this error used to hold its own copy of the
|
|
2814
|
+
// ggerganov URL, which 404'd for curated/custom names and the suggested
|
|
2815
|
+
// `curl -L` then saved the 404 HTML as a fake model.
|
|
2816
|
+
const modelPath = whisperModelPath(model, cfg.modelDir);
|
|
2817
|
+
if (!existsSync(modelPath)) {
|
|
2818
|
+
throw new Error(
|
|
2819
|
+
`whisper model not found at ${modelPath}.\n` +
|
|
2820
|
+
`Run \`ossclip setup${model === cfg.model ? "" : ` --model ${model}`}\` to download it — or manually:\n` +
|
|
2821
|
+
` curl -L -o ${modelPath} ${modelUrl(model, validModelSources(cfg.modelSources))}`,
|
|
2822
|
+
);
|
|
2823
|
+
}
|
|
2824
|
+
const whisperAnim = isInteractive()
|
|
2825
|
+
? new StageAnimator(
|
|
2826
|
+
"WHISPER ASR",
|
|
2827
|
+
`Transcribing audio stream with ${basename(modelPath)}...`,
|
|
2828
|
+
"whisper",
|
|
2829
|
+
).start()
|
|
2830
|
+
: null;
|
|
2831
|
+
if (!whisperAnim) console.log(`▸ transcribing (${basename(modelPath)})…`);
|
|
2832
|
+
transcript = await phases.time("transcribe", () =>
|
|
2833
|
+
runWhisper(
|
|
2834
|
+
{
|
|
2835
|
+
whisperPath: cfg.whisperPath,
|
|
2836
|
+
modelPath,
|
|
2837
|
+
outBase: join(work, "whisper"),
|
|
2838
|
+
// The RESOLVED language, not the raw flag — a config/model-implied
|
|
2839
|
+
// code must reach the spawn exactly as it reached the cache key.
|
|
2840
|
+
language: requestedKey.language,
|
|
2841
|
+
// Vocabulary biasing (F4) — undefined for an empty dictionary, so
|
|
2842
|
+
// the spawned args stay byte-identical to every pre-dictionary run.
|
|
2843
|
+
// From the KEY, like the language: whatever re-keys the cache is
|
|
2844
|
+
// what actually ran, so the two can never disagree.
|
|
2845
|
+
...(requestedKey.translate === true ? { translate: true } : {}),
|
|
2846
|
+
prompt: whisperPromptFor(dictionary),
|
|
2847
|
+
},
|
|
2848
|
+
audioPath,
|
|
2849
|
+
),
|
|
2722
2850
|
);
|
|
2851
|
+
if (whisperAnim) whisperAnim.stop();
|
|
2723
2852
|
}
|
|
2724
|
-
const whisperAnim = isInteractive()
|
|
2725
|
-
? new StageAnimator(
|
|
2726
|
-
"WHISPER ASR",
|
|
2727
|
-
`Transcribing audio stream with ${basename(modelPath)}...`,
|
|
2728
|
-
"whisper",
|
|
2729
|
-
).start()
|
|
2730
|
-
: null;
|
|
2731
|
-
if (!whisperAnim) console.log(`▸ transcribing (${basename(modelPath)})…`);
|
|
2732
|
-
transcript = await phases.time("transcribe", () =>
|
|
2733
|
-
runWhisper(
|
|
2734
|
-
{
|
|
2735
|
-
whisperPath: cfg.whisperPath,
|
|
2736
|
-
modelPath,
|
|
2737
|
-
outBase: join(work, "whisper"),
|
|
2738
|
-
// The RESOLVED language, not the raw flag — a config/model-implied
|
|
2739
|
-
// code must reach the spawn exactly as it reached the cache key.
|
|
2740
|
-
language: requestedKey.language,
|
|
2741
|
-
// Vocabulary biasing (F4) — undefined for an empty dictionary, so
|
|
2742
|
-
// the spawned args stay byte-identical to every pre-dictionary run.
|
|
2743
|
-
// From the KEY, like the language: whatever re-keys the cache is
|
|
2744
|
-
// what actually ran, so the two can never disagree.
|
|
2745
|
-
...(requestedKey.translate === true ? { translate: true } : {}),
|
|
2746
|
-
prompt: whisperPromptFor(dictionary),
|
|
2747
|
-
},
|
|
2748
|
-
audioPath,
|
|
2749
|
-
),
|
|
2750
|
-
);
|
|
2751
|
-
if (whisperAnim) whisperAnim.stop();
|
|
2752
2853
|
console.log(`▸ transcribed ${transcript.words.length} words`);
|
|
2753
2854
|
await writeFile(transcriptKeyPath, JSON.stringify(requestedKey, null, 2));
|
|
2754
2855
|
}
|