ossclip 0.1.36 → 0.1.37

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -2,7 +2,7 @@
2
2
 
3
3
  **A local-first CLI that turns a talking-head take into a finished short.** It cuts silence and filler words, writes word-timed kinetic captions, frames on the measured face, and has an LLM plan **code-rendered on-screen graphics** — title cards, stat cards, diagrams, terminal and chat mockups — from what was actually said.
4
4
 
5
- Transcription is local (whisper.cpp), rendering is local (Remotion); the only network calls are the LLM planning ones, on your own key or your existing Claude Code / Google Antigravity subscription. Vertical 9:16 by default, landscape 16:9 with `--aspect`.
5
+ Transcription is local (whisper.cpp — a remote server is opt-in, see below), rendering is local (Remotion); the only network calls are the LLM planning ones, on your own key or your existing Claude Code / Google Antigravity subscription. Vertical 9:16 by default, landscape 16:9 with `--aspect`.
6
6
 
7
7
  ![Left: the raw take. Right: the same take after ossclip produce — cut down, reframed on the measured face, word-timed captions](https://raw.githubusercontent.com/AhsanAyaz/ossclip/main/docs/site/assets/before-after.gif)
8
8
 
@@ -40,8 +40,18 @@ ossclip edit "<work directory>"
40
40
 
41
41
  **Scope, honestly:** ossclip is at its best polishing a take you have already cut down. `--clip` selects a single strongest window from long-form input — one clip, not N.
42
42
 
43
+ **Color grading:** `--color-grade <preset|file.cube>` grades the footage — six presets (`talking-head`, `teal-orange`, `filmic-fade`, `cwa`, `punchy`, `mono`), or any `.cube` LUT you drop in `~/.ossclip/luts`. Presets preview live in the editor, where intensity, exposure, temperature, saturation and contrast are sliders; a LUT bakes into the mezzanine at render time. `"colorGrade"` in `~/.ossclip/config.json` sets a channel's look once, and `--no-color-grade` is a hard off for one run.
44
+
45
+ **Sound effects:** `--sfx` places whooshes, dings and risers on the beats the producer planned (`--sfx-level subtle|normal|meme` — `meme` is what unlocks the record scratch). Needs `--produce`. A CC0 starter pack ships with the CLI; your own packs live in `~/.ossclip/sfx`. The editor gets an SFX lane: retime, swap, gain, mute-with-a-way-back, or add one at the playhead, audible in the preview.
46
+
47
+ **Keep a 4K source's pixels:** `--resolution auto` renders at what the source still has after the crop (capped at 2160, never under 1080p) instead of forcing 1080p — invisible on the platforms that cap at 1080p, real on YouTube. `1080` (default), `1440` and `2160` are the fixed choices.
48
+
49
+ **Frame 1 is the cover on some platforms:** `--cover-in-video` overlays the cover image on the opening frames — nothing is inserted, the overlay ends at the first spoken word, so no audio or caption timing moves. Off by default; `"coverInVideo": true` in the config turns it on for good.
50
+
43
51
  **Keep your own editor:** `ossclip analyze` exports every suggested cut as a labelled span marker — reason, duration, confidence — in the dialect your NLE actually reads (`premiere-xml`, `resolve-edl` with colours, or `fcpxml` for Final Cut). Review the markers, cut in your own timeline; nothing is applied for you. Import steps per NLE are in the repo README.
44
52
 
53
+ **Weak CPU?** On an older machine whisper is the dominant cost of a run, so transcription can go to any OpenAI-compatible `/v1/audio/transcriptions` server instead — [Groq](https://console.groq.com)'s free tier, or your own [speaches](https://github.com/speaches-ai/speaches)/whisper.cpp server, which needs no key. Export `OSSCLIP_WHISPER_URL` (plus `OSSCLIP_WHISPER_API_KEY` where the server wants one) and every run transcribes remotely; `--whisper-backend local` opts a single run back out, and `ossclip doctor` and `ossclip setup` stop asking for whisper.cpp and the model file. Your audio leaves the machine on that path — which is why it is off until you configure it. Caps and self-hosting notes in the repo README.
54
+
45
55
  AI can make mistakes: the cut, the captions and every graphic are generated — review the output before publishing.
46
56
 
47
57
  **Telemetry:** ossclip sends anonymous usage events (counts, durations, provider name) — never footage, transcripts, file names or paths. Turn it off any time with `ossclip telemetry off`, `OSSCLIP_TELEMETRY=0`, or `DO_NOT_TRACK=1`. Full detail in the repo README's Telemetry section.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "ossclip",
3
- "version": "0.1.36",
3
+ "version": "0.1.37",
4
4
  "description": "Local-first CLI video producer: cuts silence and fillers, word-timed captions, face-aware framing, and LLM-planned code-rendered graphics — transcription and rendering never leave your machine",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -36,9 +36,9 @@
36
36
  "commander": "^12.1.0",
37
37
  "tsx": "^4.19.0",
38
38
  "zod": "^3.25.76",
39
- "@ossclip/core": "0.1.36",
40
- "@ossclip/renderer": "0.1.36",
41
- "@ossclip/scenes": "0.1.36"
39
+ "@ossclip/renderer": "0.1.37",
40
+ "@ossclip/core": "0.1.37",
41
+ "@ossclip/scenes": "0.1.37"
42
42
  },
43
43
  "homepage": "https://github.com/AhsanAyaz/ossclip#readme",
44
44
  "bugs": {
package/src/analyze.ts CHANGED
@@ -117,6 +117,9 @@ export interface AnalyzeOptions {
117
117
  noiseDb?: number;
118
118
  whisperModel?: string;
119
119
  whisperLanguage?: string;
120
+ /** `--whisper-backend`, already zod-parsed by program.ts; undefined lets a
121
+ * configured `whisperUrl` decide (resolveWhisperBackend in produce). */
122
+ whisperBackend?: "local" | "remote";
120
123
  blooperMarker?: string;
121
124
  collapseRetakes?: boolean;
122
125
  sort?: "name" | "mtime";
@@ -157,6 +160,7 @@ export async function runAnalyze(
157
160
  noiseDb: opts.noiseDb,
158
161
  whisperModel: opts.whisperModel,
159
162
  whisperLanguage: opts.whisperLanguage,
163
+ whisperBackend: opts.whisperBackend,
160
164
  blooperMarker: opts.blooperMarker,
161
165
  collapseRetakes: opts.collapseRetakes,
162
166
  sort: opts.sort,
package/src/doctor.ts CHANGED
@@ -2,6 +2,7 @@ import { spawn } from "node:child_process";
2
2
  import { existsSync } from "node:fs";
3
3
  import type { OssclipConfig } from "@ossclip/core";
4
4
  import { modelUrl, validModelSources, whisperModelPath } from "./setup/manifest";
5
+ import { WHISPER_API_KEY_ENV, resolveWhisperBackend } from "./whisper-backend";
5
6
 
6
7
  /**
7
8
  * `ossclip doctor` (R18 §90a): check every prerequisite and print the exact
@@ -106,12 +107,23 @@ export async function runDoctor(cfg: OssclipConfig, p: DoctorProbes): Promise<Do
106
107
  }),
107
108
  });
108
109
 
110
+ // Remote transcription (2026-09-01 weak-CPU field report) makes the next
111
+ // two checks OPTIONAL rather than blocking: a machine that transcribes on
112
+ // Groq has no reason to own whisper.cpp or a 1.5 GB model, and doctor
113
+ // reporting two red lines on a working install is how a user concludes the
114
+ // tool is broken. The LLM-provider posture, one level up: pass with a
115
+ // detail that says why nothing is needed.
116
+ const remote = resolveWhisperBackend(undefined, cfg, p.env);
117
+ const remoteBackend = remote.ok && remote.backend.kind === "remote" ? remote.backend : null;
118
+ const notNeeded = (found: string): string =>
119
+ `${found} not found — not needed: remote transcription configured`;
120
+
109
121
  const whisperOk = await p.binRuns(cfg.whisperPath, "--help");
110
122
  checks.push({
111
123
  name: "whisper-cli",
112
- ok: whisperOk,
113
- detail: cfg.whisperPath,
114
- ...(whisperOk
124
+ ok: whisperOk || remoteBackend !== null,
125
+ detail: whisperOk ? cfg.whisperPath : remoteBackend !== null ? notNeeded(cfg.whisperPath) : cfg.whisperPath,
126
+ ...(whisperOk || remoteBackend !== null
115
127
  ? {}
116
128
  : {
117
129
  fix: viaSetup(
@@ -134,9 +146,9 @@ export async function runDoctor(cfg: OssclipConfig, p: DoctorProbes): Promise<Do
134
146
  const modelOk = p.exists(modelPath);
135
147
  checks.push({
136
148
  name: `whisper model (${cfg.model})`,
137
- ok: modelOk,
138
- detail: modelPath,
139
- ...(modelOk
149
+ ok: modelOk || remoteBackend !== null,
150
+ detail: modelOk ? modelPath : remoteBackend !== null ? notNeeded(modelPath) : modelPath,
151
+ ...(modelOk || remoteBackend !== null
140
152
  ? {}
141
153
  : {
142
154
  fix: viaSetup(
@@ -146,6 +158,26 @@ export async function runDoctor(cfg: OssclipConfig, p: DoctorProbes): Promise<Do
146
158
  }),
147
159
  });
148
160
 
161
+ // NO network call, unlike every other backend doctor could probe: a
162
+ // transcription request costs the user's metered free tier, and `doctor` is
163
+ // run repeatedly while fixing something else. This line reports the
164
+ // CONFIGURATION — the three things a 401 or a 404 would be about — and the
165
+ // provider's own status hints name the rest when a real run happens.
166
+ // Omitted entirely when remote is not configured: the local install is the
167
+ // default, and an extra "not configured" line for an opt-in feature is
168
+ // noise on every other machine.
169
+ if (remoteBackend !== null) {
170
+ checks.push({
171
+ name: "remote transcription",
172
+ ok: true,
173
+ detail:
174
+ `${remoteBackend.baseUrl} · model ${remoteBackend.model} · ` +
175
+ (remoteBackend.apiKey !== undefined
176
+ ? `${WHISPER_API_KEY_ENV} set`
177
+ : `no API key (fine for self-hosted; Groq needs ${WHISPER_API_KEY_ENV})`),
178
+ });
179
+ }
180
+
149
181
  // Provider, in the same order auto-detection uses (agy → claude CLI →
150
182
  // gemini key → anthropic key): subscription CLIs beat ambient env keys
151
183
  // since 2026-08 — a logged-in CLI is an explicit, already-paid choice
package/src/edit.ts CHANGED
@@ -70,6 +70,11 @@ import {
70
70
  portraitMimeType,
71
71
  readCoverProvenance,
72
72
  runWhisper,
73
+ // The remote transcription backend (2026-09-01) — the span re-decode goes
74
+ // through it whenever `whisperUrl` is configured.
75
+ createOpenAiCompatibleProvider,
76
+ type TranscribeRequest,
77
+ type Transcript,
73
78
  SegmentSchema,
74
79
  spliceTranscript,
75
80
  TranscriptSchema,
@@ -116,6 +121,7 @@ import { expandHome } from "./paths";
116
121
  // produce all resolve through — a second copy here would send a user to a
117
122
  // model file the rest of the tool never looks for.
118
123
  import { modelImpliedLanguage, whisperModelPath } from "./setup/manifest";
124
+ import { resolveWhisperBackend } from "./whisper-backend";
119
125
  import {
120
126
  PORTRAIT_OVERRIDE_BASENAME,
121
127
  portraitExtensionForMime,
@@ -198,6 +204,37 @@ export interface RetranscribeConfig {
198
204
  modelDir?: string;
199
205
  language?: unknown;
200
206
  dictionary?: unknown;
207
+ /** The remote-backend pair (2026-09-01): a configured URL means this span
208
+ * is decoded by an OpenAI-compatible server instead of whisper.cpp, so the
209
+ * server must see the same two keys `resolveWhisperBackend` reads. */
210
+ whisperUrl?: string;
211
+ whisperRemoteModel?: string;
212
+ }
213
+
214
+ /**
215
+ * The half of a range re-decode that is the same on BOTH backends: the
216
+ * decoder bias. Extracted (2026-09-01) so the remote path shares these rules
217
+ * verbatim instead of growing a second copy that drifts — the validation is
218
+ * the load-bearing part, and it is unchanged from the local-only version.
219
+ */
220
+ function retranscribeBias(cfg: RetranscribeConfig): { language?: string; prompt?: string } {
221
+ const dict = Array.isArray(cfg.dictionary)
222
+ && cfg.dictionary.length > 0
223
+ && cfg.dictionary.every((t) => typeof t === "string" && t.trim().length > 0)
224
+ ? (cfg.dictionary as string[]).map((t) => t.trim())
225
+ : [];
226
+ // `cfg.model` can be absent on the remote backend (a machine that never
227
+ // installed a local model), and no model name implies no language.
228
+ const language = typeof cfg.language === "string" && cfg.language.trim().length > 0
229
+ ? cfg.language.trim()
230
+ : cfg.model !== undefined
231
+ ? modelImpliedLanguage(cfg.model)
232
+ : undefined;
233
+ const prompt = whisperPromptFor(dict);
234
+ return {
235
+ ...(language !== undefined ? { language } : {}),
236
+ ...(prompt !== undefined ? { prompt } : {}),
237
+ };
201
238
  }
202
239
 
203
240
  /**
@@ -238,21 +275,38 @@ export function retranscribeSettings(
238
275
  "then `ossclip setup` to install it.",
239
276
  };
240
277
  }
241
- const dict = Array.isArray(cfg.dictionary)
242
- && cfg.dictionary.length > 0
243
- && cfg.dictionary.every((t) => typeof t === "string" && t.trim().length > 0)
244
- ? (cfg.dictionary as string[]).map((t) => t.trim())
245
- : [];
246
- const language = typeof cfg.language === "string" && cfg.language.trim().length > 0
247
- ? cfg.language.trim()
248
- : modelImpliedLanguage(cfg.model);
249
- const prompt = whisperPromptFor(dict);
250
278
  return {
251
279
  tools: { ffmpegPath: cfg.ffmpegPath, ffprobePath: cfg.ffprobePath },
252
280
  whisperPath: cfg.whisperPath,
253
281
  modelPath: whisperModelPath(cfg.model, cfg.modelDir),
254
- ...(language !== undefined ? { language } : {}),
255
- ...(prompt !== undefined ? { prompt } : {}),
282
+ ...retranscribeBias(cfg),
283
+ };
284
+ }
285
+
286
+ /**
287
+ * The same thing for the REMOTE backend (2026-09-01): no whisper binary and
288
+ * no model file, because the point of remote is that neither is installed —
289
+ * but ffmpeg still is, since the span is sliced out of `audio.wav` locally
290
+ * before it is uploaded.
291
+ *
292
+ * Pure, like its local twin: the whole "remote configured but ffmpeg isn't"
293
+ * corner is testable without a network.
294
+ */
295
+ export function retranscribeRemoteSettings(
296
+ cfg: RetranscribeConfig,
297
+ ):
298
+ | { tools: { ffmpegPath: string; ffprobePath: string }; language?: string; prompt?: string }
299
+ | { error: string } {
300
+ if (!cfg.ffmpegPath || !cfg.ffprobePath) {
301
+ return {
302
+ error:
303
+ "ffmpeg is not configured — run `ossclip doctor` to see what is missing, " +
304
+ "then `ossclip setup` to install it. (Remote transcription still slices the span locally.)",
305
+ };
306
+ }
307
+ return {
308
+ tools: { ffmpegPath: cfg.ffmpegPath, ffprobePath: cfg.ffprobePath },
309
+ ...retranscribeBias(cfg),
256
310
  };
257
311
  }
258
312
 
@@ -484,6 +538,13 @@ export async function startEditServer(
484
538
  */
485
539
  sliceAudio?: typeof extractAudioSpan;
486
540
  runWhisper?: typeof runWhisper;
541
+ /**
542
+ * The remote backend's half of the `runWhisper` seam (2026-09-01): with a
543
+ * `whisperUrl` configured the span goes to an OpenAI-compatible server
544
+ * instead of whisper.cpp, and a test must be able to observe that —
545
+ * including the failure sentence — without a network or an API key.
546
+ */
547
+ transcribeRemote?: (wavPath: string, req: TranscribeRequest) => Promise<Transcript>;
487
548
  /**
488
549
  * The sound library the SFX routes serve (`loadCfg`'s rule applied to the
489
550
  * pack loader): tests inject a hand-written library over a tmp dir, so the
@@ -955,17 +1016,67 @@ export async function startEditServer(
955
1016
  error: "this workdir has no transcript.json to re-stamp — re-run `ossclip produce`.",
956
1017
  });
957
1018
  }
958
- const settings = retranscribeSettings((opts.loadCfg ?? loadConfig)());
959
- if ("error" in settings) return send(200, { ok: false, error: settings.error });
960
- if (!existsSync(settings.modelPath)) {
961
- // The `--transcript`-only install: whisper was never needed to
962
- // make this project, so say what to run rather than 500ing.
963
- return send(200, {
964
- ok: false,
965
- error:
966
- `whisper model not found at ${settings.modelPath} — run \`ossclip setup\` ` +
967
- `to download it.`,
1019
+ const cfg = (opts.loadCfg ?? loadConfig)();
1020
+ // Which engine re-decodes the span, resolved HERE rather than in
1021
+ // retranscribeSettings (2026-09-01): the flag is a CLI thing and
1022
+ // there is no CLI in this loop, so a configured `whisperUrl` is
1023
+ // the whole switch. `undefined` as the flag can only answer ok —
1024
+ // only an explicit `--whisper-backend remote` with nothing
1025
+ // configured fails — so this is a narrowing, not a live branch.
1026
+ const backendPick = resolveWhisperBackend(undefined, cfg, process.env);
1027
+ if (!backendPick.ok) return send(200, { ok: false, error: backendPick.message });
1028
+ const backend = backendPick.backend;
1029
+ // One plan, two shapes: the local one needs a binary and a model
1030
+ // file on disk, the remote one needs neither. Both need ffmpeg —
1031
+ // the span is always sliced here.
1032
+ let plan: {
1033
+ tools: { ffmpegPath: string; ffprobePath: string };
1034
+ language?: string;
1035
+ prompt?: string;
1036
+ decode: (wavPath: string, req: TranscribeRequest) => Promise<Transcript>;
1037
+ };
1038
+ if (backend.kind === "remote") {
1039
+ const settings = retranscribeRemoteSettings(cfg);
1040
+ if ("error" in settings) return send(200, { ok: false, error: settings.error });
1041
+ const provider = createOpenAiCompatibleProvider({
1042
+ baseUrl: backend.baseUrl,
1043
+ model: backend.model,
1044
+ ...(backend.apiKey !== undefined ? { apiKey: backend.apiKey } : {}),
968
1045
  });
1046
+ plan = {
1047
+ ...settings,
1048
+ // Uploaded AS-IS, no opus sidecar (produce's rule does not
1049
+ // apply): a span is seconds long, so its wav is far under any
1050
+ // upload cap and the encode would cost more than it saves.
1051
+ decode: opts.transcribeRemote ?? ((wavPath, r) => provider.transcribe(wavPath, r)),
1052
+ };
1053
+ } else {
1054
+ const settings = retranscribeSettings(cfg);
1055
+ if ("error" in settings) return send(200, { ok: false, error: settings.error });
1056
+ if (!existsSync(settings.modelPath)) {
1057
+ // The `--transcript`-only install: whisper was never needed to
1058
+ // make this project, so say what to run rather than 500ing.
1059
+ return send(200, {
1060
+ ok: false,
1061
+ error:
1062
+ `whisper model not found at ${settings.modelPath} — run \`ossclip setup\` ` +
1063
+ `to download it.`,
1064
+ });
1065
+ }
1066
+ plan = {
1067
+ ...settings,
1068
+ decode: (wavPath, r) =>
1069
+ (opts.runWhisper ?? runWhisper)(
1070
+ {
1071
+ whisperPath: settings.whisperPath,
1072
+ modelPath: settings.modelPath,
1073
+ outBase,
1074
+ ...(r.language !== undefined ? { language: r.language } : {}),
1075
+ ...(r.prompt !== undefined ? { prompt: r.prompt } : {}),
1076
+ },
1077
+ wavPath,
1078
+ ),
1079
+ };
969
1080
  }
970
1081
  // Parsed, not cast: this file is about to be rewritten, and a
971
1082
  // truncated one must fail loudly here rather than become the new
@@ -985,22 +1096,16 @@ export async function startEditServer(
985
1096
  });
986
1097
  }
987
1098
  await (opts.sliceAudio ?? extractAudioSpan)(
988
- settings.tools,
1099
+ plan.tools,
989
1100
  audio,
990
1101
  tmpWav,
991
1102
  srcIn,
992
1103
  srcOut - srcIn,
993
1104
  );
994
- const fresh = await (opts.runWhisper ?? runWhisper)(
995
- {
996
- whisperPath: settings.whisperPath,
997
- modelPath: settings.modelPath,
998
- outBase,
999
- ...(settings.language !== undefined ? { language: settings.language } : {}),
1000
- ...(settings.prompt !== undefined ? { prompt: settings.prompt } : {}),
1001
- },
1002
- tmpWav,
1003
- );
1105
+ const fresh = await plan.decode(tmpWav, {
1106
+ ...(plan.language !== undefined ? { language: plan.language } : {}),
1107
+ ...(plan.prompt !== undefined ? { prompt: plan.prompt } : {}),
1108
+ });
1004
1109
  const restamped = alignRestamp(
1005
1110
  transcript.words.slice(range.from, range.to),
1006
1111
  fresh.words,
@@ -10,7 +10,7 @@ import { produceArgv, type ProduceAnswers, type ProduceExtras } from "./produce-
10
10
  import { assertInteractive, confirm, intro, multiselect, select, text, unwrap } from "./prompts";
11
11
 
12
12
  /**
13
- * The produce wizard. Forty-three flags (plus the positional input path)
13
+ * The produce wizard. Forty-four flags (plus the positional input path)
14
14
  * sorted into three tiers: six prompts asked directly — the input path, plus
15
15
  * five flags (--out, --cleanup, --aspect, --produce, --intent) — twelve
16
16
  * behind one "anything else?" multiselect (--sfx being the twelfth, with
@@ -28,7 +28,12 @@ import { assertInteractive, confirm, intro, multiselect, select, text, unwrap }
28
28
  * multiselect entry is the OFF switch and the positive flag exists only for
29
29
  * replay pinning), --add-jump-cuts (same mirror: auto already punches, the
30
30
  * multiselect entry is the OFF switch, and the force flag exists to beat a
31
- * future config-off), or
31
+ * future config-off),
32
+ * --whisper-backend (2026-09-01, the --color-grade shape: it selects machine
33
+ * INFRASTRUCTURE — which transcription engine this box owns, alongside
34
+ * whisperPath and modelDir — not a per-run editorial choice, and the durable
35
+ * spelling is `whisperUrl` in ~/.ossclip/config.json; an honest prompt would
36
+ * also have to explain a base URL and an API key at a menu), or
32
37
  * (final-review fix wave, Finding 1) --sort. A folder's clip order only means anything once the
33
38
  * folder has been enumerated, and that enumeration happens inside
34
39
  * `produce()` — after the wizard has already returned argv — so there is
package/src/produce.ts CHANGED
@@ -66,6 +66,10 @@ import {
66
66
  splitThenDropHidden,
67
67
  emptyOverrideDoc,
68
68
  extractAudio,
69
+ encodeUploadAudio,
70
+ REMOTE_UPLOAD_MAX_BYTES,
71
+ createOpenAiCompatibleProvider,
72
+ openaiTranscriptionsUrl,
69
73
  fillPlainCues,
70
74
  splitCues,
71
75
  landscapeLayout,
@@ -249,6 +253,7 @@ import { RenderTimelineHUD, StageAnimator, printProductionCompleteBanner } from
249
253
  import { reconcileCaptionEdits } from "./caption-report";
250
254
  import { overridesWriteLine, writeOverrideDoc } from "./overrides-write";
251
255
  import { recordedProduceArgs } from "./replay-argv";
256
+ import { remoteWhisperHost, resolveWhisperBackend } from "./whisper-backend";
252
257
  import { makeCancelSignal, renderCover, renderProduction } from "@ossclip/renderer";
253
258
  import type { RenderPhase } from "@ossclip/renderer";
254
259
  import {
@@ -311,6 +316,17 @@ export const TranscriptKeySchema = z.object({
311
316
  * files) means "no translation", the `dictionary` contract.
312
317
  */
313
318
  translate: z.boolean().optional(),
319
+ /**
320
+ * Which BACKEND decoded it (2026-09-01): `remote:<normalized endpoint>`, or
321
+ * absent for local whisper.cpp — the `dictionary`/`translate` contract, so
322
+ * every key file written before remote existed still reads as local. Two
323
+ * engines on the same audio produce different words, so without this a warm
324
+ * workdir serves the local transcript to a remote run (and vice versa) —
325
+ * the exact staleness `language` and `translate` were added for. The
326
+ * remote MODEL name rides in `model` above, so switching Groq models
327
+ * re-keys through the existing field.
328
+ */
329
+ backend: z.string().optional(),
314
330
  });
315
331
  export type TranscriptKey = z.infer<typeof TranscriptKeySchema>;
316
332
 
@@ -338,6 +354,10 @@ export function transcriptCacheReusable(
338
354
  // Absent and false are the same "no translation", so pre-flag key
339
355
  // files reuse under a non-translate request.
340
356
  (effective.translate ?? false) === (requested.translate ?? false) &&
357
+ // Absent means LOCAL on both sides, so every pre-2026-09-01 key file
358
+ // still reuses under a local request — and a remote request against
359
+ // one of them re-transcribes, which is the point.
360
+ (effective.backend ?? "") === (requested.backend ?? "") &&
341
361
  // ORDER-SENSITIVE by choice: the dictionary becomes whisper's --prompt
342
362
  // text verbatim, so a reordered list genuinely is a different decoder
343
363
  // input — treating it as equal would serve a transcript biased by a
@@ -634,6 +654,13 @@ export interface ProduceOptions {
634
654
  * together (whisper decodes better knowing the source language).
635
655
  */
636
656
  whisperTranslate?: boolean;
657
+ /**
658
+ * `--whisper-backend`, already zod-parsed to the union by program.ts
659
+ * (2026-09-01 weak-CPU field report). Undefined means "not typed", which
660
+ * is what lets a configured `whisperUrl` select remote — the flag is
661
+ * mainly `local`, the per-run opt-out.
662
+ */
663
+ whisperBackend?: "local" | "remote";
637
664
  /**
638
665
  * Vocabulary terms for this run (`--dictionary`, F4 2026-08-16), already
639
666
  * split/trimmed by the action. Wholesale beats the config's `dictionary`
@@ -2664,9 +2691,34 @@ export async function produce(inputArg: string, opts: ProduceOptions): Promise<P
2664
2691
  `--whisper-language overrides)`,
2665
2692
  );
2666
2693
  }
2694
+ // Local whisper.cpp or an OpenAI-compatible server (2026-09-01 weak-CPU
2695
+ // field report). Resolved BEFORE the key, like the language, so what
2696
+ // actually decodes is what the cache is keyed on.
2697
+ const backendPick = resolveWhisperBackend(opts.whisperBackend, cfg, process.env);
2698
+ if (!backendPick.ok) throw new Error(backendPick.message);
2699
+ const backend = backendPick.backend;
2700
+ // BEFORE the key is built, so a translate request can never cross the
2701
+ // cache with a remote one: the OpenAI-compatible API translates on a
2702
+ // DIFFERENT endpoint with a DIFFERENT default model, and swapping both
2703
+ // behind one flag would be a surprise rather than a convenience.
2704
+ if (backend.kind === "remote" && opts.whisperTranslate === true) {
2705
+ throw new Error(
2706
+ "--whisper-translate needs the local backend (the OpenAI-compatible API translates on a " +
2707
+ "different endpoint and model) — use --whisper-backend local, or drop the flag.",
2708
+ );
2709
+ }
2667
2710
  const requestedKey: TranscriptKey = {
2668
- model: requestedModel,
2711
+ // The REMOTE model name when remote — one field, both engines, so an
2712
+ // A/B between two Groq models re-keys the cache exactly like a local one.
2713
+ model: backend.kind === "remote" ? backend.model : requestedModel,
2669
2714
  ...(whisperLang.language !== undefined ? { language: whisperLang.language } : {}),
2715
+ // Spread-omitted on local so local key files stay byte-identical to
2716
+ // every one written before remote existed (the translate posture). The
2717
+ // URL goes through openaiTranscriptionsUrl so ".../v1" and ".../v1/"
2718
+ // key identically — a trailing slash is not a different server.
2719
+ ...(backend.kind === "remote"
2720
+ ? { backend: `remote:${openaiTranscriptionsUrl(backend.baseUrl)}` }
2721
+ : {}),
2670
2722
  // Omitted when off, so a non-translate run's key stays byte-identical to
2671
2723
  // every pre-flag key file (the dictionary posture).
2672
2724
  ...(opts.whisperTranslate === true ? { translate: true } : {}),
@@ -2704,51 +2756,100 @@ export async function produce(inputArg: string, opts: ProduceOptions): Promise<P
2704
2756
  `re-transcribing with ${fmt(requestedKey)}`,
2705
2757
  );
2706
2758
  }
2707
- await preflight(
2708
- cfg.whisperPath,
2709
- "Run `ossclip setup`, install whisper.cpp yourself (https://github.com/ggml-org/whisper.cpp), or set OSSCLIP_WHISPER.",
2710
- );
2711
- const model = requestedKey.model;
2712
- // whisperModelPath/modelUrl are THE resolution and URL sources (shared
2713
- // with doctor and setup) — this error used to hold its own copy of the
2714
- // ggerganov URL, which 404'd for curated/custom names and the suggested
2715
- // `curl -L` then saved the 404 HTML as a fake model.
2716
- const modelPath = whisperModelPath(model, cfg.modelDir);
2717
- if (!existsSync(modelPath)) {
2718
- throw new Error(
2719
- `whisper model not found at ${modelPath}.\n` +
2720
- `Run \`ossclip setup${model === cfg.model ? "" : ` --model ${model}`}\` to download it — or manually:\n` +
2721
- ` curl -L -o ${modelPath} ${modelUrl(model, validModelSources(cfg.modelSources))}`,
2759
+ if (backend.kind === "remote") {
2760
+ // No whisper binary and no model file on this branch — the whole point
2761
+ // of remote is that neither is installed (2026-09-01 field report).
2762
+ // ffmpeg still is: the upload sidecar is an encode.
2763
+ const host = remoteWhisperHost(backend.baseUrl);
2764
+ const uploadPath = join(work, "audio-upload.ogg");
2765
+ transcript = await phases.time("transcribe", async () => {
2766
+ await encodeUploadAudio(tools, audioPath, uploadPath);
2767
+ const bytes = statSync(uploadPath).size;
2768
+ if (bytes > REMOTE_UPLOAD_MAX_BYTES) {
2769
+ // Named here rather than paid for as somebody else's 413 after the
2770
+ // whole upload: chunking is out of scope for v1, so the error has
2771
+ // to carry both escape hatches itself.
2772
+ throw new Error(
2773
+ `the compressed audio is ${(bytes / 1_000_000).toFixed(1)} MB, over the ` +
2774
+ `${(REMOTE_UPLOAD_MAX_BYTES / 1_000_000).toFixed(0)} MB single-file limit for remote ` +
2775
+ `transcription (about 100 minutes of speech at this bitrate).\n` +
2776
+ `Transcribe locally with --whisper-backend local, split the take, or point ` +
2777
+ `OSSCLIP_WHISPER_URL at a server with a larger cap (Groq's dev tier allows 100 MB).`,
2778
+ );
2779
+ }
2780
+ const anim = isInteractive()
2781
+ ? new StageAnimator(
2782
+ "REMOTE ASR",
2783
+ `Transcribing via ${host} (${backend.model})...`,
2784
+ "whisper",
2785
+ ).start()
2786
+ : null;
2787
+ if (!anim) console.log(`▸ transcribing remotely (${host}, ${backend.model})…`);
2788
+ try {
2789
+ return await createOpenAiCompatibleProvider({
2790
+ baseUrl: backend.baseUrl,
2791
+ model: backend.model,
2792
+ ...(backend.apiKey !== undefined ? { apiKey: backend.apiKey } : {}),
2793
+ }).transcribe(uploadPath, {
2794
+ // From the KEY, like the local branch: whatever re-keys the cache
2795
+ // is what actually decoded, so the two can never disagree.
2796
+ language: requestedKey.language,
2797
+ prompt: whisperPromptFor(dictionary),
2798
+ });
2799
+ } finally {
2800
+ // In a finally, unlike the local branch's trailing stop(): an HTTP
2801
+ // failure here is EXPECTED (a wrong key, a rate limit), and a
2802
+ // spinner still animating would overwrite the hint the user needs.
2803
+ anim?.stop();
2804
+ }
2805
+ });
2806
+ } else {
2807
+ await preflight(
2808
+ cfg.whisperPath,
2809
+ "Run `ossclip setup`, install whisper.cpp yourself (https://github.com/ggml-org/whisper.cpp), or set OSSCLIP_WHISPER.",
2810
+ );
2811
+ const model = requestedKey.model;
2812
+ // whisperModelPath/modelUrl are THE resolution and URL sources (shared
2813
+ // with doctor and setup) — this error used to hold its own copy of the
2814
+ // ggerganov URL, which 404'd for curated/custom names and the suggested
2815
+ // `curl -L` then saved the 404 HTML as a fake model.
2816
+ const modelPath = whisperModelPath(model, cfg.modelDir);
2817
+ if (!existsSync(modelPath)) {
2818
+ throw new Error(
2819
+ `whisper model not found at ${modelPath}.\n` +
2820
+ `Run \`ossclip setup${model === cfg.model ? "" : ` --model ${model}`}\` to download it — or manually:\n` +
2821
+ ` curl -L -o ${modelPath} ${modelUrl(model, validModelSources(cfg.modelSources))}`,
2822
+ );
2823
+ }
2824
+ const whisperAnim = isInteractive()
2825
+ ? new StageAnimator(
2826
+ "WHISPER ASR",
2827
+ `Transcribing audio stream with ${basename(modelPath)}...`,
2828
+ "whisper",
2829
+ ).start()
2830
+ : null;
2831
+ if (!whisperAnim) console.log(`▸ transcribing (${basename(modelPath)})…`);
2832
+ transcript = await phases.time("transcribe", () =>
2833
+ runWhisper(
2834
+ {
2835
+ whisperPath: cfg.whisperPath,
2836
+ modelPath,
2837
+ outBase: join(work, "whisper"),
2838
+ // The RESOLVED language, not the raw flag — a config/model-implied
2839
+ // code must reach the spawn exactly as it reached the cache key.
2840
+ language: requestedKey.language,
2841
+ // Vocabulary biasing (F4) — undefined for an empty dictionary, so
2842
+ // the spawned args stay byte-identical to every pre-dictionary run.
2843
+ // From the KEY, like the language: whatever re-keys the cache is
2844
+ // what actually ran, so the two can never disagree.
2845
+ ...(requestedKey.translate === true ? { translate: true } : {}),
2846
+ prompt: whisperPromptFor(dictionary),
2847
+ },
2848
+ audioPath,
2849
+ ),
2722
2850
  );
2851
+ if (whisperAnim) whisperAnim.stop();
2723
2852
  }
2724
- const whisperAnim = isInteractive()
2725
- ? new StageAnimator(
2726
- "WHISPER ASR",
2727
- `Transcribing audio stream with ${basename(modelPath)}...`,
2728
- "whisper",
2729
- ).start()
2730
- : null;
2731
- if (!whisperAnim) console.log(`▸ transcribing (${basename(modelPath)})…`);
2732
- transcript = await phases.time("transcribe", () =>
2733
- runWhisper(
2734
- {
2735
- whisperPath: cfg.whisperPath,
2736
- modelPath,
2737
- outBase: join(work, "whisper"),
2738
- // The RESOLVED language, not the raw flag — a config/model-implied
2739
- // code must reach the spawn exactly as it reached the cache key.
2740
- language: requestedKey.language,
2741
- // Vocabulary biasing (F4) — undefined for an empty dictionary, so
2742
- // the spawned args stay byte-identical to every pre-dictionary run.
2743
- // From the KEY, like the language: whatever re-keys the cache is
2744
- // what actually ran, so the two can never disagree.
2745
- ...(requestedKey.translate === true ? { translate: true } : {}),
2746
- prompt: whisperPromptFor(dictionary),
2747
- },
2748
- audioPath,
2749
- ),
2750
- );
2751
- if (whisperAnim) whisperAnim.stop();
2752
2853
  console.log(`▸ transcribed ${transcript.words.length} words`);
2753
2854
  await writeFile(transcriptKeyPath, JSON.stringify(requestedKey, null, 2));
2754
2855
  }
package/src/program.ts CHANGED
@@ -411,6 +411,12 @@ export function buildProgram(): Command {
411
411
  "(whisper's -tr; pair with --whisper-language for the SOURCE language)",
412
412
  false,
413
413
  )
414
+ .option(
415
+ "--whisper-backend <backend>",
416
+ "local | remote. local (default) runs whisper.cpp on this machine; remote posts the " +
417
+ "audio to the OpenAI-compatible server in OSSCLIP_WHISPER_URL (config: whisperUrl) — " +
418
+ "configuring that URL already implies remote, so this flag is mainly `local` to opt out",
419
+ )
414
420
  // COMMA-SEPARATED in one value, not variadic: a variadic option swallows
415
421
  // the optional positional [input] whenever the flag precedes the path,
416
422
  // and commander offers no way to give the positional priority.
@@ -661,6 +667,14 @@ export function buildProgram(): Command {
661
667
  opts.whisperLanguage !== undefined
662
668
  ? z.string().trim().min(1, "--whisper-language needs a code, e.g. ur").parse(opts.whisperLanguage)
663
669
  : undefined;
670
+ // An enum, unlike --whisper-language: there are exactly two backends,
671
+ // and a typo'd `--whisper-backend groq` silently running local whisper
672
+ // on the weak CPU the flag exists to spare is the --source-fit crop
673
+ // all over again.
674
+ const whisperBackend =
675
+ opts.whisperBackend !== undefined
676
+ ? z.enum(["local", "remote"]).parse(opts.whisperBackend)
677
+ : undefined;
664
678
  // --add-jump-cuts / --no-jump-cuts land on DIFFERENT commander keys
665
679
  // (see the option declarations for why the pair can't share one);
666
680
  // jumpCutsFlag reunites them into the tri-state ProduceOptions
@@ -720,6 +734,9 @@ export function buildProgram(): Command {
720
734
  whisperModel: opts.whisperModel,
721
735
  whisperLanguage,
722
736
  whisperTranslate: opts.whisperTranslate === true,
737
+ // undefined = "not typed", so a configured whisperUrl decides
738
+ // (resolveWhisperBackend at the use site).
739
+ whisperBackend,
723
740
  // Split/trim/drop-empties (dictionaryFlag) — undefined stays
724
741
  // undefined so the config's dictionary can supply the default.
725
742
  dictionary: dictionaryFlag(opts.dictionary),
@@ -855,6 +872,15 @@ export function buildProgram(): Command {
855
872
  "(whisper's -tr; pair with --whisper-language for the SOURCE language)",
856
873
  false,
857
874
  )
875
+ .option(
876
+ "--whisper-backend <backend>",
877
+ // Same sentence as produce's, deliberately: `transcribe` is the command
878
+ // a user drives while SETTING remote transcription up, so the half that
879
+ // says the URL is the real switch cannot be the half that is dropped here.
880
+ "local | remote. local (default) runs whisper.cpp on this machine; remote posts the " +
881
+ "audio to the OpenAI-compatible server in OSSCLIP_WHISPER_URL (config: whisperUrl) — " +
882
+ "configuring that URL already implies remote, so this flag is mainly `local` to opt out",
883
+ )
858
884
  .action(async (input: string, opts) => {
859
885
  const cleanup = CleanupLevelSchema.parse(opts.cleanup);
860
886
  const result = await produce(input, {
@@ -871,6 +897,12 @@ export function buildProgram(): Command {
871
897
  opts.whisperLanguage !== undefined
872
898
  ? z.string().trim().min(1, "--whisper-language needs a code, e.g. ur").parse(opts.whisperLanguage)
873
899
  : undefined,
900
+ // Parsed, not coerced — produce's reasoning: a typo must error, not
901
+ // fall back to the local backend the flag exists to avoid.
902
+ whisperBackend:
903
+ opts.whisperBackend !== undefined
904
+ ? z.enum(["local", "remote"]).parse(opts.whisperBackend)
905
+ : undefined,
874
906
  });
875
907
  telemetry.record("transcribe_completed", {
876
908
  cleanup_level: cleanup,
@@ -907,6 +939,15 @@ export function buildProgram(): Command {
907
939
  "--whisper-language <code>",
908
940
  "transcription language code for a multilingual model, e.g. ur | de | auto (whisper defaults to en)",
909
941
  )
942
+ .option(
943
+ "--whisper-backend <backend>",
944
+ // Same sentence as produce's, deliberately: `transcribe` is the command
945
+ // a user drives while SETTING remote transcription up, so the half that
946
+ // says the URL is the real switch cannot be the half that is dropped here.
947
+ "local | remote. local (default) runs whisper.cpp on this machine; remote posts the " +
948
+ "audio to the OpenAI-compatible server in OSSCLIP_WHISPER_URL (config: whisperUrl) — " +
949
+ "configuring that URL already implies remote, so this flag is mainly `local` to opt out",
950
+ )
910
951
  .option(
911
952
  "--blooper-marker <word>",
912
953
  "mark the flubbed take wherever you say this word out loud (e.g. blooper). Off unless given",
@@ -935,6 +976,12 @@ export function buildProgram(): Command {
935
976
  opts.whisperLanguage !== undefined
936
977
  ? z.string().trim().min(1, "--whisper-language needs a code, e.g. ur").parse(opts.whisperLanguage)
937
978
  : undefined,
979
+ // Parsed, not coerced — produce's reasoning: a typo must error, not
980
+ // fall back to the local backend the flag exists to avoid.
981
+ whisperBackend:
982
+ opts.whisperBackend !== undefined
983
+ ? z.enum(["local", "remote"]).parse(opts.whisperBackend)
984
+ : undefined,
938
985
  blooperMarker: opts.blooperMarker,
939
986
  collapseRetakes: opts.collapseRetakes,
940
987
  sort: opts.sort === "mtime" ? "mtime" : "name",
package/src/setup/plan.ts CHANGED
@@ -9,6 +9,7 @@ import {
9
9
  whisperAsset,
10
10
  whisperModelPath,
11
11
  } from "./manifest";
12
+ import { resolveWhisperBackend } from "../whisper-backend";
12
13
 
13
14
  /**
14
15
  * The planning half of `ossclip setup` — pure over injected probes, like
@@ -111,10 +112,25 @@ export async function planSetup(
111
112
  }
112
113
  }
113
114
 
115
+ // Remote transcription (2026-09-01 weak-CPU field report) makes whisper.cpp
116
+ // and the model OPTIONAL: on the machine the report came from, downloading
117
+ // a 1.5 GB model to run an engine that is too slow to use is exactly the
118
+ // cliff remote exists to remove. Reported as `satisfied` with the reason,
119
+ // never as a silent skip — and a local install that ALREADY works still
120
+ // reports itself (ground rule one: setup never uninstalls, and a user who
121
+ // has both keeps the `--whisper-backend local` escape hatch working).
122
+ const remote = resolveWhisperBackend(undefined, cfg, p.env);
123
+ const remoteDetail =
124
+ remote.ok && remote.backend.kind === "remote"
125
+ ? `remote transcription configured (${remote.backend.baseUrl}) — local whisper not needed`
126
+ : null;
127
+
114
128
  const whisperOk = await p.binRuns(cfg.whisperPath, "--help");
115
129
  const whisperForceable = isManaged(cfg.whisperPath, opts.configDir);
116
130
  if (whisperOk && !(opts.force && whisperForceable)) {
117
131
  steps.push({ kind: "whisper", status: "satisfied", detail: cfg.whisperPath });
132
+ } else if (remoteDetail !== null) {
133
+ steps.push({ kind: "whisper", status: "satisfied", detail: remoteDetail });
118
134
  } else {
119
135
  const asset = whisperAsset(p.platform, p.arch);
120
136
  if (asset) {
@@ -153,6 +169,8 @@ export async function planSetup(
153
169
  const known = MODELS[model];
154
170
  if (p.exists(modelPath)) {
155
171
  steps.push({ kind: "model", status: "satisfied", detail: modelPath });
172
+ } else if (remoteDetail !== null) {
173
+ steps.push({ kind: "model", status: "satisfied", detail: remoteDetail });
156
174
  } else if (isAbsolute(model)) {
157
175
  steps.push({
158
176
  kind: "model",
@@ -0,0 +1,103 @@
1
+ import type { OssclipConfig } from "@ossclip/core";
2
+
3
+ /**
4
+ * Which transcription backend a run uses — local whisper.cpp (the default,
5
+ * forever) or an OpenAI-compatible `/v1/audio/transcriptions` server.
6
+ *
7
+ * Why (2026-09-01 field report): on a weak CPU — an i3 2nd gen — whisper is
8
+ * the dominant cost of a produce run, and Groq's free tier makes it seconds.
9
+ * Remote is OPT-IN: presence of a URL is the switch, so nobody's existing
10
+ * install changes behavior by upgrading.
11
+ *
12
+ * `publishConfigured`'s mould (publish.ts), with one deliberate difference:
13
+ * the API key is OPTIONAL. Self-hosted servers (speaches, whisper.cpp
14
+ * server) run keyless, so requiring a key the way publish does would lock
15
+ * out the privacy-minded half of the audience.
16
+ *
17
+ * Pure over (flag, config, env) so the whole matrix is testable without a
18
+ * config file or a poked process.env.
19
+ */
20
+
21
+ /** Env-only, like every other secret (env.ts's rule): keys never live in config.json. */
22
+ export const WHISPER_API_KEY_ENV = "OSSCLIP_WHISPER_API_KEY";
23
+
24
+ /**
25
+ * Groq's word-timestamped turbo model — the one the quickstart in the README
26
+ * points at. The default lives HERE rather than in config.ts's DEFAULTS
27
+ * because it is only meaningful once a remote URL exists, and a self-hosted
28
+ * box (whose model names look like "Systran/faster-whisper-large-v3") sets
29
+ * `whisperRemoteModel` anyway.
30
+ */
31
+ export const DEFAULT_REMOTE_WHISPER_MODEL = "whisper-large-v3-turbo";
32
+
33
+ export type WhisperBackend =
34
+ | { kind: "local" }
35
+ | { kind: "remote"; baseUrl: string; model: string; apiKey?: string };
36
+
37
+ export type WhisperBackendResult =
38
+ | { ok: true; backend: WhisperBackend }
39
+ | { ok: false; message: string };
40
+
41
+ /**
42
+ * `--whisper-backend` (already zod-parsed by program.ts) beats the config,
43
+ * and "local" ALWAYS wins — it is the escape hatch a user reaches for when
44
+ * the remote server is down or the audio must not leave the machine, so it
45
+ * can never be overridden by a URL sitting in config.json.
46
+ *
47
+ * A typed `--whisper-backend remote` with nothing configured is an ERROR
48
+ * naming both spellings, not a silent fall back to local: the user asked for
49
+ * remote precisely because local is what they are trying to avoid.
50
+ */
51
+ export function resolveWhisperBackend(
52
+ flag: "local" | "remote" | undefined,
53
+ cfg: Pick<OssclipConfig, "whisperUrl" | "whisperRemoteModel">,
54
+ env: NodeJS.ProcessEnv,
55
+ ): WhisperBackendResult {
56
+ if (flag === "local") return { ok: true, backend: { kind: "local" } };
57
+ // typeof + trim, never truthiness: `whisperUrl: " "` in a hand-edited
58
+ // config.json must read as "not configured", not as a URL we then POST to.
59
+ const url = typeof cfg.whisperUrl === "string" ? cfg.whisperUrl.trim() : "";
60
+ if (url.length === 0) {
61
+ if (flag === "remote") {
62
+ return {
63
+ ok: false,
64
+ message:
65
+ "--whisper-backend remote needs a transcription server: set OSSCLIP_WHISPER_URL in the " +
66
+ 'environment (or ~/.ossclip/.env), or "whisperUrl" in ~/.ossclip/config.json — the ' +
67
+ "OpenAI-compatible base ending in /v1, e.g. https://api.groq.com/openai/v1.",
68
+ };
69
+ }
70
+ return { ok: true, backend: { kind: "local" } };
71
+ }
72
+ const model =
73
+ typeof cfg.whisperRemoteModel === "string" && cfg.whisperRemoteModel.trim().length > 0
74
+ ? cfg.whisperRemoteModel.trim()
75
+ : DEFAULT_REMOTE_WHISPER_MODEL;
76
+ const key = env[WHISPER_API_KEY_ENV]?.trim() ?? "";
77
+ return {
78
+ ok: true,
79
+ backend: {
80
+ kind: "remote",
81
+ baseUrl: url,
82
+ model,
83
+ // Omitted rather than "" when unset, so the provider sends NO
84
+ // Authorization header at all — a keyless self-hosted server is a
85
+ // supported configuration, not a missing key.
86
+ ...(key.length > 0 ? { apiKey: key } : {}),
87
+ },
88
+ };
89
+ }
90
+
91
+ /**
92
+ * The host for a one-line stage/status label. Falls back to the raw string
93
+ * when `new URL` refuses it: the value is user-typed config, and a stage line
94
+ * must never be the thing that throws — the POST that follows will report a
95
+ * bad URL with far better context.
96
+ */
97
+ export function remoteWhisperHost(baseUrl: string): string {
98
+ try {
99
+ return new URL(baseUrl).host;
100
+ } catch {
101
+ return baseUrl;
102
+ }
103
+ }