ossclip 0.1.36 → 0.1.38

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,7 +4,7 @@
4
4
  <meta charset="UTF-8" />
5
5
  <meta name="viewport" content="width=device-width, initial-scale=1.0" />
6
6
  <title>ossclip editor</title>
7
- <script type="module" crossorigin src="/assets/index-pWbFr8vc.js"></script>
7
+ <script type="module" crossorigin src="/assets/index-DpySfDtS.js"></script>
8
8
  <link rel="stylesheet" crossorigin href="/assets/index-Bx2VQLP8.css">
9
9
  </head>
10
10
  <body>
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "ossclip",
3
- "version": "0.1.36",
3
+ "version": "0.1.38",
4
4
  "description": "Local-first CLI video producer: cuts silence and fillers, word-timed captions, face-aware framing, and LLM-planned code-rendered graphics — transcription and rendering never leave your machine",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -36,9 +36,9 @@
36
36
  "commander": "^12.1.0",
37
37
  "tsx": "^4.19.0",
38
38
  "zod": "^3.25.76",
39
- "@ossclip/core": "0.1.36",
40
- "@ossclip/renderer": "0.1.36",
41
- "@ossclip/scenes": "0.1.36"
39
+ "@ossclip/core": "0.1.38",
40
+ "@ossclip/scenes": "0.1.38",
41
+ "@ossclip/renderer": "0.1.38"
42
42
  },
43
43
  "homepage": "https://github.com/AhsanAyaz/ossclip#readme",
44
44
  "bugs": {
package/src/analyze.ts CHANGED
@@ -117,6 +117,9 @@ export interface AnalyzeOptions {
117
117
  noiseDb?: number;
118
118
  whisperModel?: string;
119
119
  whisperLanguage?: string;
120
+ /** `--whisper-backend`, already zod-parsed by program.ts; undefined lets a
121
+ * configured `whisperUrl` decide (resolveWhisperBackend in produce). */
122
+ whisperBackend?: "local" | "remote";
120
123
  blooperMarker?: string;
121
124
  collapseRetakes?: boolean;
122
125
  sort?: "name" | "mtime";
@@ -157,6 +160,7 @@ export async function runAnalyze(
157
160
  noiseDb: opts.noiseDb,
158
161
  whisperModel: opts.whisperModel,
159
162
  whisperLanguage: opts.whisperLanguage,
163
+ whisperBackend: opts.whisperBackend,
160
164
  blooperMarker: opts.blooperMarker,
161
165
  collapseRetakes: opts.collapseRetakes,
162
166
  sort: opts.sort,
package/src/doctor.ts CHANGED
@@ -2,6 +2,7 @@ import { spawn } from "node:child_process";
2
2
  import { existsSync } from "node:fs";
3
3
  import type { OssclipConfig } from "@ossclip/core";
4
4
  import { modelUrl, validModelSources, whisperModelPath } from "./setup/manifest";
5
+ import { WHISPER_API_KEY_ENV, resolveWhisperBackend } from "./whisper-backend";
5
6
 
6
7
  /**
7
8
  * `ossclip doctor` (R18 §90a): check every prerequisite and print the exact
@@ -106,12 +107,23 @@ export async function runDoctor(cfg: OssclipConfig, p: DoctorProbes): Promise<Do
106
107
  }),
107
108
  });
108
109
 
110
+ // Remote transcription (2026-09-01 weak-CPU field report) makes the next
111
+ // two checks OPTIONAL rather than blocking: a machine that transcribes on
112
+ // Groq has no reason to own whisper.cpp or a 1.5 GB model, and doctor
113
+ // reporting two red lines on a working install is how a user concludes the
114
+ // tool is broken. The LLM-provider posture, one level up: pass with a
115
+ // detail that says why nothing is needed.
116
+ const remote = resolveWhisperBackend(undefined, cfg, p.env);
117
+ const remoteBackend = remote.ok && remote.backend.kind === "remote" ? remote.backend : null;
118
+ const notNeeded = (found: string): string =>
119
+ `${found} not found — not needed: remote transcription configured`;
120
+
109
121
  const whisperOk = await p.binRuns(cfg.whisperPath, "--help");
110
122
  checks.push({
111
123
  name: "whisper-cli",
112
- ok: whisperOk,
113
- detail: cfg.whisperPath,
114
- ...(whisperOk
124
+ ok: whisperOk || remoteBackend !== null,
125
+ detail: whisperOk ? cfg.whisperPath : remoteBackend !== null ? notNeeded(cfg.whisperPath) : cfg.whisperPath,
126
+ ...(whisperOk || remoteBackend !== null
115
127
  ? {}
116
128
  : {
117
129
  fix: viaSetup(
@@ -134,9 +146,9 @@ export async function runDoctor(cfg: OssclipConfig, p: DoctorProbes): Promise<Do
134
146
  const modelOk = p.exists(modelPath);
135
147
  checks.push({
136
148
  name: `whisper model (${cfg.model})`,
137
- ok: modelOk,
138
- detail: modelPath,
139
- ...(modelOk
149
+ ok: modelOk || remoteBackend !== null,
150
+ detail: modelOk ? modelPath : remoteBackend !== null ? notNeeded(modelPath) : modelPath,
151
+ ...(modelOk || remoteBackend !== null
140
152
  ? {}
141
153
  : {
142
154
  fix: viaSetup(
@@ -146,6 +158,26 @@ export async function runDoctor(cfg: OssclipConfig, p: DoctorProbes): Promise<Do
146
158
  }),
147
159
  });
148
160
 
161
+ // NO network call, unlike every other backend doctor could probe: a
162
+ // transcription request costs the user's metered free tier, and `doctor` is
163
+ // run repeatedly while fixing something else. This line reports the
164
+ // CONFIGURATION — the three things a 401 or a 404 would be about — and the
165
+ // provider's own status hints name the rest when a real run happens.
166
+ // Omitted entirely when remote is not configured: the local install is the
167
+ // default, and an extra "not configured" line for an opt-in feature is
168
+ // noise on every other machine.
169
+ if (remoteBackend !== null) {
170
+ checks.push({
171
+ name: "remote transcription",
172
+ ok: true,
173
+ detail:
174
+ `${remoteBackend.baseUrl} · model ${remoteBackend.model} · ` +
175
+ (remoteBackend.apiKey !== undefined
176
+ ? `${WHISPER_API_KEY_ENV} set`
177
+ : `no API key (fine for self-hosted; Groq needs ${WHISPER_API_KEY_ENV})`),
178
+ });
179
+ }
180
+
149
181
  // Provider, in the same order auto-detection uses (agy → claude CLI →
150
182
  // gemini key → anthropic key): subscription CLIs beat ambient env keys
151
183
  // since 2026-08 — a logged-in CLI is an explicit, already-paid choice
package/src/edit.ts CHANGED
@@ -70,6 +70,11 @@ import {
70
70
  portraitMimeType,
71
71
  readCoverProvenance,
72
72
  runWhisper,
73
+ // The remote transcription backend (2026-09-01) — the span re-decode goes
74
+ // through it whenever `whisperUrl` is configured.
75
+ createOpenAiCompatibleProvider,
76
+ type TranscribeRequest,
77
+ type Transcript,
73
78
  SegmentSchema,
74
79
  spliceTranscript,
75
80
  TranscriptSchema,
@@ -116,6 +121,7 @@ import { expandHome } from "./paths";
116
121
  // produce all resolve through — a second copy here would send a user to a
117
122
  // model file the rest of the tool never looks for.
118
123
  import { modelImpliedLanguage, whisperModelPath } from "./setup/manifest";
124
+ import { resolveWhisperBackend } from "./whisper-backend";
119
125
  import {
120
126
  PORTRAIT_OVERRIDE_BASENAME,
121
127
  portraitExtensionForMime,
@@ -198,6 +204,37 @@ export interface RetranscribeConfig {
198
204
  modelDir?: string;
199
205
  language?: unknown;
200
206
  dictionary?: unknown;
207
+ /** The remote-backend pair (2026-09-01): a configured URL means this span
208
+ * is decoded by an OpenAI-compatible server instead of whisper.cpp, so the
209
+ * server must see the same two keys `resolveWhisperBackend` reads. */
210
+ whisperUrl?: string;
211
+ whisperRemoteModel?: string;
212
+ }
213
+
214
+ /**
215
+ * The half of a range re-decode that is the same on BOTH backends: the
216
+ * decoder bias. Extracted (2026-09-01) so the remote path shares these rules
217
+ * verbatim instead of growing a second copy that drifts — the validation is
218
+ * the load-bearing part, and it is unchanged from the local-only version.
219
+ */
220
+ function retranscribeBias(cfg: RetranscribeConfig): { language?: string; prompt?: string } {
221
+ const dict = Array.isArray(cfg.dictionary)
222
+ && cfg.dictionary.length > 0
223
+ && cfg.dictionary.every((t) => typeof t === "string" && t.trim().length > 0)
224
+ ? (cfg.dictionary as string[]).map((t) => t.trim())
225
+ : [];
226
+ // `cfg.model` can be absent on the remote backend (a machine that never
227
+ // installed a local model), and no model name implies no language.
228
+ const language = typeof cfg.language === "string" && cfg.language.trim().length > 0
229
+ ? cfg.language.trim()
230
+ : cfg.model !== undefined
231
+ ? modelImpliedLanguage(cfg.model)
232
+ : undefined;
233
+ const prompt = whisperPromptFor(dict);
234
+ return {
235
+ ...(language !== undefined ? { language } : {}),
236
+ ...(prompt !== undefined ? { prompt } : {}),
237
+ };
201
238
  }
202
239
 
203
240
  /**
@@ -238,21 +275,38 @@ export function retranscribeSettings(
238
275
  "then `ossclip setup` to install it.",
239
276
  };
240
277
  }
241
- const dict = Array.isArray(cfg.dictionary)
242
- && cfg.dictionary.length > 0
243
- && cfg.dictionary.every((t) => typeof t === "string" && t.trim().length > 0)
244
- ? (cfg.dictionary as string[]).map((t) => t.trim())
245
- : [];
246
- const language = typeof cfg.language === "string" && cfg.language.trim().length > 0
247
- ? cfg.language.trim()
248
- : modelImpliedLanguage(cfg.model);
249
- const prompt = whisperPromptFor(dict);
250
278
  return {
251
279
  tools: { ffmpegPath: cfg.ffmpegPath, ffprobePath: cfg.ffprobePath },
252
280
  whisperPath: cfg.whisperPath,
253
281
  modelPath: whisperModelPath(cfg.model, cfg.modelDir),
254
- ...(language !== undefined ? { language } : {}),
255
- ...(prompt !== undefined ? { prompt } : {}),
282
+ ...retranscribeBias(cfg),
283
+ };
284
+ }
285
+
286
+ /**
287
+ * The same thing for the REMOTE backend (2026-09-01): no whisper binary and
288
+ * no model file, because the point of remote is that neither is installed —
289
+ * but ffmpeg still is, since the span is sliced out of `audio.wav` locally
290
+ * before it is uploaded.
291
+ *
292
+ * Pure, like its local twin: the whole "remote configured but ffmpeg isn't"
293
+ * corner is testable without a network.
294
+ */
295
+ export function retranscribeRemoteSettings(
296
+ cfg: RetranscribeConfig,
297
+ ):
298
+ | { tools: { ffmpegPath: string; ffprobePath: string }; language?: string; prompt?: string }
299
+ | { error: string } {
300
+ if (!cfg.ffmpegPath || !cfg.ffprobePath) {
301
+ return {
302
+ error:
303
+ "ffmpeg is not configured — run `ossclip doctor` to see what is missing, " +
304
+ "then `ossclip setup` to install it. (Remote transcription still slices the span locally.)",
305
+ };
306
+ }
307
+ return {
308
+ tools: { ffmpegPath: cfg.ffmpegPath, ffprobePath: cfg.ffprobePath },
309
+ ...retranscribeBias(cfg),
256
310
  };
257
311
  }
258
312
 
@@ -484,6 +538,13 @@ export async function startEditServer(
484
538
  */
485
539
  sliceAudio?: typeof extractAudioSpan;
486
540
  runWhisper?: typeof runWhisper;
541
+ /**
542
+ * The remote backend's half of the `runWhisper` seam (2026-09-01): with a
543
+ * `whisperUrl` configured the span goes to an OpenAI-compatible server
544
+ * instead of whisper.cpp, and a test must be able to observe that —
545
+ * including the failure sentence — without a network or an API key.
546
+ */
547
+ transcribeRemote?: (wavPath: string, req: TranscribeRequest) => Promise<Transcript>;
487
548
  /**
488
549
  * The sound library the SFX routes serve (`loadCfg`'s rule applied to the
489
550
  * pack loader): tests inject a hand-written library over a tmp dir, so the
@@ -955,17 +1016,67 @@ export async function startEditServer(
955
1016
  error: "this workdir has no transcript.json to re-stamp — re-run `ossclip produce`.",
956
1017
  });
957
1018
  }
958
- const settings = retranscribeSettings((opts.loadCfg ?? loadConfig)());
959
- if ("error" in settings) return send(200, { ok: false, error: settings.error });
960
- if (!existsSync(settings.modelPath)) {
961
- // The `--transcript`-only install: whisper was never needed to
962
- // make this project, so say what to run rather than 500ing.
963
- return send(200, {
964
- ok: false,
965
- error:
966
- `whisper model not found at ${settings.modelPath} — run \`ossclip setup\` ` +
967
- `to download it.`,
1019
+ const cfg = (opts.loadCfg ?? loadConfig)();
1020
+ // Which engine re-decodes the span, resolved HERE rather than in
1021
+ // retranscribeSettings (2026-09-01): the flag is a CLI thing and
1022
+ // there is no CLI in this loop, so a configured `whisperUrl` is
1023
+ // the whole switch. `undefined` as the flag can only answer ok —
1024
+ // only an explicit `--whisper-backend remote` with nothing
1025
+ // configured fails — so this is a narrowing, not a live branch.
1026
+ const backendPick = resolveWhisperBackend(undefined, cfg, process.env);
1027
+ if (!backendPick.ok) return send(200, { ok: false, error: backendPick.message });
1028
+ const backend = backendPick.backend;
1029
+ // One plan, two shapes: the local one needs a binary and a model
1030
+ // file on disk, the remote one needs neither. Both need ffmpeg —
1031
+ // the span is always sliced here.
1032
+ let plan: {
1033
+ tools: { ffmpegPath: string; ffprobePath: string };
1034
+ language?: string;
1035
+ prompt?: string;
1036
+ decode: (wavPath: string, req: TranscribeRequest) => Promise<Transcript>;
1037
+ };
1038
+ if (backend.kind === "remote") {
1039
+ const settings = retranscribeRemoteSettings(cfg);
1040
+ if ("error" in settings) return send(200, { ok: false, error: settings.error });
1041
+ const provider = createOpenAiCompatibleProvider({
1042
+ baseUrl: backend.baseUrl,
1043
+ model: backend.model,
1044
+ ...(backend.apiKey !== undefined ? { apiKey: backend.apiKey } : {}),
968
1045
  });
1046
+ plan = {
1047
+ ...settings,
1048
+ // Uploaded AS-IS, no opus sidecar (produce's rule does not
1049
+ // apply): a span is seconds long, so its wav is far under any
1050
+ // upload cap and the encode would cost more than it saves.
1051
+ decode: opts.transcribeRemote ?? ((wavPath, r) => provider.transcribe(wavPath, r)),
1052
+ };
1053
+ } else {
1054
+ const settings = retranscribeSettings(cfg);
1055
+ if ("error" in settings) return send(200, { ok: false, error: settings.error });
1056
+ if (!existsSync(settings.modelPath)) {
1057
+ // The `--transcript`-only install: whisper was never needed to
1058
+ // make this project, so say what to run rather than 500ing.
1059
+ return send(200, {
1060
+ ok: false,
1061
+ error:
1062
+ `whisper model not found at ${settings.modelPath} — run \`ossclip setup\` ` +
1063
+ `to download it.`,
1064
+ });
1065
+ }
1066
+ plan = {
1067
+ ...settings,
1068
+ decode: (wavPath, r) =>
1069
+ (opts.runWhisper ?? runWhisper)(
1070
+ {
1071
+ whisperPath: settings.whisperPath,
1072
+ modelPath: settings.modelPath,
1073
+ outBase,
1074
+ ...(r.language !== undefined ? { language: r.language } : {}),
1075
+ ...(r.prompt !== undefined ? { prompt: r.prompt } : {}),
1076
+ },
1077
+ wavPath,
1078
+ ),
1079
+ };
969
1080
  }
970
1081
  // Parsed, not cast: this file is about to be rewritten, and a
971
1082
  // truncated one must fail loudly here rather than become the new
@@ -985,22 +1096,16 @@ export async function startEditServer(
985
1096
  });
986
1097
  }
987
1098
  await (opts.sliceAudio ?? extractAudioSpan)(
988
- settings.tools,
1099
+ plan.tools,
989
1100
  audio,
990
1101
  tmpWav,
991
1102
  srcIn,
992
1103
  srcOut - srcIn,
993
1104
  );
994
- const fresh = await (opts.runWhisper ?? runWhisper)(
995
- {
996
- whisperPath: settings.whisperPath,
997
- modelPath: settings.modelPath,
998
- outBase,
999
- ...(settings.language !== undefined ? { language: settings.language } : {}),
1000
- ...(settings.prompt !== undefined ? { prompt: settings.prompt } : {}),
1001
- },
1002
- tmpWav,
1003
- );
1105
+ const fresh = await plan.decode(tmpWav, {
1106
+ ...(plan.language !== undefined ? { language: plan.language } : {}),
1107
+ ...(plan.prompt !== undefined ? { prompt: plan.prompt } : {}),
1108
+ });
1004
1109
  const restamped = alignRestamp(
1005
1110
  transcript.words.slice(range.from, range.to),
1006
1111
  fresh.words,
@@ -10,7 +10,7 @@ import { produceArgv, type ProduceAnswers, type ProduceExtras } from "./produce-
10
10
  import { assertInteractive, confirm, intro, multiselect, select, text, unwrap } from "./prompts";
11
11
 
12
12
  /**
13
- * The produce wizard. Forty-three flags (plus the positional input path)
13
+ * The produce wizard. Forty-four flags (plus the positional input path)
14
14
  * sorted into three tiers: six prompts asked directly — the input path, plus
15
15
  * five flags (--out, --cleanup, --aspect, --produce, --intent) — twelve
16
16
  * behind one "anything else?" multiselect (--sfx being the twelfth, with
@@ -28,7 +28,12 @@ import { assertInteractive, confirm, intro, multiselect, select, text, unwrap }
28
28
  * multiselect entry is the OFF switch and the positive flag exists only for
29
29
  * replay pinning), --add-jump-cuts (same mirror: auto already punches, the
30
30
  * multiselect entry is the OFF switch, and the force flag exists to beat a
31
- * future config-off), or
31
+ * future config-off),
32
+ * --whisper-backend (2026-09-01, the --color-grade shape: it selects machine
33
+ * INFRASTRUCTURE — which transcription engine this box owns, alongside
34
+ * whisperPath and modelDir — not a per-run editorial choice, and the durable
35
+ * spelling is `whisperUrl` in ~/.ossclip/config.json; an honest prompt would
36
+ * also have to explain a base URL and an API key at a menu), or
32
37
  * (final-review fix wave, Finding 1) --sort. A folder's clip order only means anything once the
33
38
  * folder has been enumerated, and that enumeration happens inside
34
39
  * `produce()` — after the wizard has already returned argv — so there is
package/src/produce.ts CHANGED
@@ -66,6 +66,10 @@ import {
66
66
  splitThenDropHidden,
67
67
  emptyOverrideDoc,
68
68
  extractAudio,
69
+ encodeUploadAudio,
70
+ REMOTE_UPLOAD_MAX_BYTES,
71
+ createOpenAiCompatibleProvider,
72
+ openaiTranscriptionsUrl,
69
73
  fillPlainCues,
70
74
  splitCues,
71
75
  landscapeLayout,
@@ -249,6 +253,7 @@ import { RenderTimelineHUD, StageAnimator, printProductionCompleteBanner } from
249
253
  import { reconcileCaptionEdits } from "./caption-report";
250
254
  import { overridesWriteLine, writeOverrideDoc } from "./overrides-write";
251
255
  import { recordedProduceArgs } from "./replay-argv";
256
+ import { remoteWhisperHost, resolveWhisperBackend } from "./whisper-backend";
252
257
  import { makeCancelSignal, renderCover, renderProduction } from "@ossclip/renderer";
253
258
  import type { RenderPhase } from "@ossclip/renderer";
254
259
  import {
@@ -311,6 +316,17 @@ export const TranscriptKeySchema = z.object({
311
316
  * files) means "no translation", the `dictionary` contract.
312
317
  */
313
318
  translate: z.boolean().optional(),
319
+ /**
320
+ * Which BACKEND decoded it (2026-09-01): `remote:<normalized endpoint>`, or
321
+ * absent for local whisper.cpp — the `dictionary`/`translate` contract, so
322
+ * every key file written before remote existed still reads as local. Two
323
+ * engines on the same audio produce different words, so without this a warm
324
+ * workdir serves the local transcript to a remote run (and vice versa) —
325
+ * the exact staleness `language` and `translate` were added for. The
326
+ * remote MODEL name rides in `model` above, so switching Groq models
327
+ * re-keys through the existing field.
328
+ */
329
+ backend: z.string().optional(),
314
330
  });
315
331
  export type TranscriptKey = z.infer<typeof TranscriptKeySchema>;
316
332
 
@@ -338,6 +354,10 @@ export function transcriptCacheReusable(
338
354
  // Absent and false are the same "no translation", so pre-flag key
339
355
  // files reuse under a non-translate request.
340
356
  (effective.translate ?? false) === (requested.translate ?? false) &&
357
+ // Absent means LOCAL on both sides, so every pre-2026-09-01 key file
358
+ // still reuses under a local request — and a remote request against
359
+ // one of them re-transcribes, which is the point.
360
+ (effective.backend ?? "") === (requested.backend ?? "") &&
341
361
  // ORDER-SENSITIVE by choice: the dictionary becomes whisper's --prompt
342
362
  // text verbatim, so a reordered list genuinely is a different decoder
343
363
  // input — treating it as equal would serve a transcript biased by a
@@ -634,6 +654,13 @@ export interface ProduceOptions {
634
654
  * together (whisper decodes better knowing the source language).
635
655
  */
636
656
  whisperTranslate?: boolean;
657
+ /**
658
+ * `--whisper-backend`, already zod-parsed to the union by program.ts
659
+ * (2026-09-01 weak-CPU field report). Undefined means "not typed", which
660
+ * is what lets a configured `whisperUrl` select remote — the flag is
661
+ * mainly `local`, the per-run opt-out.
662
+ */
663
+ whisperBackend?: "local" | "remote";
637
664
  /**
638
665
  * Vocabulary terms for this run (`--dictionary`, F4 2026-08-16), already
639
666
  * split/trimmed by the action. Wholesale beats the config's `dictionary`
@@ -2664,9 +2691,34 @@ export async function produce(inputArg: string, opts: ProduceOptions): Promise<P
2664
2691
  `--whisper-language overrides)`,
2665
2692
  );
2666
2693
  }
2694
+ // Local whisper.cpp or an OpenAI-compatible server (2026-09-01 weak-CPU
2695
+ // field report). Resolved BEFORE the key, like the language, so what
2696
+ // actually decodes is what the cache is keyed on.
2697
+ const backendPick = resolveWhisperBackend(opts.whisperBackend, cfg, process.env);
2698
+ if (!backendPick.ok) throw new Error(backendPick.message);
2699
+ const backend = backendPick.backend;
2700
+ // BEFORE the key is built, so a translate request can never cross the
2701
+ // cache with a remote one: the OpenAI-compatible API translates on a
2702
+ // DIFFERENT endpoint with a DIFFERENT default model, and swapping both
2703
+ // behind one flag would be a surprise rather than a convenience.
2704
+ if (backend.kind === "remote" && opts.whisperTranslate === true) {
2705
+ throw new Error(
2706
+ "--whisper-translate needs the local backend (the OpenAI-compatible API translates on a " +
2707
+ "different endpoint and model) — use --whisper-backend local, or drop the flag.",
2708
+ );
2709
+ }
2667
2710
  const requestedKey: TranscriptKey = {
2668
- model: requestedModel,
2711
+ // The REMOTE model name when remote — one field, both engines, so an
2712
+ // A/B between two Groq models re-keys the cache exactly like a local one.
2713
+ model: backend.kind === "remote" ? backend.model : requestedModel,
2669
2714
  ...(whisperLang.language !== undefined ? { language: whisperLang.language } : {}),
2715
+ // Spread-omitted on local so local key files stay byte-identical to
2716
+ // every one written before remote existed (the translate posture). The
2717
+ // URL goes through openaiTranscriptionsUrl so ".../v1" and ".../v1/"
2718
+ // key identically — a trailing slash is not a different server.
2719
+ ...(backend.kind === "remote"
2720
+ ? { backend: `remote:${openaiTranscriptionsUrl(backend.baseUrl)}` }
2721
+ : {}),
2670
2722
  // Omitted when off, so a non-translate run's key stays byte-identical to
2671
2723
  // every pre-flag key file (the dictionary posture).
2672
2724
  ...(opts.whisperTranslate === true ? { translate: true } : {}),
@@ -2704,51 +2756,100 @@ export async function produce(inputArg: string, opts: ProduceOptions): Promise<P
2704
2756
  `re-transcribing with ${fmt(requestedKey)}`,
2705
2757
  );
2706
2758
  }
2707
- await preflight(
2708
- cfg.whisperPath,
2709
- "Run `ossclip setup`, install whisper.cpp yourself (https://github.com/ggml-org/whisper.cpp), or set OSSCLIP_WHISPER.",
2710
- );
2711
- const model = requestedKey.model;
2712
- // whisperModelPath/modelUrl are THE resolution and URL sources (shared
2713
- // with doctor and setup) — this error used to hold its own copy of the
2714
- // ggerganov URL, which 404'd for curated/custom names and the suggested
2715
- // `curl -L` then saved the 404 HTML as a fake model.
2716
- const modelPath = whisperModelPath(model, cfg.modelDir);
2717
- if (!existsSync(modelPath)) {
2718
- throw new Error(
2719
- `whisper model not found at ${modelPath}.\n` +
2720
- `Run \`ossclip setup${model === cfg.model ? "" : ` --model ${model}`}\` to download it — or manually:\n` +
2721
- ` curl -L -o ${modelPath} ${modelUrl(model, validModelSources(cfg.modelSources))}`,
2759
+ if (backend.kind === "remote") {
2760
+ // No whisper binary and no model file on this branch — the whole point
2761
+ // of remote is that neither is installed (2026-09-01 field report).
2762
+ // ffmpeg still is: the upload sidecar is an encode.
2763
+ const host = remoteWhisperHost(backend.baseUrl);
2764
+ const uploadPath = join(work, "audio-upload.ogg");
2765
+ transcript = await phases.time("transcribe", async () => {
2766
+ await encodeUploadAudio(tools, audioPath, uploadPath);
2767
+ const bytes = statSync(uploadPath).size;
2768
+ if (bytes > REMOTE_UPLOAD_MAX_BYTES) {
2769
+ // Named here rather than paid for as somebody else's 413 after the
2770
+ // whole upload: chunking is out of scope for v1, so the error has
2771
+ // to carry both escape hatches itself.
2772
+ throw new Error(
2773
+ `the compressed audio is ${(bytes / 1_000_000).toFixed(1)} MB, over the ` +
2774
+ `${(REMOTE_UPLOAD_MAX_BYTES / 1_000_000).toFixed(0)} MB single-file limit for remote ` +
2775
+ `transcription (about 100 minutes of speech at this bitrate).\n` +
2776
+ `Transcribe locally with --whisper-backend local, split the take, or point ` +
2777
+ `OSSCLIP_WHISPER_URL at a server with a larger cap (Groq's dev tier allows 100 MB).`,
2778
+ );
2779
+ }
2780
+ const anim = isInteractive()
2781
+ ? new StageAnimator(
2782
+ "REMOTE ASR",
2783
+ `Transcribing via ${host} (${backend.model})...`,
2784
+ "whisper",
2785
+ ).start()
2786
+ : null;
2787
+ if (!anim) console.log(`▸ transcribing remotely (${host}, ${backend.model})…`);
2788
+ try {
2789
+ return await createOpenAiCompatibleProvider({
2790
+ baseUrl: backend.baseUrl,
2791
+ model: backend.model,
2792
+ ...(backend.apiKey !== undefined ? { apiKey: backend.apiKey } : {}),
2793
+ }).transcribe(uploadPath, {
2794
+ // From the KEY, like the local branch: whatever re-keys the cache
2795
+ // is what actually decoded, so the two can never disagree.
2796
+ language: requestedKey.language,
2797
+ prompt: whisperPromptFor(dictionary),
2798
+ });
2799
+ } finally {
2800
+ // In a finally, unlike the local branch's trailing stop(): an HTTP
2801
+ // failure here is EXPECTED (a wrong key, a rate limit), and a
2802
+ // spinner still animating would overwrite the hint the user needs.
2803
+ anim?.stop();
2804
+ }
2805
+ });
2806
+ } else {
2807
+ await preflight(
2808
+ cfg.whisperPath,
2809
+ "Run `ossclip setup`, install whisper.cpp yourself (https://github.com/ggml-org/whisper.cpp), or set OSSCLIP_WHISPER.",
2810
+ );
2811
+ const model = requestedKey.model;
2812
+ // whisperModelPath/modelUrl are THE resolution and URL sources (shared
2813
+ // with doctor and setup) — this error used to hold its own copy of the
2814
+ // ggerganov URL, which 404'd for curated/custom names and the suggested
2815
+ // `curl -L` then saved the 404 HTML as a fake model.
2816
+ const modelPath = whisperModelPath(model, cfg.modelDir);
2817
+ if (!existsSync(modelPath)) {
2818
+ throw new Error(
2819
+ `whisper model not found at ${modelPath}.\n` +
2820
+ `Run \`ossclip setup${model === cfg.model ? "" : ` --model ${model}`}\` to download it — or manually:\n` +
2821
+ ` curl -L -o ${modelPath} ${modelUrl(model, validModelSources(cfg.modelSources))}`,
2822
+ );
2823
+ }
2824
+ const whisperAnim = isInteractive()
2825
+ ? new StageAnimator(
2826
+ "WHISPER ASR",
2827
+ `Transcribing audio stream with ${basename(modelPath)}...`,
2828
+ "whisper",
2829
+ ).start()
2830
+ : null;
2831
+ if (!whisperAnim) console.log(`▸ transcribing (${basename(modelPath)})…`);
2832
+ transcript = await phases.time("transcribe", () =>
2833
+ runWhisper(
2834
+ {
2835
+ whisperPath: cfg.whisperPath,
2836
+ modelPath,
2837
+ outBase: join(work, "whisper"),
2838
+ // The RESOLVED language, not the raw flag — a config/model-implied
2839
+ // code must reach the spawn exactly as it reached the cache key.
2840
+ language: requestedKey.language,
2841
+ // Vocabulary biasing (F4) — undefined for an empty dictionary, so
2842
+ // the spawned args stay byte-identical to every pre-dictionary run.
2843
+ // From the KEY, like the language: whatever re-keys the cache is
2844
+ // what actually ran, so the two can never disagree.
2845
+ ...(requestedKey.translate === true ? { translate: true } : {}),
2846
+ prompt: whisperPromptFor(dictionary),
2847
+ },
2848
+ audioPath,
2849
+ ),
2722
2850
  );
2851
+ if (whisperAnim) whisperAnim.stop();
2723
2852
  }
2724
- const whisperAnim = isInteractive()
2725
- ? new StageAnimator(
2726
- "WHISPER ASR",
2727
- `Transcribing audio stream with ${basename(modelPath)}...`,
2728
- "whisper",
2729
- ).start()
2730
- : null;
2731
- if (!whisperAnim) console.log(`▸ transcribing (${basename(modelPath)})…`);
2732
- transcript = await phases.time("transcribe", () =>
2733
- runWhisper(
2734
- {
2735
- whisperPath: cfg.whisperPath,
2736
- modelPath,
2737
- outBase: join(work, "whisper"),
2738
- // The RESOLVED language, not the raw flag — a config/model-implied
2739
- // code must reach the spawn exactly as it reached the cache key.
2740
- language: requestedKey.language,
2741
- // Vocabulary biasing (F4) — undefined for an empty dictionary, so
2742
- // the spawned args stay byte-identical to every pre-dictionary run.
2743
- // From the KEY, like the language: whatever re-keys the cache is
2744
- // what actually ran, so the two can never disagree.
2745
- ...(requestedKey.translate === true ? { translate: true } : {}),
2746
- prompt: whisperPromptFor(dictionary),
2747
- },
2748
- audioPath,
2749
- ),
2750
- );
2751
- if (whisperAnim) whisperAnim.stop();
2752
2853
  console.log(`▸ transcribed ${transcript.words.length} words`);
2753
2854
  await writeFile(transcriptKeyPath, JSON.stringify(requestedKey, null, 2));
2754
2855
  }