@gentbajko/slopify 0.6.1 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/README.md +36 -1
  2. package/SUBTITLES.md +3 -3
  3. package/dist/adapter-registry.js +12 -2
  4. package/dist/adapters/image/google.js +12 -15
  5. package/dist/adapters/image/models.js +72 -0
  6. package/dist/adapters/image/openai.js +3 -5
  7. package/dist/adapters/llm/catalogue-files.js +23 -0
  8. package/dist/adapters/llm/claude-code.js +3 -4
  9. package/dist/adapters/llm/codex-models.js +46 -0
  10. package/dist/adapters/llm/codex.js +7 -10
  11. package/dist/adapters/llm/gemini-models.js +85 -0
  12. package/dist/adapters/llm/gemini-workspace.js +24 -1
  13. package/dist/adapters/llm/gemini.js +4 -7
  14. package/dist/adapters/llm/openrouter.js +12 -2
  15. package/dist/adapters/tts/cartesia.js +9 -1
  16. package/dist/adapters/tts/elevenlabs.js +33 -2
  17. package/dist/adapters/tts/inworld-async.js +150 -0
  18. package/dist/adapters/tts/inworld-text.js +32 -0
  19. package/dist/adapters/tts/inworld.js +170 -0
  20. package/dist/adapters/tts/openai.js +31 -4
  21. package/dist/assets/models.yaml +470 -0
  22. package/dist/catalog/registry.js +147 -0
  23. package/dist/catalog/schema.js +91 -0
  24. package/dist/catalog/store.js +90 -0
  25. package/dist/catalog/validate.js +43 -0
  26. package/dist/edge/cli.js +5 -0
  27. package/dist/edge/events/hub.js +9 -2
  28. package/dist/edge/events/preview-cache.js +40 -0
  29. package/dist/edge/http/actions.js +2 -0
  30. package/dist/edge/http/app.js +28 -1
  31. package/dist/edge/http/audio-preview.js +46 -0
  32. package/dist/edge/http/planning.js +101 -0
  33. package/dist/edge/http/projects.js +4 -2
  34. package/dist/edge/http/providers.js +44 -0
  35. package/dist/edge/http/update.js +57 -0
  36. package/dist/edge/update-worker.js +44 -0
  37. package/dist/kernel/audio-preview.js +188 -0
  38. package/dist/kernel/db/migrations/0003-batch-queue.sql +11 -0
  39. package/dist/kernel/ports/llm.js +1 -0
  40. package/dist/kernel/ports/text.js +26 -0
  41. package/dist/kernel/runner/providers.js +80 -28
  42. package/dist/kernel/runner/queue.js +65 -0
  43. package/dist/main.js +142 -27
  44. package/dist/model-catalog.js +37 -0
  45. package/dist/slices/admission/repo.js +6 -1
  46. package/dist/slices/admission/start.js +9 -8
  47. package/dist/slices/article/continuation.js +8 -1
  48. package/dist/slices/article/segments.js +1 -0
  49. package/dist/slices/batch/index.js +82 -0
  50. package/dist/slices/control/index.js +6 -1
  51. package/dist/slices/control/providers.js +16 -1
  52. package/dist/slices/estimate/index.js +81 -0
  53. package/dist/slices/narration/live.js +21 -0
  54. package/dist/slices/narration/plan.js +58 -0
  55. package/dist/slices/narration/run.js +15 -36
  56. package/dist/slices/research/run.js +7 -0
  57. package/dist/slices/settings/model.js +2 -0
  58. package/dist/slices/settings/models.js +89 -0
  59. package/dist/slices/storage/staging.js +2 -1
  60. package/dist/slices/subtitles/captions.js +4 -1
  61. package/dist/slices/subtitles/layout.js +19 -0
  62. package/dist/slices/subtitles/model.js +9 -0
  63. package/dist/slices/subtitles/prepare.js +6 -1
  64. package/dist/slices/thumbnail/run.js +3 -0
  65. package/dist/updater/candidate.js +28 -0
  66. package/dist/updater/forward.js +35 -0
  67. package/dist/updater/install-flow.js +26 -0
  68. package/dist/updater/install.js +50 -0
  69. package/dist/updater/model.js +24 -0
  70. package/dist/updater/plan.js +135 -0
  71. package/dist/updater/readiness.js +15 -0
  72. package/dist/updater/registry.js +19 -0
  73. package/dist/updater/service.js +128 -0
  74. package/dist/updater/worker.js +136 -0
  75. package/dist/web/assets/index-6zz8telY.css +1 -0
  76. package/dist/web/assets/index-CbYEcBOa.js +130 -0
  77. package/dist/web/index.html +2 -2
  78. package/package.json +2 -1
  79. package/dist/web/assets/index-CNYs0noB.js +0 -130
  80. package/dist/web/assets/index-Mk0bvBg-.css +0 -1
package/README.md CHANGED
@@ -1,4 +1,15 @@
1
- # Slopify
1
+ <h1 align="center">
2
+ <a href="https://slopify.stream"><img src="https://slopify.stream/assets/favicon.svg" width="40" height="40" align="middle" alt="" /></a>
3
+ Slopify
4
+ </h1>
5
+
6
+ <p align="center">
7
+ <a href="https://slopify.stream">slopify.stream</a>
8
+ &nbsp;·&nbsp;
9
+ <a href="https://www.patreon.com/cw/GentBajko"><img src="https://slopify.stream/assets/patreon-green.svg" width="14" height="14" align="middle" alt="" /> Patreon</a>
10
+ &nbsp;·&nbsp;
11
+ <a href="https://buymeacoffee.com/gentbajko"><img src="https://slopify.stream/assets/buymeacoffee-green.svg" width="14" height="14" align="middle" alt="" /> Buy Me a Coffee</a>
12
+ </p>
2
13
 
3
14
  A prompt and a few keywords in. A narrated slideshow video out. Your keys, your machine, free.
4
15
 
@@ -14,6 +25,23 @@ Six stages run as a graph: research, article, narration, images, thumbnail, vide
14
25
  Generate any of them, or provide the output yourself and that stage is skipped. The
15
26
  video is a slideshow with alternating zoom over the narration, rendered with ffmpeg.
16
27
 
28
+ ## Inworld narration
29
+
30
+ In Settings, add the **Base64 credentials** from Inworld's API Keys page, then add an
31
+ Inworld voice ID (for example `Dennis`, or a voice from your workspace). In Play,
32
+ choose **Realtime TTS-2** or **Realtime TTS-2 Flash** and that voice.
33
+
34
+ Short text streams immediately. TTS-2 text over 4,000 characters uses one async job,
35
+ up to 100,000 characters; Inworld caps On-Demand accounts at 10,000. Audio becomes
36
+ available once that job finishes. Flash uses streamed parts of at most 4,000 characters.
37
+ For longer articles, select paragraph chunking. Successful status checks keep long jobs
38
+ alive, and automatic polling/download retries reuse the accepted job. Pausing stops
39
+ local requests; Inworld may still finish and bill an accepted job. Resuming after a
40
+ pause or app restart starts a new request for unfinished narration.
41
+
42
+ See [Inworld's async API](https://docs.inworld.ai/api-reference/ttsAPI/texttospeech/synthesize-speech-async)
43
+ for account limits. Both model IDs are bundled; Inworld's LLM catalogue does not list TTS models.
44
+
17
45
  ## Options
18
46
 
19
47
  | Flag | Environment variable | Default |
@@ -67,6 +95,13 @@ Never your keys, prompts, keywords, titles, article text, filenames, or anything
67
95
  your machine. A notice says all of this the first time you run it, before the machine
68
96
  id exists, and the Usage screen shows you your own numbers at any time.
69
97
 
98
+ ## Supporting the project
99
+
100
+ Slopify is free and always will be. If it is worth something to you, there is
101
+ [Patreon](https://www.patreon.com/cw/GentBajko) and
102
+ [Buy Me a Coffee](https://buymeacoffee.com/gentbajko). The people who do are listed
103
+ in [SUPPORTERS.md](https://github.com/GentBajko/slopify/blob/main/SUPPORTERS.md).
104
+
70
105
  ## Licence
71
106
 
72
107
  MIT. The ffmpeg binary fetched at install time is a separate GPL-3.0-or-later program,
package/SUBTITLES.md CHANGED
@@ -1,12 +1,12 @@
1
- # Subtitles in Slopify 0.6
1
+ # Subtitles in Slopify
2
2
 
3
3
  1. In Play, enable narration or provide matching English audio and article text.
4
4
  2. In Subtitles, choose **Subtitle files (.srt + .vtt)** or **Burn into video + files**.
5
- 3. For burned captions, select a bundled or system font, or upload a `.ttf` or `.otf` file (up to 32 MiB). Preview the font and choose a size from 16 to 120.
5
+ 3. For burned captions, select a bundled or system font, or upload a `.ttf` or `.otf` file (up to 32 MiB). Choose a size from 16 to 120 and a position: top, upper-middle, center, lower-middle or bottom. The side preview shows the selected font, size and position in a 16:9 or 9:16 frame.
6
6
  4. Start the run. Slopify times the article against the actual narration on your computer. The first use downloads an approximately 95 MB English speech model; later runs work offline with the cached model. Allow roughly 1 GB of available memory during alignment. No extra API key, Python, or compiler is needed.
7
7
  5. Download SRT/VTT beside the final export. Files mode adds an optional native caption track to the in-app video preview. Burned captions remain visible in the downloaded MP4.
8
8
 
9
- For a completed project, open its final Video or Audio export section, set subtitles, and click **Save subtitles**. This rebuilds only the local export from saved narration and images. Changing font or size reuses word timing when the audio and spoken text are unchanged. A paused run saves these choices until Resume; pause an active run before editing. Failed alignment or rendering keeps the previous finished export.
9
+ For a completed project, open its final Video or Audio export section, set subtitles, and click **Save subtitles**. This rebuilds only the local export from saved narration and images. Changing font, size or position reuses word timing when the audio and spoken text are unchanged. A paused run saves these choices until Resume; pause an active run before editing. Failed alignment or rendering keeps the previous finished export.
10
10
 
11
11
  Audio Off disables subtitles. Video Off produces WAV audio with separate subtitle files. Uploaded audio must match the article; substantial mismatches fail with a message to correct the transcript. Review timing and spelling before publishing. English is supported first; unusual pronunciations and non-English passages can fail alignment.
12
12
 
@@ -4,10 +4,13 @@ import { openAiImage } from "./adapters/image/openai.js";
4
4
  import { replicateImage } from "./adapters/image/replicate.js";
5
5
  import { claudeCodeLlm } from "./adapters/llm/claude-code.js";
6
6
  import { codexLlm } from "./adapters/llm/codex.js";
7
+ import { nodeCodexModels } from "./adapters/llm/codex-models.js";
7
8
  import { geminiLlm } from "./adapters/llm/gemini.js";
9
+ import { nodeGeminiModels } from "./adapters/llm/gemini-models.js";
8
10
  import { openRouterLlm } from "./adapters/llm/openrouter.js";
9
11
  import { cartesiaTts } from "./adapters/tts/cartesia.js";
10
12
  import { elevenLabsTts } from "./adapters/tts/elevenlabs.js";
13
+ import { inworldTts } from "./adapters/tts/inworld.js";
11
14
  import { openAiTts } from "./adapters/tts/openai.js";
12
15
  import { cliBinary } from "./slices/settings/cli-paths.js";
13
16
  import { keyForAttempt } from "./slices/settings/keys.js";
@@ -26,8 +29,14 @@ export function buildRegistry(deps) {
26
29
  ["openrouter", openRouterLlm({ fetch: deps.fetch, key: keyOf("openrouter") })],
27
30
  // Each CLI authenticates with its own login.
28
31
  ["claude-code", claudeCodeLlm({ run: cliFor("claude-code") })],
29
- ["codex", codexLlm({ run: cliFor("codex") })],
30
- ["gemini", geminiLlm({ run: cliFor("gemini") })],
32
+ ["codex", codexLlm({ run: cliFor("codex"), readModels: () => nodeCodexModels() })],
33
+ [
34
+ "gemini",
35
+ geminiLlm({
36
+ run: cliFor("gemini"),
37
+ readModels: () => nodeGeminiModels(cliBinary(deps.db, "gemini")),
38
+ }),
39
+ ],
31
40
  ]);
32
41
  // One key per provider, so each adapter is handed the reader for its own row and no other.
33
42
  // OpenAI keeps two rows because it ships an adapter in two families and a user may key one
@@ -36,6 +45,7 @@ export function buildRegistry(deps) {
36
45
  ["elevenlabs", elevenLabsTts({ fetch: deps.fetch, key: keyOf("elevenlabs") })],
37
46
  ["openai-tts", openAiTts({ fetch: deps.fetch, key: keyOf("openai-tts") })],
38
47
  ["cartesia", cartesiaTts({ fetch: deps.fetch, key: keyOf("cartesia") })],
48
+ ["inworld", inworldTts({ fetch: deps.fetch, key: keyOf("inworld"), clock: deps.clock })],
39
49
  ]);
40
50
  // Four image providers behind one port, each handed the reader for its own key row.
41
51
  // Replicate also takes the clock: `Prefer: wait` gives up after 60 s and the prediction has
@@ -3,28 +3,21 @@ import { redact } from "../../kernel/log.js";
3
3
  import { providerError } from "../../kernel/ports/model.js";
4
4
  import { retryAfter } from "../retry-after.js";
5
5
  import { describeBytes, sniffImage } from "./bytes.js";
6
+ import { discoverGoogleImages } from "./models.js";
6
7
  // The HTTP gateway adapter for Google's own image generation, billed to a Gemini API key
7
8
  // rather than to a host reselling the same models. Like the OpenAI one it hands back the
8
- // bytes with the call, so there is no link to follow; unlike it, the whole interaction is
9
- // one `input` string and a `response_format`, with no per-model size table to keep.
9
+ // bytes with the call, so there is no link to follow.
10
10
  export const googleImagesBase = "https://generativelanguage.googleapis.com/v1beta";
11
- // The dropdown is filled from what the provider offers, and Google's model list carries every
12
- // text and embedding model too, so the image shortlist is this adapter's own data. Named as
13
- // Google markets them: "Nano Banana" is the family, `gemini-*-image` is what the API answers
14
- // to, and the picker shows the name a user would recognise. Newest first.
15
- //
16
- // A Gemini model without the `-image` suffix returns text and cannot be used here, whatever
17
- // its version number: `gemini-3.8-flash` is newer than all of these and generates no images.
11
+ // Offline choices only; the picker normally loads the provider catalogue.
18
12
  export const googleImageModels = [
19
13
  { id: "gemini-3.1-flash-image", name: "Nano Banana 2" },
20
14
  { id: "gemini-3.1-flash-lite-image", name: "Nano Banana 2 Lite" },
21
15
  { id: "gemini-3-pro-image", name: "Nano Banana Pro" },
22
16
  { id: "gemini-2.5-flash-image", name: "Nano Banana" },
23
17
  ];
24
- // The API takes the aspect in the run's own words, so the closest supported size is exact
25
- // and the render crops nothing. `2K` is the middle of the documented ladder: enough to
26
- // survive the 4x pre-scale the zoom needs without paying for 4K on every slide.
27
- const imageSize = "2K";
18
+ // 2.5 and Flash Lite cannot generate 2K images. Unknown models keep their own default
19
+ // resolution until their capabilities are known; discovery alone does not describe sizes.
20
+ const highResolutionModel = /^gemini-(?:3(?:\.1)?-pro|3\.1-flash)-image(?:-preview(?:-\d{2}-\d{2})?)?$/;
28
21
  // A wire payload is narrowed, never cast. The image arrives as one content part of a
29
22
  // `model_output` step, beside any text the model chose to write, so both the step and the
30
23
  // part are searched for by type rather than read off a fixed index.
@@ -58,7 +51,7 @@ const refusalWords = /\b(safety|blocked|content polic|prohibited|violat)/i;
58
51
  export function googleImage(deps) {
59
52
  return {
60
53
  id: "google-image",
61
- models: () => Promise.resolve(googleImageModels),
54
+ models: () => discoverGoogleImages(deps),
62
55
  generate: async (req) => {
63
56
  const response = await deps.fetch(`${googleImagesBase}/interactions`, {
64
57
  method: "POST",
@@ -72,7 +65,11 @@ export function googleImage(deps) {
72
65
  // The stage sends Number as that many independent calls, one piece each, so one
73
66
  // image per request is what it asks for. Nothing about style is set: the stage
74
67
  // asks for the provider's own.
75
- response_format: { type: "image", aspect_ratio: req.aspect, image_size: imageSize },
68
+ response_format: {
69
+ type: "image",
70
+ aspect_ratio: req.aspect,
71
+ ...(highResolutionModel.test(req.model) ? { image_size: "2K" } : {}),
72
+ },
76
73
  }),
77
74
  });
78
75
  if (!response.ok) {
@@ -0,0 +1,72 @@
1
+ import { z } from "zod";
2
+ import { providerError } from "../../kernel/ports/model.js";
3
+ const openAiModels = z.object({ data: z.array(z.object({ id: z.string() })) });
4
+ const googleModels = z.object({
5
+ models: z
6
+ .array(z.object({
7
+ name: z.string(),
8
+ displayName: z.string().optional(),
9
+ supportedGenerationMethods: z.array(z.string()).optional(),
10
+ }))
11
+ .default([]),
12
+ nextPageToken: z.string().optional(),
13
+ });
14
+ export async function discoverOpenAiImages(deps) {
15
+ const response = await deps.fetch("https://api.openai.com/v1/models", {
16
+ headers: { Authorization: `Bearer ${requireKey(deps)}` },
17
+ signal: AbortSignal.timeout(10_000),
18
+ });
19
+ if (!response.ok)
20
+ throw unavailable();
21
+ const parsed = openAiModels.safeParse(await response.json());
22
+ if (!parsed.success)
23
+ throw unavailable();
24
+ return parsed.data.data
25
+ .filter(({ id }) => id.startsWith("gpt-image-"))
26
+ .map(({ id }) => ({ id, name: id }));
27
+ }
28
+ export async function discoverGoogleImages(deps) {
29
+ const headers = { "x-goog-api-key": requireKey(deps) };
30
+ const signal = AbortSignal.timeout(10_000);
31
+ const result = [];
32
+ const tokens = new Set();
33
+ let page = "";
34
+ // Bound a malformed provider's pagination while allowing 20,000 model records.
35
+ for (let count = 0; count < 20; count++) {
36
+ const url = new URL("https://generativelanguage.googleapis.com/v1beta/models");
37
+ url.searchParams.set("pageSize", "1000");
38
+ if (page !== "")
39
+ url.searchParams.set("pageToken", page);
40
+ const response = await deps.fetch(url, { headers, signal });
41
+ if (!response.ok)
42
+ throw unavailable();
43
+ const parsed = googleModels.safeParse(await response.json());
44
+ if (!parsed.success)
45
+ throw unavailable();
46
+ for (const model of parsed.data.models) {
47
+ const id = model.name.replace(/^models\//, "");
48
+ // Imagen uses a different generation API. Only Gemini image models share this adapter.
49
+ if (id.startsWith("gemini-") &&
50
+ /-image(?:-|$)/.test(id) &&
51
+ model.supportedGenerationMethods?.includes("generateContent") === true) {
52
+ result.push({ id, name: model.displayName ?? id });
53
+ }
54
+ }
55
+ page = parsed.data.nextPageToken ?? "";
56
+ if (page === "")
57
+ return result;
58
+ if (tokens.has(page))
59
+ throw unavailable();
60
+ tokens.add(page);
61
+ }
62
+ throw unavailable();
63
+ }
64
+ function requireKey(deps) {
65
+ const key = deps.key();
66
+ if (key === undefined || key === "")
67
+ throw providerError({ kind: "missing_key", message: "Save an API key to load models." });
68
+ return key;
69
+ }
70
+ function unavailable() {
71
+ return providerError({ kind: "other", message: "The provider model list could not be loaded." });
72
+ }
@@ -3,15 +3,13 @@ import { redact } from "../../kernel/log.js";
3
3
  import { providerError } from "../../kernel/ports/model.js";
4
4
  import { retryAfter } from "../retry-after.js";
5
5
  import { describeBytes, sniffImage } from "./bytes.js";
6
+ import { discoverOpenAiImages } from "./models.js";
6
7
  // The HTTP gateway adapter for OpenAI's images endpoint: the platform's own `fetch` and
7
8
  // nothing else, because the whole call is one request. Unlike fal and Replicate this one
8
9
  // hands back the image itself - a GPT image model always answers with base64, never a URL -
9
10
  // so there is no link to follow.
10
11
  export const openAiImagesBase = "https://api.openai.com/v1";
11
- // The dropdown is filled from what the provider offers, and `/v1/models`
12
- // lists every model on the account, chat and embeddings among them, so the image
13
- // shortlist is this adapter's own data. These are the four GPT image models OpenAI
14
- // documents; adding the next one is a line here and no code change anywhere else.
12
+ // Offline choices only; the picker normally loads the provider catalogue.
15
13
  export const openAiImageModels = [
16
14
  { id: "gpt-image-2", name: "GPT Image 2" },
17
15
  { id: "gpt-image-1.5", name: "GPT Image 1.5" },
@@ -52,7 +50,7 @@ export function sizeFor(model, aspect) {
52
50
  export function openAiImage(deps) {
53
51
  return {
54
52
  id: "openai-image",
55
- models: () => Promise.resolve(openAiImageModels),
53
+ models: () => discoverOpenAiImages(deps),
56
54
  generate: async (req) => {
57
55
  const response = await deps.fetch(`${openAiImagesBase}/images/generations`, {
58
56
  method: "POST",
@@ -0,0 +1,23 @@
1
+ import { open } from "node:fs/promises";
2
+ // Bound both the initial read and a file that grows between stat and read. Model
3
+ // metadata is data only: none of the installed CLI's modules are evaluated.
4
+ export async function readCatalogueFile(path, maxBytes) {
5
+ const file = await open(path, "r");
6
+ try {
7
+ const info = await file.stat();
8
+ if (!info.isFile() || info.size > maxBytes)
9
+ throw new Error("Invalid model metadata file");
10
+ const bytes = Buffer.alloc(maxBytes + 1);
11
+ let size = 0;
12
+ while (size <= maxBytes) {
13
+ const chunk = await file.read(bytes, size, bytes.length - size, null);
14
+ if (chunk.bytesRead === 0)
15
+ return bytes.toString("utf8", 0, size);
16
+ size += chunk.bytesRead;
17
+ }
18
+ throw new Error("Model metadata file is too large");
19
+ }
20
+ finally {
21
+ await file.close();
22
+ }
23
+ }
@@ -8,10 +8,9 @@ import { lines } from "./sse-lines.js";
8
8
  // may not import `slices/**`, and `slices/settings/cli-status.ts` already probes the binary per
9
9
  // request. The registry `main.ts` builds is where this adapter and that probe meet.
10
10
  export const claudeCodeBinary = "claude";
11
- // The CLI takes an alias for the latest model of a family (`claude --help`, 2.1.258).
12
- // ceiling: a fixed list, because the CLI has no offline command that prints the models an
13
- // account may use. A user whose plan carries a model not listed here cannot pick it;
14
- // reading the list off the CLI is the upgrade when it can print one.
11
+ // Official stable family aliases resolve to the latest model available to the
12
+ // installed CLI/account. Full model IDs remain available through custom entry.
13
+ // https://code.claude.com/docs/en/model-config
15
14
  export const claudeCodeModels = [
16
15
  { id: "fable", name: "Claude Fable (latest)" },
17
16
  { id: "opus", name: "Claude Opus (latest)" },
@@ -0,0 +1,46 @@
1
+ import { homedir } from "node:os";
2
+ import { join } from "node:path";
3
+ import { z } from "zod";
4
+ import { readCatalogueFile } from "./catalogue-files.js";
5
+ const safeText = z
6
+ .string()
7
+ .trim()
8
+ .min(1)
9
+ .max(256)
10
+ .refine((text) => [...text].every((character) => character.charCodeAt(0) >= 32 && character.charCodeAt(0) !== 127));
11
+ // Deliberately omit model instructions, capabilities and every authentication
12
+ // file. Visibility refers to the CLI picker, not supported_in_api: some visible
13
+ // CLI-only models cannot be called through the public API.
14
+ const cache = z.object({ models: z.array(z.unknown()).max(1000) });
15
+ const model = z.object({
16
+ slug: safeText.refine((id) => !/\s/.test(id)),
17
+ display_name: safeText.optional(),
18
+ visibility: z.literal("list"),
19
+ priority: z.number().finite().optional(),
20
+ });
21
+ export async function nodeCodexModels(env = process.env) {
22
+ try {
23
+ const directory = env.CODEX_HOME?.trim() || join(homedir(), ".codex");
24
+ const source = await readCatalogueFile(join(directory, "models_cache.json"), 8 * 1024 * 1024);
25
+ const parsed = cache.parse(JSON.parse(source));
26
+ const models = parsed.models
27
+ .flatMap((entry) => {
28
+ const parsed = model.safeParse(entry);
29
+ return parsed.success ? [parsed.data] : [];
30
+ })
31
+ .sort((left, right) => (left.priority ?? Infinity) - (right.priority ?? Infinity));
32
+ const unique = new Map();
33
+ for (const item of models) {
34
+ if (!unique.has(item.slug)) {
35
+ unique.set(item.slug, { id: item.slug, name: item.display_name ?? item.slug });
36
+ }
37
+ }
38
+ if (unique.size === 0)
39
+ throw new Error("No visible models");
40
+ return [...unique.values()];
41
+ }
42
+ catch {
43
+ // Do not expose cache contents, home paths or raw JSON errors to the browser.
44
+ throw new Error("Codex model metadata is unavailable. Open Codex once to refresh its model list, or enter a custom model ID.");
45
+ }
46
+ }
@@ -7,15 +7,9 @@ import { lines } from "./sse-lines.js";
7
7
  // vocabulary: Codex writes a JSONL thread of `thread.started`, `item.*` and `turn.*`
8
8
  // events. No key here either - the CLI's own login authenticates it.
9
9
  export const codexBinary = "codex";
10
- // ceiling: a fixed list. `codex` has no offline command that prints the models an account
11
- // may use - `codex doctor` reports the install, not the catalogue, and the model refresh
12
- // it does at start needs the login this list is meant to be readable without. These two
13
- // are the ids this machine's `~/.codex/config.toml` names; reading the real catalogue is
14
- // the upgrade when the CLI grows a command that prints it.
15
- export const codexModels = [
16
- { id: "gpt-5.1-codex-max", name: "GPT-5.1 Codex Max" },
17
- { id: "gpt-5.6-sol", name: "GPT-5.6 Sol" },
18
- ];
10
+ // Codex publishes its account-specific picker in models_cache.json. With no
11
+ // cache, offer custom entry rather than pretending a pinned model is current.
12
+ export const codexModels = [];
19
13
  // `codex exec --help` (0.149.1) for the flags. `-c web_search=<mode>` is a TOML override, and
20
14
  // the binary's own error names the modes: "unknown variant `bogus`, expected one of `disabled`,
21
15
  // `cached`, `indexed`, `live`". `live` is the grounded mode research asks for and `disabled` is
@@ -31,6 +25,9 @@ export function codexArgs(req) {
31
25
  "--skip-git-repo-check",
32
26
  "-c",
33
27
  `web_search="${req.webSearch === true ? "live" : "disabled"}"`,
28
+ ...(req.thinkingConfig?.effort
29
+ ? ["-c", `model_reasoning_effort="${req.thinkingConfig.effort}"`]
30
+ : []),
34
31
  ...(req.model === "" ? [] : ["-m", req.model]),
35
32
  // The prompt is one argv element after `--`, so a leading dash is text, not a flag.
36
33
  "--",
@@ -115,7 +112,7 @@ export function codexLlm(deps) {
115
112
  id: "codex",
116
113
  // Prose arrives as whole messages; other JSONL events carry activity separately.
117
114
  capabilities: { streams: true, reportsUsage: true, webSearch: true },
118
- models: () => Promise.resolve(codexModels),
115
+ models: deps.readModels ?? (() => Promise.resolve(codexModels)),
119
116
  complete,
120
117
  };
121
118
  }
@@ -0,0 +1,85 @@
1
+ import { constants } from "node:fs";
2
+ import { access, realpath } from "node:fs/promises";
3
+ import { delimiter, dirname, isAbsolute, join } from "node:path";
4
+ import { cliCommand } from "../../kernel/cli-command.js";
5
+ import { readCatalogueFile } from "./catalogue-files.js";
6
+ // Official CLI aliases follow the installed CLI's model routing and account.
7
+ // https://geminicli.com/docs/cli/model/
8
+ export const geminiModels = [
9
+ { id: "auto", name: "Gemini Auto (CLI default)" },
10
+ { id: "pro", name: "Gemini Pro (CLI alias)" },
11
+ { id: "flash", name: "Gemini Flash (CLI alias)" },
12
+ { id: "flash-lite", name: "Gemini Flash-Lite (CLI alias)" },
13
+ ];
14
+ export async function nodeGeminiModels(binary) {
15
+ try {
16
+ const command = cliCommand(binary);
17
+ const entry = await executablePath(command.args[0] ?? command.file);
18
+ const candidates = new Set();
19
+ let directory = dirname(entry);
20
+ // npm and bun can install the core package nested under gemini-cli or
21
+ // hoisted beside it. Resolve the configured executable's symlinks first.
22
+ for (let depth = 0; depth < 10; depth += 1) {
23
+ const suffix = "gemini-cli-core/dist/src/config/models.js";
24
+ candidates.add(join(directory, "node_modules/@google", suffix));
25
+ candidates.add(join(directory, suffix));
26
+ const parent = dirname(directory);
27
+ if (parent === directory)
28
+ break;
29
+ directory = parent;
30
+ }
31
+ for (const candidate of candidates) {
32
+ try {
33
+ const models = parseGeminiModels(await readCatalogueFile(candidate, 256 * 1024));
34
+ if (models.length > geminiModels.length)
35
+ return models;
36
+ }
37
+ catch {
38
+ // A candidate is an optional package layout, not the final discovery result.
39
+ }
40
+ }
41
+ throw new Error("No installed model metadata");
42
+ }
43
+ catch {
44
+ throw new Error("Gemini CLI model metadata is unavailable. Use a documented CLI alias or enter a custom model ID.");
45
+ }
46
+ }
47
+ function parseGeminiModels(source) {
48
+ const models = new Map(geminiModels.map((model) => [model.id, model]));
49
+ // Read literal exported model constants only. Never import or execute an
50
+ // installed package, parse comments as entries, or inspect login/settings.
51
+ const declarations = /^export (?:const|let) ([A-Z][A-Z0-9_]*)\s*=\s*(['"])([^'"\r\n]+)\2\s*;/gm;
52
+ const uncommented = source.replace(/\/\*[\s\S]*?\*\//g, "");
53
+ for (const match of uncommented.matchAll(declarations)) {
54
+ const name = match[1];
55
+ const id = match[3];
56
+ if (name === undefined ||
57
+ id === undefined ||
58
+ !name.includes("MODEL") ||
59
+ name.includes("EMBEDDING") ||
60
+ !/^(?:gemini|gemma)-[a-z0-9.-]+$/.test(id))
61
+ continue;
62
+ if (!/^(?:(?:PREVIEW|DEFAULT|SECONDARY)_GEMINI_|GEMMA_)/.test(name))
63
+ continue;
64
+ models.set(id, { id, name: id });
65
+ }
66
+ return [...models.values()];
67
+ }
68
+ async function executablePath(binary) {
69
+ if (isAbsolute(binary) || /[\\/]/.test(binary))
70
+ return realpath(binary);
71
+ const path = Object.entries(process.env).find(([key]) => key.toLowerCase() === "path")?.[1] ?? "";
72
+ for (const directory of path.split(delimiter).slice(0, 128)) {
73
+ if (directory === "")
74
+ continue;
75
+ const candidate = join(directory.replace(/^"|"$/g, ""), binary);
76
+ try {
77
+ await access(candidate, constants.X_OK);
78
+ return await realpath(candidate);
79
+ }
80
+ catch {
81
+ // Continue normal PATH lookup; no process is launched for discovery.
82
+ }
83
+ }
84
+ throw new Error("Gemini executable was not found");
85
+ }
@@ -4,7 +4,7 @@ import { join } from "node:path";
4
4
  // Gemini 0.16 supports stream-json and tools.core, but headless default mode still
5
5
  // exposes file-reading tools. Isolate each writing call while retaining ~/.gemini
6
6
  // authentication. Nothing is written into the user's settings or project.
7
- export function geminiWorkspace(webSearch) {
7
+ export function geminiWorkspace(webSearch, request) {
8
8
  const directory = mkdtempSync(join(tmpdir(), "slopify-gemini-"));
9
9
  try {
10
10
  const settings = join(directory, "settings.json");
@@ -12,6 +12,29 @@ export function geminiWorkspace(webSearch) {
12
12
  const trust = join(directory, "trusted-folders.json");
13
13
  const mcpAllowlist = `slopify-disabled-${directory.split(/[\\/]/).at(-1)}`;
14
14
  writeFileSync(settings, JSON.stringify({
15
+ ...(request?.thinkingConfig
16
+ ? {
17
+ modelConfigs: {
18
+ customOverrides: [
19
+ {
20
+ match: { model: request.model },
21
+ modelConfig: {
22
+ generateContentConfig: {
23
+ thinkingConfig: {
24
+ ...(request.thinkingConfig.level === undefined
25
+ ? {}
26
+ : { thinkingLevel: request.thinkingConfig.level }),
27
+ ...(request.thinkingConfig.budget === undefined
28
+ ? {}
29
+ : { thinkingBudget: request.thinkingConfig.budget }),
30
+ },
31
+ },
32
+ },
33
+ },
34
+ ],
35
+ },
36
+ }
37
+ : {}),
15
38
  tools: {
16
39
  core: webSearch ? ["google_web_search"] : [],
17
40
  allowed: webSearch ? ["google_web_search"] : [],
@@ -1,15 +1,12 @@
1
1
  import { z } from "zod";
2
2
  import { redact } from "../../kernel/log.js";
3
3
  import { providerError } from "../../kernel/ports/model.js";
4
+ import { geminiModels } from "./gemini-models.js";
4
5
  import { geminiWorkspace } from "./gemini-workspace.js";
5
6
  import { cliEvent, cliShaped, endedWithout, promptOf } from "./run-cli.js";
6
7
  import { lines } from "./sse-lines.js";
7
8
  export const geminiBinary = "gemini";
8
- export const geminiModels = [
9
- { id: "gemini-2.5-pro", name: "Gemini 2.5 Pro" },
10
- { id: "gemini-2.5-flash", name: "Gemini 2.5 Flash" },
11
- { id: "gemini-2.5-flash-lite", name: "Gemini 2.5 Flash-Lite" },
12
- ];
9
+ export { geminiModels } from "./gemini-models.js";
13
10
  // Verified against installed Gemini CLI 0.16.0's config/nonInteractiveCli sources.
14
11
  // Its --allowed-tools flag bypasses approval; tools.core in the isolated settings
15
12
  // actually restricts discovery. Never request yolo/auto_edit.
@@ -46,7 +43,7 @@ export function geminiLlm(deps) {
46
43
  const binary = deps.binary ?? geminiBinary;
47
44
  async function* complete(req) {
48
45
  req.signal.throwIfAborted();
49
- const workspace = geminiWorkspace(req.webSearch === true);
46
+ const workspace = geminiWorkspace(req.webSearch === true, req);
50
47
  let run;
51
48
  try {
52
49
  run = deps.run(binary, geminiArgs(req, workspace.mcpAllowlist), req.signal, workspace.options);
@@ -103,7 +100,7 @@ export function geminiLlm(deps) {
103
100
  return {
104
101
  id: "gemini",
105
102
  capabilities: { streams: true, reportsUsage: true, webSearch: true },
106
- models: () => Promise.resolve(geminiModels),
103
+ models: deps.readModels ?? (() => Promise.resolve(geminiModels)),
107
104
  complete,
108
105
  };
109
106
  }
@@ -14,7 +14,11 @@ const appHeaders = {
14
14
  // A wire payload is narrowed, never cast: everything unlisted is dropped at the seam so
15
15
  // no vendor shape can leak past this file.
16
16
  const modelList = z.object({
17
- data: z.array(z.object({ id: z.string(), name: z.string().optional() })),
17
+ data: z.array(z.object({
18
+ id: z.string(),
19
+ name: z.string().optional(),
20
+ architecture: z.object({ output_modalities: z.array(z.string()).optional() }).optional(),
21
+ })),
18
22
  });
19
23
  const errorBody = z.object({
20
24
  error: z.object({ message: z.string(), code: z.union([z.number(), z.string()]).optional() }),
@@ -43,6 +47,7 @@ export function openRouterLlm(deps) {
43
47
  role: message.role,
44
48
  content: message.content,
45
49
  })),
50
+ ...(req.thinkingConfig?.effort ? { reasoning: { effort: req.thinkingConfig.effort } } : {}),
46
51
  stream: true,
47
52
  // The usage-accounting flag: without it the final chunk carries no token counts
48
53
  // and the Usage page would have nothing to count.
@@ -112,6 +117,7 @@ export function openRouterLlm(deps) {
112
117
  capabilities: { streams: true, reportsUsage: true, webSearch: true },
113
118
  models: async () => {
114
119
  const response = await deps.fetch(`${openRouterBase}/models`, {
120
+ signal: AbortSignal.timeout(10_000),
115
121
  headers: headers(deps.key()),
116
122
  });
117
123
  if (!response.ok) {
@@ -124,7 +130,11 @@ export function openRouterLlm(deps) {
124
130
  message: "OpenRouter's model list was not in the shape this app can read",
125
131
  });
126
132
  }
127
- return parsed.data.data.map((model) => ({ id: model.id, name: model.name ?? model.id }));
133
+ // /models defaults to text output. Also reject an explicit non-text row
134
+ // if a gateway response includes one; Slopify sends chat completions here.
135
+ return parsed.data.data
136
+ .filter((model) => model.architecture?.output_modalities?.includes("text") !== false)
137
+ .map((model) => ({ id: model.id, name: model.name ?? model.id }));
128
138
  },
129
139
  complete,
130
140
  };
@@ -11,6 +11,13 @@ export const cartesiaBase = "https://api.cartesia.ai";
11
11
  // constant rather than left to the account's default.
12
12
  export const cartesiaVersion = "2026-03-01";
13
13
  export const cartesiaModel = "sonic-3.5";
14
+ // Cartesia publishes model IDs in its docs, with no model-list API. Stable family aliases
15
+ // receive new snapshots automatically: https://docs.cartesia.ai/build-with-cartesia/tts-models/latest
16
+ export const cartesiaModels = [
17
+ { id: "sonic-3.6", name: "Sonic 3.6" },
18
+ { id: "sonic-3.5", name: "Sonic 3.5" },
19
+ { id: "sonic-3", name: "Sonic 3" },
20
+ ];
14
21
  // mp3 is the port's container; `bit_rate` is required for it and `sample_rate` fixes the
15
22
  // rate the concatenation then keeps.
16
23
  const outputFormat = { container: "mp3", bit_rate: 128_000, sample_rate: 44_100 };
@@ -24,6 +31,7 @@ export function cartesiaTts(deps) {
24
31
  return {
25
32
  id: "cartesia",
26
33
  capabilities: { streams: true },
34
+ models: async () => cartesiaModels,
27
35
  synthesize: async (req) => {
28
36
  const response = await deps.fetch(`${cartesiaBase}/tts/bytes`, {
29
37
  method: "POST",
@@ -36,7 +44,7 @@ export function cartesiaTts(deps) {
36
44
  // No pre-check on length; Cartesia's own limit surfaces as its
37
45
  // error. `language` is left out so the model reads it off the transcript.
38
46
  body: JSON.stringify({
39
- model_id: cartesiaModel,
47
+ model_id: req.model ?? cartesiaModel,
40
48
  transcript: req.text,
41
49
  voice: { mode: "id", id: req.voiceId },
42
50
  output_format: outputFormat,