@gentbajko/slopify 0.6.1 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +36 -1
- package/SUBTITLES.md +3 -3
- package/dist/adapter-registry.js +12 -2
- package/dist/adapters/image/google.js +12 -15
- package/dist/adapters/image/models.js +72 -0
- package/dist/adapters/image/openai.js +3 -5
- package/dist/adapters/llm/catalogue-files.js +23 -0
- package/dist/adapters/llm/claude-code.js +3 -4
- package/dist/adapters/llm/codex-models.js +46 -0
- package/dist/adapters/llm/codex.js +4 -10
- package/dist/adapters/llm/gemini-models.js +85 -0
- package/dist/adapters/llm/gemini.js +3 -6
- package/dist/adapters/llm/openrouter.js +11 -2
- package/dist/adapters/tts/cartesia.js +9 -1
- package/dist/adapters/tts/elevenlabs.js +33 -2
- package/dist/adapters/tts/inworld-async.js +150 -0
- package/dist/adapters/tts/inworld-text.js +32 -0
- package/dist/adapters/tts/inworld.js +170 -0
- package/dist/adapters/tts/openai.js +31 -4
- package/dist/edge/cli.js +5 -0
- package/dist/edge/events/hub.js +9 -2
- package/dist/edge/events/preview-cache.js +40 -0
- package/dist/edge/http/actions.js +2 -0
- package/dist/edge/http/app.js +26 -1
- package/dist/edge/http/audio-preview.js +46 -0
- package/dist/edge/http/providers.js +16 -0
- package/dist/edge/http/update.js +57 -0
- package/dist/edge/update-worker.js +44 -0
- package/dist/kernel/audio-preview.js +188 -0
- package/dist/kernel/runner/providers.js +72 -22
- package/dist/main.js +104 -24
- package/dist/model-catalog.js +37 -0
- package/dist/slices/control/index.js +1 -0
- package/dist/slices/control/providers.js +2 -0
- package/dist/slices/narration/live.js +21 -0
- package/dist/slices/narration/run.js +9 -9
- package/dist/slices/research/run.js +3 -0
- package/dist/slices/settings/model.js +2 -0
- package/dist/slices/settings/models.js +89 -0
- package/dist/slices/subtitles/captions.js +4 -1
- package/dist/slices/subtitles/layout.js +19 -0
- package/dist/slices/subtitles/model.js +9 -0
- package/dist/slices/subtitles/prepare.js +6 -1
- package/dist/updater/candidate.js +28 -0
- package/dist/updater/forward.js +35 -0
- package/dist/updater/install-flow.js +26 -0
- package/dist/updater/install.js +50 -0
- package/dist/updater/model.js +24 -0
- package/dist/updater/plan.js +135 -0
- package/dist/updater/readiness.js +15 -0
- package/dist/updater/registry.js +19 -0
- package/dist/updater/service.js +128 -0
- package/dist/updater/worker.js +136 -0
- package/dist/web/assets/index-6QIQ-TzO.css +1 -0
- package/dist/web/assets/{index-CNYs0noB.js → index-CNgWL-ct.js} +24 -24
- package/dist/web/index.html +2 -2
- package/package.json +1 -1
- package/dist/web/assets/index-Mk0bvBg-.css +0 -1
package/README.md
CHANGED
|
@@ -1,4 +1,15 @@
|
|
|
1
|
-
|
|
1
|
+
<h1 align="center">
|
|
2
|
+
<a href="https://slopify.stream"><img src="https://slopify.stream/assets/favicon.svg" width="40" height="40" align="middle" alt="" /></a>
|
|
3
|
+
Slopify
|
|
4
|
+
</h1>
|
|
5
|
+
|
|
6
|
+
<p align="center">
|
|
7
|
+
<a href="https://slopify.stream">slopify.stream</a>
|
|
8
|
+
·
|
|
9
|
+
<a href="https://www.patreon.com/cw/GentBajko"><img src="https://slopify.stream/assets/patreon-green.svg" width="14" height="14" align="middle" alt="" /> Patreon</a>
|
|
10
|
+
·
|
|
11
|
+
<a href="https://buymeacoffee.com/gentbajko"><img src="https://slopify.stream/assets/buymeacoffee-green.svg" width="14" height="14" align="middle" alt="" /> Buy Me a Coffee</a>
|
|
12
|
+
</p>
|
|
2
13
|
|
|
3
14
|
A prompt and a few keywords in. A narrated slideshow video out. Your keys, your machine, free.
|
|
4
15
|
|
|
@@ -14,6 +25,23 @@ Six stages run as a graph: research, article, narration, images, thumbnail, vide
|
|
|
14
25
|
Generate any of them, or provide the output yourself and that stage is skipped. The
|
|
15
26
|
video is a slideshow with alternating zoom over the narration, rendered with ffmpeg.
|
|
16
27
|
|
|
28
|
+
## Inworld narration
|
|
29
|
+
|
|
30
|
+
In Settings, add the **Base64 credentials** from Inworld's API Keys page, then add an
|
|
31
|
+
Inworld voice ID (for example `Dennis`, or a voice from your workspace). In Play,
|
|
32
|
+
choose **Realtime TTS-2** or **Realtime TTS-2 Flash** and that voice.
|
|
33
|
+
|
|
34
|
+
Short text streams immediately. TTS-2 text over 4,000 characters uses one async job,
|
|
35
|
+
up to 100,000 characters; Inworld caps On-Demand accounts at 10,000. Audio becomes
|
|
36
|
+
available once that job finishes. Flash uses streamed parts of at most 4,000 characters.
|
|
37
|
+
For longer articles, select paragraph chunking. Successful status checks keep long jobs
|
|
38
|
+
alive, and automatic polling/download retries reuse the accepted job. Pausing stops
|
|
39
|
+
local requests; Inworld may still finish and bill an accepted job. Resuming after a
|
|
40
|
+
pause or app restart starts a new request for unfinished narration.
|
|
41
|
+
|
|
42
|
+
See [Inworld's async API](https://docs.inworld.ai/api-reference/ttsAPI/texttospeech/synthesize-speech-async)
|
|
43
|
+
for account limits. Both model IDs are bundled; Inworld's LLM catalogue does not list TTS models.
|
|
44
|
+
|
|
17
45
|
## Options
|
|
18
46
|
|
|
19
47
|
| Flag | Environment variable | Default |
|
|
@@ -67,6 +95,13 @@ Never your keys, prompts, keywords, titles, article text, filenames, or anything
|
|
|
67
95
|
your machine. A notice says all of this the first time you run it, before the machine
|
|
68
96
|
id exists, and the Usage screen shows you your own numbers at any time.
|
|
69
97
|
|
|
98
|
+
## Supporting the project
|
|
99
|
+
|
|
100
|
+
Slopify is free and always will be. If it is worth something to you, there is
|
|
101
|
+
[Patreon](https://www.patreon.com/cw/GentBajko) and
|
|
102
|
+
[Buy Me a Coffee](https://buymeacoffee.com/gentbajko). The people who do are listed
|
|
103
|
+
in [SUPPORTERS.md](https://github.com/GentBajko/slopify/blob/main/SUPPORTERS.md).
|
|
104
|
+
|
|
70
105
|
## Licence
|
|
71
106
|
|
|
72
107
|
MIT. The ffmpeg binary fetched at install time is a separate GPL-3.0-or-later program,
|
package/SUBTITLES.md
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
|
-
# Subtitles in Slopify
|
|
1
|
+
# Subtitles in Slopify
|
|
2
2
|
|
|
3
3
|
1. In Play, enable narration or provide matching English audio and article text.
|
|
4
4
|
2. In Subtitles, choose **Subtitle files (.srt + .vtt)** or **Burn into video + files**.
|
|
5
|
-
3. For burned captions, select a bundled or system font, or upload a `.ttf` or `.otf` file (up to 32 MiB).
|
|
5
|
+
3. For burned captions, select a bundled or system font, or upload a `.ttf` or `.otf` file (up to 32 MiB). Choose a size from 16 to 120 and a position: top, upper-middle, center, lower-middle or bottom. The side preview shows the selected font, size and position in a 16:9 or 9:16 frame.
|
|
6
6
|
4. Start the run. Slopify times the article against the actual narration on your computer. The first use downloads an approximately 95 MB English speech model; later runs work offline with the cached model. Allow roughly 1 GB of available memory during alignment. No extra API key, Python, or compiler is needed.
|
|
7
7
|
5. Download SRT/VTT beside the final export. Files mode adds an optional native caption track to the in-app video preview. Burned captions remain visible in the downloaded MP4.
|
|
8
8
|
|
|
9
|
-
For a completed project, open its final Video or Audio export section, set subtitles, and click **Save subtitles**. This rebuilds only the local export from saved narration and images. Changing font or
|
|
9
|
+
For a completed project, open its final Video or Audio export section, set subtitles, and click **Save subtitles**. This rebuilds only the local export from saved narration and images. Changing font, size or position reuses word timing when the audio and spoken text are unchanged. A paused run saves these choices until Resume; pause an active run before editing. Failed alignment or rendering keeps the previous finished export.
|
|
10
10
|
|
|
11
11
|
Audio Off disables subtitles. Video Off produces WAV audio with separate subtitle files. Uploaded audio must match the article; substantial mismatches fail with a message to correct the transcript. Review timing and spelling before publishing. English is supported first; unusual pronunciations and non-English passages can fail alignment.
|
|
12
12
|
|
package/dist/adapter-registry.js
CHANGED
|
@@ -4,10 +4,13 @@ import { openAiImage } from "./adapters/image/openai.js";
|
|
|
4
4
|
import { replicateImage } from "./adapters/image/replicate.js";
|
|
5
5
|
import { claudeCodeLlm } from "./adapters/llm/claude-code.js";
|
|
6
6
|
import { codexLlm } from "./adapters/llm/codex.js";
|
|
7
|
+
import { nodeCodexModels } from "./adapters/llm/codex-models.js";
|
|
7
8
|
import { geminiLlm } from "./adapters/llm/gemini.js";
|
|
9
|
+
import { nodeGeminiModels } from "./adapters/llm/gemini-models.js";
|
|
8
10
|
import { openRouterLlm } from "./adapters/llm/openrouter.js";
|
|
9
11
|
import { cartesiaTts } from "./adapters/tts/cartesia.js";
|
|
10
12
|
import { elevenLabsTts } from "./adapters/tts/elevenlabs.js";
|
|
13
|
+
import { inworldTts } from "./adapters/tts/inworld.js";
|
|
11
14
|
import { openAiTts } from "./adapters/tts/openai.js";
|
|
12
15
|
import { cliBinary } from "./slices/settings/cli-paths.js";
|
|
13
16
|
import { keyForAttempt } from "./slices/settings/keys.js";
|
|
@@ -26,8 +29,14 @@ export function buildRegistry(deps) {
|
|
|
26
29
|
["openrouter", openRouterLlm({ fetch: deps.fetch, key: keyOf("openrouter") })],
|
|
27
30
|
// Each CLI authenticates with its own login.
|
|
28
31
|
["claude-code", claudeCodeLlm({ run: cliFor("claude-code") })],
|
|
29
|
-
["codex", codexLlm({ run: cliFor("codex") })],
|
|
30
|
-
[
|
|
32
|
+
["codex", codexLlm({ run: cliFor("codex"), readModels: () => nodeCodexModels() })],
|
|
33
|
+
[
|
|
34
|
+
"gemini",
|
|
35
|
+
geminiLlm({
|
|
36
|
+
run: cliFor("gemini"),
|
|
37
|
+
readModels: () => nodeGeminiModels(cliBinary(deps.db, "gemini")),
|
|
38
|
+
}),
|
|
39
|
+
],
|
|
31
40
|
]);
|
|
32
41
|
// One key per provider, so each adapter is handed the reader for its own row and no other.
|
|
33
42
|
// OpenAI keeps two rows because it ships an adapter in two families and a user may key one
|
|
@@ -36,6 +45,7 @@ export function buildRegistry(deps) {
|
|
|
36
45
|
["elevenlabs", elevenLabsTts({ fetch: deps.fetch, key: keyOf("elevenlabs") })],
|
|
37
46
|
["openai-tts", openAiTts({ fetch: deps.fetch, key: keyOf("openai-tts") })],
|
|
38
47
|
["cartesia", cartesiaTts({ fetch: deps.fetch, key: keyOf("cartesia") })],
|
|
48
|
+
["inworld", inworldTts({ fetch: deps.fetch, key: keyOf("inworld"), clock: deps.clock })],
|
|
39
49
|
]);
|
|
40
50
|
// Four image providers behind one port, each handed the reader for its own key row.
|
|
41
51
|
// Replicate also takes the clock: `Prefer: wait` gives up after 60 s and the prediction has
|
|
@@ -3,28 +3,21 @@ import { redact } from "../../kernel/log.js";
|
|
|
3
3
|
import { providerError } from "../../kernel/ports/model.js";
|
|
4
4
|
import { retryAfter } from "../retry-after.js";
|
|
5
5
|
import { describeBytes, sniffImage } from "./bytes.js";
|
|
6
|
+
import { discoverGoogleImages } from "./models.js";
|
|
6
7
|
// The HTTP gateway adapter for Google's own image generation, billed to a Gemini API key
|
|
7
8
|
// rather than to a host reselling the same models. Like the OpenAI one it hands back the
|
|
8
|
-
// bytes with the call, so there is no link to follow
|
|
9
|
-
// one `input` string and a `response_format`, with no per-model size table to keep.
|
|
9
|
+
// bytes with the call, so there is no link to follow.
|
|
10
10
|
export const googleImagesBase = "https://generativelanguage.googleapis.com/v1beta";
|
|
11
|
-
//
|
|
12
|
-
// text and embedding model too, so the image shortlist is this adapter's own data. Named as
|
|
13
|
-
// Google markets them: "Nano Banana" is the family, `gemini-*-image` is what the API answers
|
|
14
|
-
// to, and the picker shows the name a user would recognise. Newest first.
|
|
15
|
-
//
|
|
16
|
-
// A Gemini model without the `-image` suffix returns text and cannot be used here, whatever
|
|
17
|
-
// its version number: `gemini-3.8-flash` is newer than all of these and generates no images.
|
|
11
|
+
// Offline choices only; the picker normally loads the provider catalogue.
|
|
18
12
|
export const googleImageModels = [
|
|
19
13
|
{ id: "gemini-3.1-flash-image", name: "Nano Banana 2" },
|
|
20
14
|
{ id: "gemini-3.1-flash-lite-image", name: "Nano Banana 2 Lite" },
|
|
21
15
|
{ id: "gemini-3-pro-image", name: "Nano Banana Pro" },
|
|
22
16
|
{ id: "gemini-2.5-flash-image", name: "Nano Banana" },
|
|
23
17
|
];
|
|
24
|
-
//
|
|
25
|
-
//
|
|
26
|
-
|
|
27
|
-
const imageSize = "2K";
|
|
18
|
+
// 2.5 and Flash Lite cannot generate 2K images. Unknown models keep their own default
|
|
19
|
+
// resolution until their capabilities are known; discovery alone does not describe sizes.
|
|
20
|
+
const highResolutionModel = /^gemini-(?:3(?:\.1)?-pro|3\.1-flash)-image(?:-preview(?:-\d{2}-\d{2})?)?$/;
|
|
28
21
|
// A wire payload is narrowed, never cast. The image arrives as one content part of a
|
|
29
22
|
// `model_output` step, beside any text the model chose to write, so both the step and the
|
|
30
23
|
// part are searched for by type rather than read off a fixed index.
|
|
@@ -58,7 +51,7 @@ const refusalWords = /\b(safety|blocked|content polic|prohibited|violat)/i;
|
|
|
58
51
|
export function googleImage(deps) {
|
|
59
52
|
return {
|
|
60
53
|
id: "google-image",
|
|
61
|
-
models: () =>
|
|
54
|
+
models: () => discoverGoogleImages(deps),
|
|
62
55
|
generate: async (req) => {
|
|
63
56
|
const response = await deps.fetch(`${googleImagesBase}/interactions`, {
|
|
64
57
|
method: "POST",
|
|
@@ -72,7 +65,11 @@ export function googleImage(deps) {
|
|
|
72
65
|
// The stage sends Number as that many independent calls, one piece each, so one
|
|
73
66
|
// image per request is what it asks for. Nothing about style is set: the stage
|
|
74
67
|
// asks for the provider's own.
|
|
75
|
-
response_format: {
|
|
68
|
+
response_format: {
|
|
69
|
+
type: "image",
|
|
70
|
+
aspect_ratio: req.aspect,
|
|
71
|
+
...(highResolutionModel.test(req.model) ? { image_size: "2K" } : {}),
|
|
72
|
+
},
|
|
76
73
|
}),
|
|
77
74
|
});
|
|
78
75
|
if (!response.ok) {
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
import { providerError } from "../../kernel/ports/model.js";
|
|
3
|
+
const openAiModels = z.object({ data: z.array(z.object({ id: z.string() })) });
|
|
4
|
+
const googleModels = z.object({
|
|
5
|
+
models: z
|
|
6
|
+
.array(z.object({
|
|
7
|
+
name: z.string(),
|
|
8
|
+
displayName: z.string().optional(),
|
|
9
|
+
supportedGenerationMethods: z.array(z.string()).optional(),
|
|
10
|
+
}))
|
|
11
|
+
.default([]),
|
|
12
|
+
nextPageToken: z.string().optional(),
|
|
13
|
+
});
|
|
14
|
+
export async function discoverOpenAiImages(deps) {
|
|
15
|
+
const response = await deps.fetch("https://api.openai.com/v1/models", {
|
|
16
|
+
headers: { Authorization: `Bearer ${requireKey(deps)}` },
|
|
17
|
+
signal: AbortSignal.timeout(10_000),
|
|
18
|
+
});
|
|
19
|
+
if (!response.ok)
|
|
20
|
+
throw unavailable();
|
|
21
|
+
const parsed = openAiModels.safeParse(await response.json());
|
|
22
|
+
if (!parsed.success)
|
|
23
|
+
throw unavailable();
|
|
24
|
+
return parsed.data.data
|
|
25
|
+
.filter(({ id }) => id.startsWith("gpt-image-"))
|
|
26
|
+
.map(({ id }) => ({ id, name: id }));
|
|
27
|
+
}
|
|
28
|
+
export async function discoverGoogleImages(deps) {
|
|
29
|
+
const headers = { "x-goog-api-key": requireKey(deps) };
|
|
30
|
+
const signal = AbortSignal.timeout(10_000);
|
|
31
|
+
const result = [];
|
|
32
|
+
const tokens = new Set();
|
|
33
|
+
let page = "";
|
|
34
|
+
// Bound a malformed provider's pagination while allowing 20,000 model records.
|
|
35
|
+
for (let count = 0; count < 20; count++) {
|
|
36
|
+
const url = new URL("https://generativelanguage.googleapis.com/v1beta/models");
|
|
37
|
+
url.searchParams.set("pageSize", "1000");
|
|
38
|
+
if (page !== "")
|
|
39
|
+
url.searchParams.set("pageToken", page);
|
|
40
|
+
const response = await deps.fetch(url, { headers, signal });
|
|
41
|
+
if (!response.ok)
|
|
42
|
+
throw unavailable();
|
|
43
|
+
const parsed = googleModels.safeParse(await response.json());
|
|
44
|
+
if (!parsed.success)
|
|
45
|
+
throw unavailable();
|
|
46
|
+
for (const model of parsed.data.models) {
|
|
47
|
+
const id = model.name.replace(/^models\//, "");
|
|
48
|
+
// Imagen uses a different generation API. Only Gemini image models share this adapter.
|
|
49
|
+
if (id.startsWith("gemini-") &&
|
|
50
|
+
/-image(?:-|$)/.test(id) &&
|
|
51
|
+
model.supportedGenerationMethods?.includes("generateContent") === true) {
|
|
52
|
+
result.push({ id, name: model.displayName ?? id });
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
page = parsed.data.nextPageToken ?? "";
|
|
56
|
+
if (page === "")
|
|
57
|
+
return result;
|
|
58
|
+
if (tokens.has(page))
|
|
59
|
+
throw unavailable();
|
|
60
|
+
tokens.add(page);
|
|
61
|
+
}
|
|
62
|
+
throw unavailable();
|
|
63
|
+
}
|
|
64
|
+
function requireKey(deps) {
|
|
65
|
+
const key = deps.key();
|
|
66
|
+
if (key === undefined || key === "")
|
|
67
|
+
throw providerError({ kind: "missing_key", message: "Save an API key to load models." });
|
|
68
|
+
return key;
|
|
69
|
+
}
|
|
70
|
+
function unavailable() {
|
|
71
|
+
return providerError({ kind: "other", message: "The provider model list could not be loaded." });
|
|
72
|
+
}
|
|
@@ -3,15 +3,13 @@ import { redact } from "../../kernel/log.js";
|
|
|
3
3
|
import { providerError } from "../../kernel/ports/model.js";
|
|
4
4
|
import { retryAfter } from "../retry-after.js";
|
|
5
5
|
import { describeBytes, sniffImage } from "./bytes.js";
|
|
6
|
+
import { discoverOpenAiImages } from "./models.js";
|
|
6
7
|
// The HTTP gateway adapter for OpenAI's images endpoint: the platform's own `fetch` and
|
|
7
8
|
// nothing else, because the whole call is one request. Unlike fal and Replicate this one
|
|
8
9
|
// hands back the image itself - a GPT image model always answers with base64, never a URL -
|
|
9
10
|
// so there is no link to follow.
|
|
10
11
|
export const openAiImagesBase = "https://api.openai.com/v1";
|
|
11
|
-
//
|
|
12
|
-
// lists every model on the account, chat and embeddings among them, so the image
|
|
13
|
-
// shortlist is this adapter's own data. These are the four GPT image models OpenAI
|
|
14
|
-
// documents; adding the next one is a line here and no code change anywhere else.
|
|
12
|
+
// Offline choices only; the picker normally loads the provider catalogue.
|
|
15
13
|
export const openAiImageModels = [
|
|
16
14
|
{ id: "gpt-image-2", name: "GPT Image 2" },
|
|
17
15
|
{ id: "gpt-image-1.5", name: "GPT Image 1.5" },
|
|
@@ -52,7 +50,7 @@ export function sizeFor(model, aspect) {
|
|
|
52
50
|
export function openAiImage(deps) {
|
|
53
51
|
return {
|
|
54
52
|
id: "openai-image",
|
|
55
|
-
models: () =>
|
|
53
|
+
models: () => discoverOpenAiImages(deps),
|
|
56
54
|
generate: async (req) => {
|
|
57
55
|
const response = await deps.fetch(`${openAiImagesBase}/images/generations`, {
|
|
58
56
|
method: "POST",
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
import { open } from "node:fs/promises";
|
|
2
|
+
// Bound both the initial read and a file that grows between stat and read. Model
|
|
3
|
+
// metadata is data only: none of the installed CLI's modules are evaluated.
|
|
4
|
+
export async function readCatalogueFile(path, maxBytes) {
|
|
5
|
+
const file = await open(path, "r");
|
|
6
|
+
try {
|
|
7
|
+
const info = await file.stat();
|
|
8
|
+
if (!info.isFile() || info.size > maxBytes)
|
|
9
|
+
throw new Error("Invalid model metadata file");
|
|
10
|
+
const bytes = Buffer.alloc(maxBytes + 1);
|
|
11
|
+
let size = 0;
|
|
12
|
+
while (size <= maxBytes) {
|
|
13
|
+
const chunk = await file.read(bytes, size, bytes.length - size, null);
|
|
14
|
+
if (chunk.bytesRead === 0)
|
|
15
|
+
return bytes.toString("utf8", 0, size);
|
|
16
|
+
size += chunk.bytesRead;
|
|
17
|
+
}
|
|
18
|
+
throw new Error("Model metadata file is too large");
|
|
19
|
+
}
|
|
20
|
+
finally {
|
|
21
|
+
await file.close();
|
|
22
|
+
}
|
|
23
|
+
}
|
|
@@ -8,10 +8,9 @@ import { lines } from "./sse-lines.js";
|
|
|
8
8
|
// may not import `slices/**`, and `slices/settings/cli-status.ts` already probes the binary per
|
|
9
9
|
// request. The registry `main.ts` builds is where this adapter and that probe meet.
|
|
10
10
|
export const claudeCodeBinary = "claude";
|
|
11
|
-
//
|
|
12
|
-
//
|
|
13
|
-
//
|
|
14
|
-
// reading the list off the CLI is the upgrade when it can print one.
|
|
11
|
+
// Official stable family aliases resolve to the latest model available to the
|
|
12
|
+
// installed CLI/account. Full model IDs remain available through custom entry.
|
|
13
|
+
// https://code.claude.com/docs/en/model-config
|
|
15
14
|
export const claudeCodeModels = [
|
|
16
15
|
{ id: "fable", name: "Claude Fable (latest)" },
|
|
17
16
|
{ id: "opus", name: "Claude Opus (latest)" },
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { homedir } from "node:os";
|
|
2
|
+
import { join } from "node:path";
|
|
3
|
+
import { z } from "zod";
|
|
4
|
+
import { readCatalogueFile } from "./catalogue-files.js";
|
|
5
|
+
const safeText = z
|
|
6
|
+
.string()
|
|
7
|
+
.trim()
|
|
8
|
+
.min(1)
|
|
9
|
+
.max(256)
|
|
10
|
+
.refine((text) => [...text].every((character) => character.charCodeAt(0) >= 32 && character.charCodeAt(0) !== 127));
|
|
11
|
+
// Deliberately omit model instructions, capabilities and every authentication
|
|
12
|
+
// file. Visibility refers to the CLI picker, not supported_in_api: some visible
|
|
13
|
+
// CLI-only models cannot be called through the public API.
|
|
14
|
+
const cache = z.object({ models: z.array(z.unknown()).max(1000) });
|
|
15
|
+
const model = z.object({
|
|
16
|
+
slug: safeText.refine((id) => !/\s/.test(id)),
|
|
17
|
+
display_name: safeText.optional(),
|
|
18
|
+
visibility: z.literal("list"),
|
|
19
|
+
priority: z.number().finite().optional(),
|
|
20
|
+
});
|
|
21
|
+
export async function nodeCodexModels(env = process.env) {
|
|
22
|
+
try {
|
|
23
|
+
const directory = env.CODEX_HOME?.trim() || join(homedir(), ".codex");
|
|
24
|
+
const source = await readCatalogueFile(join(directory, "models_cache.json"), 8 * 1024 * 1024);
|
|
25
|
+
const parsed = cache.parse(JSON.parse(source));
|
|
26
|
+
const models = parsed.models
|
|
27
|
+
.flatMap((entry) => {
|
|
28
|
+
const parsed = model.safeParse(entry);
|
|
29
|
+
return parsed.success ? [parsed.data] : [];
|
|
30
|
+
})
|
|
31
|
+
.sort((left, right) => (left.priority ?? Infinity) - (right.priority ?? Infinity));
|
|
32
|
+
const unique = new Map();
|
|
33
|
+
for (const item of models) {
|
|
34
|
+
if (!unique.has(item.slug)) {
|
|
35
|
+
unique.set(item.slug, { id: item.slug, name: item.display_name ?? item.slug });
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
if (unique.size === 0)
|
|
39
|
+
throw new Error("No visible models");
|
|
40
|
+
return [...unique.values()];
|
|
41
|
+
}
|
|
42
|
+
catch {
|
|
43
|
+
// Do not expose cache contents, home paths or raw JSON errors to the browser.
|
|
44
|
+
throw new Error("Codex model metadata is unavailable. Open Codex once to refresh its model list, or enter a custom model ID.");
|
|
45
|
+
}
|
|
46
|
+
}
|
|
@@ -7,15 +7,9 @@ import { lines } from "./sse-lines.js";
|
|
|
7
7
|
// vocabulary: Codex writes a JSONL thread of `thread.started`, `item.*` and `turn.*`
|
|
8
8
|
// events. No key here either - the CLI's own login authenticates it.
|
|
9
9
|
export const codexBinary = "codex";
|
|
10
|
-
//
|
|
11
|
-
//
|
|
12
|
-
|
|
13
|
-
// are the ids this machine's `~/.codex/config.toml` names; reading the real catalogue is
|
|
14
|
-
// the upgrade when the CLI grows a command that prints it.
|
|
15
|
-
export const codexModels = [
|
|
16
|
-
{ id: "gpt-5.1-codex-max", name: "GPT-5.1 Codex Max" },
|
|
17
|
-
{ id: "gpt-5.6-sol", name: "GPT-5.6 Sol" },
|
|
18
|
-
];
|
|
10
|
+
// Codex publishes its account-specific picker in models_cache.json. With no
|
|
11
|
+
// cache, offer custom entry rather than pretending a pinned model is current.
|
|
12
|
+
export const codexModels = [];
|
|
19
13
|
// `codex exec --help` (0.149.1) for the flags. `-c web_search=<mode>` is a TOML override, and
|
|
20
14
|
// the binary's own error names the modes: "unknown variant `bogus`, expected one of `disabled`,
|
|
21
15
|
// `cached`, `indexed`, `live`". `live` is the grounded mode research asks for and `disabled` is
|
|
@@ -115,7 +109,7 @@ export function codexLlm(deps) {
|
|
|
115
109
|
id: "codex",
|
|
116
110
|
// Prose arrives as whole messages; other JSONL events carry activity separately.
|
|
117
111
|
capabilities: { streams: true, reportsUsage: true, webSearch: true },
|
|
118
|
-
models: () => Promise.resolve(codexModels),
|
|
112
|
+
models: deps.readModels ?? (() => Promise.resolve(codexModels)),
|
|
119
113
|
complete,
|
|
120
114
|
};
|
|
121
115
|
}
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
import { constants } from "node:fs";
|
|
2
|
+
import { access, realpath } from "node:fs/promises";
|
|
3
|
+
import { delimiter, dirname, isAbsolute, join } from "node:path";
|
|
4
|
+
import { cliCommand } from "../../kernel/cli-command.js";
|
|
5
|
+
import { readCatalogueFile } from "./catalogue-files.js";
|
|
6
|
+
// Official CLI aliases follow the installed CLI's model routing and account.
|
|
7
|
+
// https://geminicli.com/docs/cli/model/
|
|
8
|
+
export const geminiModels = [
|
|
9
|
+
{ id: "auto", name: "Gemini Auto (CLI default)" },
|
|
10
|
+
{ id: "pro", name: "Gemini Pro (CLI alias)" },
|
|
11
|
+
{ id: "flash", name: "Gemini Flash (CLI alias)" },
|
|
12
|
+
{ id: "flash-lite", name: "Gemini Flash-Lite (CLI alias)" },
|
|
13
|
+
];
|
|
14
|
+
export async function nodeGeminiModels(binary) {
|
|
15
|
+
try {
|
|
16
|
+
const command = cliCommand(binary);
|
|
17
|
+
const entry = await executablePath(command.args[0] ?? command.file);
|
|
18
|
+
const candidates = new Set();
|
|
19
|
+
let directory = dirname(entry);
|
|
20
|
+
// npm and bun can install the core package nested under gemini-cli or
|
|
21
|
+
// hoisted beside it. Resolve the configured executable's symlinks first.
|
|
22
|
+
for (let depth = 0; depth < 10; depth += 1) {
|
|
23
|
+
const suffix = "gemini-cli-core/dist/src/config/models.js";
|
|
24
|
+
candidates.add(join(directory, "node_modules/@google", suffix));
|
|
25
|
+
candidates.add(join(directory, suffix));
|
|
26
|
+
const parent = dirname(directory);
|
|
27
|
+
if (parent === directory)
|
|
28
|
+
break;
|
|
29
|
+
directory = parent;
|
|
30
|
+
}
|
|
31
|
+
for (const candidate of candidates) {
|
|
32
|
+
try {
|
|
33
|
+
const models = parseGeminiModels(await readCatalogueFile(candidate, 256 * 1024));
|
|
34
|
+
if (models.length > geminiModels.length)
|
|
35
|
+
return models;
|
|
36
|
+
}
|
|
37
|
+
catch {
|
|
38
|
+
// A candidate is an optional package layout, not the final discovery result.
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
throw new Error("No installed model metadata");
|
|
42
|
+
}
|
|
43
|
+
catch {
|
|
44
|
+
throw new Error("Gemini CLI model metadata is unavailable. Use a documented CLI alias or enter a custom model ID.");
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
function parseGeminiModels(source) {
|
|
48
|
+
const models = new Map(geminiModels.map((model) => [model.id, model]));
|
|
49
|
+
// Read literal exported model constants only. Never import or execute an
|
|
50
|
+
// installed package, parse comments as entries, or inspect login/settings.
|
|
51
|
+
const declarations = /^export (?:const|let) ([A-Z][A-Z0-9_]*)\s*=\s*(['"])([^'"\r\n]+)\2\s*;/gm;
|
|
52
|
+
const uncommented = source.replace(/\/\*[\s\S]*?\*\//g, "");
|
|
53
|
+
for (const match of uncommented.matchAll(declarations)) {
|
|
54
|
+
const name = match[1];
|
|
55
|
+
const id = match[3];
|
|
56
|
+
if (name === undefined ||
|
|
57
|
+
id === undefined ||
|
|
58
|
+
!name.includes("MODEL") ||
|
|
59
|
+
name.includes("EMBEDDING") ||
|
|
60
|
+
!/^(?:gemini|gemma)-[a-z0-9.-]+$/.test(id))
|
|
61
|
+
continue;
|
|
62
|
+
if (!/^(?:(?:PREVIEW|DEFAULT|SECONDARY)_GEMINI_|GEMMA_)/.test(name))
|
|
63
|
+
continue;
|
|
64
|
+
models.set(id, { id, name: id });
|
|
65
|
+
}
|
|
66
|
+
return [...models.values()];
|
|
67
|
+
}
|
|
68
|
+
async function executablePath(binary) {
|
|
69
|
+
if (isAbsolute(binary) || /[\\/]/.test(binary))
|
|
70
|
+
return realpath(binary);
|
|
71
|
+
const path = Object.entries(process.env).find(([key]) => key.toLowerCase() === "path")?.[1] ?? "";
|
|
72
|
+
for (const directory of path.split(delimiter).slice(0, 128)) {
|
|
73
|
+
if (directory === "")
|
|
74
|
+
continue;
|
|
75
|
+
const candidate = join(directory.replace(/^"|"$/g, ""), binary);
|
|
76
|
+
try {
|
|
77
|
+
await access(candidate, constants.X_OK);
|
|
78
|
+
return await realpath(candidate);
|
|
79
|
+
}
|
|
80
|
+
catch {
|
|
81
|
+
// Continue normal PATH lookup; no process is launched for discovery.
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
throw new Error("Gemini executable was not found");
|
|
85
|
+
}
|
|
@@ -1,15 +1,12 @@
|
|
|
1
1
|
import { z } from "zod";
|
|
2
2
|
import { redact } from "../../kernel/log.js";
|
|
3
3
|
import { providerError } from "../../kernel/ports/model.js";
|
|
4
|
+
import { geminiModels } from "./gemini-models.js";
|
|
4
5
|
import { geminiWorkspace } from "./gemini-workspace.js";
|
|
5
6
|
import { cliEvent, cliShaped, endedWithout, promptOf } from "./run-cli.js";
|
|
6
7
|
import { lines } from "./sse-lines.js";
|
|
7
8
|
export const geminiBinary = "gemini";
|
|
8
|
-
export
|
|
9
|
-
{ id: "gemini-2.5-pro", name: "Gemini 2.5 Pro" },
|
|
10
|
-
{ id: "gemini-2.5-flash", name: "Gemini 2.5 Flash" },
|
|
11
|
-
{ id: "gemini-2.5-flash-lite", name: "Gemini 2.5 Flash-Lite" },
|
|
12
|
-
];
|
|
9
|
+
export { geminiModels } from "./gemini-models.js";
|
|
13
10
|
// Verified against installed Gemini CLI 0.16.0's config/nonInteractiveCli sources.
|
|
14
11
|
// Its --allowed-tools flag bypasses approval; tools.core in the isolated settings
|
|
15
12
|
// actually restricts discovery. Never request yolo/auto_edit.
|
|
@@ -103,7 +100,7 @@ export function geminiLlm(deps) {
|
|
|
103
100
|
return {
|
|
104
101
|
id: "gemini",
|
|
105
102
|
capabilities: { streams: true, reportsUsage: true, webSearch: true },
|
|
106
|
-
models: () => Promise.resolve(geminiModels),
|
|
103
|
+
models: deps.readModels ?? (() => Promise.resolve(geminiModels)),
|
|
107
104
|
complete,
|
|
108
105
|
};
|
|
109
106
|
}
|
|
@@ -14,7 +14,11 @@ const appHeaders = {
|
|
|
14
14
|
// A wire payload is narrowed, never cast: everything unlisted is dropped at the seam so
|
|
15
15
|
// no vendor shape can leak past this file.
|
|
16
16
|
const modelList = z.object({
|
|
17
|
-
data: z.array(z.object({
|
|
17
|
+
data: z.array(z.object({
|
|
18
|
+
id: z.string(),
|
|
19
|
+
name: z.string().optional(),
|
|
20
|
+
architecture: z.object({ output_modalities: z.array(z.string()).optional() }).optional(),
|
|
21
|
+
})),
|
|
18
22
|
});
|
|
19
23
|
const errorBody = z.object({
|
|
20
24
|
error: z.object({ message: z.string(), code: z.union([z.number(), z.string()]).optional() }),
|
|
@@ -112,6 +116,7 @@ export function openRouterLlm(deps) {
|
|
|
112
116
|
capabilities: { streams: true, reportsUsage: true, webSearch: true },
|
|
113
117
|
models: async () => {
|
|
114
118
|
const response = await deps.fetch(`${openRouterBase}/models`, {
|
|
119
|
+
signal: AbortSignal.timeout(10_000),
|
|
115
120
|
headers: headers(deps.key()),
|
|
116
121
|
});
|
|
117
122
|
if (!response.ok) {
|
|
@@ -124,7 +129,11 @@ export function openRouterLlm(deps) {
|
|
|
124
129
|
message: "OpenRouter's model list was not in the shape this app can read",
|
|
125
130
|
});
|
|
126
131
|
}
|
|
127
|
-
|
|
132
|
+
// /models defaults to text output. Also reject an explicit non-text row
|
|
133
|
+
// if a gateway response includes one; Slopify sends chat completions here.
|
|
134
|
+
return parsed.data.data
|
|
135
|
+
.filter((model) => model.architecture?.output_modalities?.includes("text") !== false)
|
|
136
|
+
.map((model) => ({ id: model.id, name: model.name ?? model.id }));
|
|
128
137
|
},
|
|
129
138
|
complete,
|
|
130
139
|
};
|
|
@@ -11,6 +11,13 @@ export const cartesiaBase = "https://api.cartesia.ai";
|
|
|
11
11
|
// constant rather than left to the account's default.
|
|
12
12
|
export const cartesiaVersion = "2026-03-01";
|
|
13
13
|
export const cartesiaModel = "sonic-3.5";
|
|
14
|
+
// Cartesia publishes model IDs in its docs, with no model-list API. Stable family aliases
|
|
15
|
+
// receive new snapshots automatically: https://docs.cartesia.ai/build-with-cartesia/tts-models/latest
|
|
16
|
+
export const cartesiaModels = [
|
|
17
|
+
{ id: "sonic-3.6", name: "Sonic 3.6" },
|
|
18
|
+
{ id: "sonic-3.5", name: "Sonic 3.5" },
|
|
19
|
+
{ id: "sonic-3", name: "Sonic 3" },
|
|
20
|
+
];
|
|
14
21
|
// mp3 is the port's container; `bit_rate` is required for it and `sample_rate` fixes the
|
|
15
22
|
// rate the concatenation then keeps.
|
|
16
23
|
const outputFormat = { container: "mp3", bit_rate: 128_000, sample_rate: 44_100 };
|
|
@@ -24,6 +31,7 @@ export function cartesiaTts(deps) {
|
|
|
24
31
|
return {
|
|
25
32
|
id: "cartesia",
|
|
26
33
|
capabilities: { streams: true },
|
|
34
|
+
models: async () => cartesiaModels,
|
|
27
35
|
synthesize: async (req) => {
|
|
28
36
|
const response = await deps.fetch(`${cartesiaBase}/tts/bytes`, {
|
|
29
37
|
method: "POST",
|
|
@@ -36,7 +44,7 @@ export function cartesiaTts(deps) {
|
|
|
36
44
|
// No pre-check on length; Cartesia's own limit surfaces as its
|
|
37
45
|
// error. `language` is left out so the model reads it off the transcript.
|
|
38
46
|
body: JSON.stringify({
|
|
39
|
-
model_id: cartesiaModel,
|
|
47
|
+
model_id: req.model ?? cartesiaModel,
|
|
40
48
|
transcript: req.text,
|
|
41
49
|
voice: { mode: "id", id: req.voiceId },
|
|
42
50
|
output_format: outputFormat,
|
|
@@ -10,9 +10,22 @@ export const elevenLabsBase = "https://api.elevenlabs.io/v1";
|
|
|
10
10
|
// attempt wrapper measures its 120 s as an idle timeout between chunks only when bytes
|
|
11
11
|
// keep arriving.
|
|
12
12
|
export const elevenLabsModel = "eleven_multilingual_v2";
|
|
13
|
+
export const elevenLabsModels = [
|
|
14
|
+
{ id: "eleven_v3", name: "Eleven v3" },
|
|
15
|
+
{ id: "eleven_multilingual_v2", name: "Eleven Multilingual v2" },
|
|
16
|
+
{ id: "eleven_flash_v2_5", name: "Eleven Flash v2.5" },
|
|
17
|
+
{ id: "eleven_flash_v2", name: "Eleven Flash v2" },
|
|
18
|
+
{ id: "eleven_turbo_v2_5", name: "Eleven Turbo v2.5 (deprecated)" },
|
|
19
|
+
{ id: "eleven_turbo_v2", name: "Eleven Turbo v2 (deprecated)" },
|
|
20
|
+
];
|
|
13
21
|
// mp3 at the port's container, 44.1 kHz, 128 kbps: `kernel/ports/tts.ts` fixes mp3 and the
|
|
14
22
|
// concatenation keeps the provider's own sample rate.
|
|
15
23
|
export const elevenLabsFormat = "mp3_44100_128";
|
|
24
|
+
const modelList = z.array(z.object({
|
|
25
|
+
model_id: z.string().min(1),
|
|
26
|
+
name: z.string().optional(),
|
|
27
|
+
can_do_text_to_speech: z.boolean().optional(),
|
|
28
|
+
}));
|
|
16
29
|
// A wire payload is narrowed, never cast. `detail` is an object on a handled failure and
|
|
17
30
|
// a string on the framework's own; anything else falls back to the raw text.
|
|
18
31
|
const errorBody = z.object({
|
|
@@ -25,6 +38,24 @@ export function elevenLabsTts(deps) {
|
|
|
25
38
|
return {
|
|
26
39
|
id: "elevenlabs",
|
|
27
40
|
capabilities: { streams: true },
|
|
41
|
+
models: async () => {
|
|
42
|
+
const response = await deps.fetch(`${elevenLabsBase}/models`, {
|
|
43
|
+
headers: { "xi-api-key": keyOf(deps) },
|
|
44
|
+
signal: AbortSignal.timeout(10_000),
|
|
45
|
+
});
|
|
46
|
+
if (!response.ok)
|
|
47
|
+
throw await failure(response);
|
|
48
|
+
const parsed = modelList.safeParse(safeJson(await response.text()));
|
|
49
|
+
if (!parsed.success) {
|
|
50
|
+
throw providerError({
|
|
51
|
+
kind: "other",
|
|
52
|
+
message: "ElevenLabs' model list was not in the shape this app can read",
|
|
53
|
+
});
|
|
54
|
+
}
|
|
55
|
+
return parsed.data
|
|
56
|
+
.filter((model) => model.can_do_text_to_speech === true)
|
|
57
|
+
.map((model) => ({ id: model.model_id, name: model.name || model.model_id }));
|
|
58
|
+
},
|
|
28
59
|
synthesize: async (req) => {
|
|
29
60
|
const voice = encodeURIComponent(req.voiceId);
|
|
30
61
|
const response = await deps.fetch(`${elevenLabsBase}/text-to-speech/${voice}/stream?output_format=${elevenLabsFormat}`, {
|
|
@@ -33,7 +64,7 @@ export function elevenLabsTts(deps) {
|
|
|
33
64
|
headers: { "xi-api-key": keyOf(deps), "Content-Type": "application/json" },
|
|
34
65
|
// No pre-check on length. A text past the model's limit comes
|
|
35
66
|
// back as the provider's own 400 and that is what the stage shows.
|
|
36
|
-
body: JSON.stringify({ text: req.text, model_id: elevenLabsModel }),
|
|
67
|
+
body: JSON.stringify({ text: req.text, model_id: req.model ?? elevenLabsModel }),
|
|
37
68
|
});
|
|
38
69
|
if (!response.ok) {
|
|
39
70
|
throw await failure(response, req.voiceId);
|
|
@@ -75,7 +106,7 @@ async function failure(response, voiceId) {
|
|
|
75
106
|
kind: kindOf(response.status),
|
|
76
107
|
// A rejected voice ID has to be named, and it is the one part of the
|
|
77
108
|
// request the user chose. It is not secret, unlike everything else on the wire.
|
|
78
|
-
message: `ElevenLabs answered ${response.status} for voice ${voiceId}: ${message}`,
|
|
109
|
+
message: `ElevenLabs answered ${response.status}${voiceId === undefined ? "" : ` for voice ${voiceId}`}: ${message}`,
|
|
79
110
|
...(retryAfterMs === undefined ? {} : { retryAfterMs }),
|
|
80
111
|
});
|
|
81
112
|
}
|