@gentbajko/slopify 0.6.1 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +36 -1
- package/SUBTITLES.md +3 -3
- package/dist/adapter-registry.js +12 -2
- package/dist/adapters/image/google.js +12 -15
- package/dist/adapters/image/models.js +72 -0
- package/dist/adapters/image/openai.js +3 -5
- package/dist/adapters/llm/catalogue-files.js +23 -0
- package/dist/adapters/llm/claude-code.js +3 -4
- package/dist/adapters/llm/codex-models.js +46 -0
- package/dist/adapters/llm/codex.js +7 -10
- package/dist/adapters/llm/gemini-models.js +85 -0
- package/dist/adapters/llm/gemini-workspace.js +24 -1
- package/dist/adapters/llm/gemini.js +4 -7
- package/dist/adapters/llm/openrouter.js +12 -2
- package/dist/adapters/tts/cartesia.js +9 -1
- package/dist/adapters/tts/elevenlabs.js +33 -2
- package/dist/adapters/tts/inworld-async.js +150 -0
- package/dist/adapters/tts/inworld-text.js +32 -0
- package/dist/adapters/tts/inworld.js +170 -0
- package/dist/adapters/tts/openai.js +31 -4
- package/dist/assets/models.yaml +470 -0
- package/dist/catalog/registry.js +147 -0
- package/dist/catalog/schema.js +91 -0
- package/dist/catalog/store.js +90 -0
- package/dist/catalog/validate.js +43 -0
- package/dist/edge/cli.js +5 -0
- package/dist/edge/events/hub.js +9 -2
- package/dist/edge/events/preview-cache.js +40 -0
- package/dist/edge/http/actions.js +2 -0
- package/dist/edge/http/app.js +28 -1
- package/dist/edge/http/audio-preview.js +46 -0
- package/dist/edge/http/planning.js +101 -0
- package/dist/edge/http/projects.js +4 -2
- package/dist/edge/http/providers.js +44 -0
- package/dist/edge/http/update.js +57 -0
- package/dist/edge/update-worker.js +44 -0
- package/dist/kernel/audio-preview.js +188 -0
- package/dist/kernel/db/migrations/0003-batch-queue.sql +11 -0
- package/dist/kernel/ports/llm.js +1 -0
- package/dist/kernel/ports/text.js +26 -0
- package/dist/kernel/runner/providers.js +80 -28
- package/dist/kernel/runner/queue.js +65 -0
- package/dist/main.js +142 -27
- package/dist/model-catalog.js +37 -0
- package/dist/slices/admission/repo.js +6 -1
- package/dist/slices/admission/start.js +9 -8
- package/dist/slices/article/continuation.js +8 -1
- package/dist/slices/article/segments.js +1 -0
- package/dist/slices/batch/index.js +82 -0
- package/dist/slices/control/index.js +6 -1
- package/dist/slices/control/providers.js +16 -1
- package/dist/slices/estimate/index.js +81 -0
- package/dist/slices/narration/live.js +21 -0
- package/dist/slices/narration/plan.js +58 -0
- package/dist/slices/narration/run.js +15 -36
- package/dist/slices/research/run.js +7 -0
- package/dist/slices/settings/model.js +2 -0
- package/dist/slices/settings/models.js +89 -0
- package/dist/slices/storage/staging.js +2 -1
- package/dist/slices/subtitles/captions.js +4 -1
- package/dist/slices/subtitles/layout.js +19 -0
- package/dist/slices/subtitles/model.js +9 -0
- package/dist/slices/subtitles/prepare.js +6 -1
- package/dist/slices/thumbnail/run.js +3 -0
- package/dist/updater/candidate.js +28 -0
- package/dist/updater/forward.js +35 -0
- package/dist/updater/install-flow.js +26 -0
- package/dist/updater/install.js +50 -0
- package/dist/updater/model.js +24 -0
- package/dist/updater/plan.js +135 -0
- package/dist/updater/readiness.js +15 -0
- package/dist/updater/registry.js +19 -0
- package/dist/updater/service.js +128 -0
- package/dist/updater/worker.js +136 -0
- package/dist/web/assets/index-6zz8telY.css +1 -0
- package/dist/web/assets/index-CbYEcBOa.js +130 -0
- package/dist/web/index.html +2 -2
- package/package.json +2 -1
- package/dist/web/assets/index-CNYs0noB.js +0 -130
- package/dist/web/assets/index-Mk0bvBg-.css +0 -1
|
@@ -10,9 +10,22 @@ export const elevenLabsBase = "https://api.elevenlabs.io/v1";
|
|
|
10
10
|
// attempt wrapper measures its 120 s as an idle timeout between chunks only when bytes
|
|
11
11
|
// keep arriving.
|
|
12
12
|
export const elevenLabsModel = "eleven_multilingual_v2";
|
|
13
|
+
export const elevenLabsModels = [
|
|
14
|
+
{ id: "eleven_v3", name: "Eleven v3" },
|
|
15
|
+
{ id: "eleven_multilingual_v2", name: "Eleven Multilingual v2" },
|
|
16
|
+
{ id: "eleven_flash_v2_5", name: "Eleven Flash v2.5" },
|
|
17
|
+
{ id: "eleven_flash_v2", name: "Eleven Flash v2" },
|
|
18
|
+
{ id: "eleven_turbo_v2_5", name: "Eleven Turbo v2.5 (deprecated)" },
|
|
19
|
+
{ id: "eleven_turbo_v2", name: "Eleven Turbo v2 (deprecated)" },
|
|
20
|
+
];
|
|
13
21
|
// mp3 at the port's container, 44.1 kHz, 128 kbps: `kernel/ports/tts.ts` fixes mp3 and the
|
|
14
22
|
// concatenation keeps the provider's own sample rate.
|
|
15
23
|
export const elevenLabsFormat = "mp3_44100_128";
|
|
24
|
+
const modelList = z.array(z.object({
|
|
25
|
+
model_id: z.string().min(1),
|
|
26
|
+
name: z.string().optional(),
|
|
27
|
+
can_do_text_to_speech: z.boolean().optional(),
|
|
28
|
+
}));
|
|
16
29
|
// A wire payload is narrowed, never cast. `detail` is an object on a handled failure and
|
|
17
30
|
// a string on the framework's own; anything else falls back to the raw text.
|
|
18
31
|
const errorBody = z.object({
|
|
@@ -25,6 +38,24 @@ export function elevenLabsTts(deps) {
|
|
|
25
38
|
return {
|
|
26
39
|
id: "elevenlabs",
|
|
27
40
|
capabilities: { streams: true },
|
|
41
|
+
models: async () => {
|
|
42
|
+
const response = await deps.fetch(`${elevenLabsBase}/models`, {
|
|
43
|
+
headers: { "xi-api-key": keyOf(deps) },
|
|
44
|
+
signal: AbortSignal.timeout(10_000),
|
|
45
|
+
});
|
|
46
|
+
if (!response.ok)
|
|
47
|
+
throw await failure(response);
|
|
48
|
+
const parsed = modelList.safeParse(safeJson(await response.text()));
|
|
49
|
+
if (!parsed.success) {
|
|
50
|
+
throw providerError({
|
|
51
|
+
kind: "other",
|
|
52
|
+
message: "ElevenLabs' model list was not in the shape this app can read",
|
|
53
|
+
});
|
|
54
|
+
}
|
|
55
|
+
return parsed.data
|
|
56
|
+
.filter((model) => model.can_do_text_to_speech === true)
|
|
57
|
+
.map((model) => ({ id: model.model_id, name: model.name || model.model_id }));
|
|
58
|
+
},
|
|
28
59
|
synthesize: async (req) => {
|
|
29
60
|
const voice = encodeURIComponent(req.voiceId);
|
|
30
61
|
const response = await deps.fetch(`${elevenLabsBase}/text-to-speech/${voice}/stream?output_format=${elevenLabsFormat}`, {
|
|
@@ -33,7 +64,7 @@ export function elevenLabsTts(deps) {
|
|
|
33
64
|
headers: { "xi-api-key": keyOf(deps), "Content-Type": "application/json" },
|
|
34
65
|
// No pre-check on length. A text past the model's limit comes
|
|
35
66
|
// back as the provider's own 400 and that is what the stage shows.
|
|
36
|
-
body: JSON.stringify({ text: req.text, model_id: elevenLabsModel }),
|
|
67
|
+
body: JSON.stringify({ text: req.text, model_id: req.model ?? elevenLabsModel }),
|
|
37
68
|
});
|
|
38
69
|
if (!response.ok) {
|
|
39
70
|
throw await failure(response, req.voiceId);
|
|
@@ -75,7 +106,7 @@ async function failure(response, voiceId) {
|
|
|
75
106
|
kind: kindOf(response.status),
|
|
76
107
|
// A rejected voice ID has to be named, and it is the one part of the
|
|
77
108
|
// request the user chose. It is not secret, unlike everything else on the wire.
|
|
78
|
-
message: `ElevenLabs answered ${response.status} for voice ${voiceId}: ${message}`,
|
|
109
|
+
message: `ElevenLabs answered ${response.status}${voiceId === undefined ? "" : ` for voice ${voiceId}`}: ${message}`,
|
|
79
110
|
...(retryAfterMs === undefined ? {} : { retryAfterMs }),
|
|
80
111
|
});
|
|
81
112
|
}
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
import { redact } from "../../kernel/log.js";
|
|
3
|
+
import { providerError } from "../../kernel/ports/model.js";
|
|
4
|
+
import { retryAfter } from "../retry-after.js";
|
|
5
|
+
const operationName = /^workspaces\/[A-Za-z0-9_-]+\/ttsAsyncJobs\/[A-Za-z0-9_-]+\/operations\/[A-Za-z0-9_-]+$/;
|
|
6
|
+
const operationSchema = z.object({
|
|
7
|
+
name: z.string().regex(operationName),
|
|
8
|
+
done: z.boolean().optional(),
|
|
9
|
+
error: z.object({ code: z.number().optional(), message: z.string().optional() }).nullish(),
|
|
10
|
+
response: z.object({ audioUri: z.string() }).nullish(),
|
|
11
|
+
});
|
|
12
|
+
// Async is documented for TTS-2 only. Flash continues using the streaming API.
|
|
13
|
+
// A continuation survives automatic retries of this call; pausing/restarting a
|
|
14
|
+
// stage stops local polling but cannot cancel an already accepted remote job.
|
|
15
|
+
export async function* inworldAsync(deps, request, key) {
|
|
16
|
+
if (request.text.length > 100_000) {
|
|
17
|
+
throw providerError({
|
|
18
|
+
kind: "unsupported",
|
|
19
|
+
message: "Inworld async accepts up to 100,000 characters per request (10,000 for On-Demand accounts). Select paragraph chunking for longer articles.",
|
|
20
|
+
});
|
|
21
|
+
}
|
|
22
|
+
const headers = { Authorization: `Basic ${key}`, "Content-Type": "application/json" };
|
|
23
|
+
const saved = request.continuation?.read();
|
|
24
|
+
if (saved && !operationName.test(saved))
|
|
25
|
+
throw invalid("returned an invalid operation name");
|
|
26
|
+
request.signal.throwIfAborted();
|
|
27
|
+
let response = await deps.fetch(saved
|
|
28
|
+
? `https://api.inworld.ai/lro/v1alpha/${saved}`
|
|
29
|
+
: "https://api.inworld.ai/tts/v1/voice:synthesizeAsync", {
|
|
30
|
+
method: saved ? "GET" : "POST",
|
|
31
|
+
headers,
|
|
32
|
+
signal: request.signal,
|
|
33
|
+
redirect: "error",
|
|
34
|
+
...(saved
|
|
35
|
+
? {}
|
|
36
|
+
: {
|
|
37
|
+
body: JSON.stringify({
|
|
38
|
+
text: request.text,
|
|
39
|
+
voiceId: request.voiceId,
|
|
40
|
+
modelId: request.model ?? "inworld-tts-2",
|
|
41
|
+
audioConfig: { audioEncoding: "MP3", sampleRateHertz: 48_000, bitRate: 128_000 },
|
|
42
|
+
}),
|
|
43
|
+
}),
|
|
44
|
+
});
|
|
45
|
+
let expected = saved;
|
|
46
|
+
for (;;) {
|
|
47
|
+
request.signal.throwIfAborted();
|
|
48
|
+
await check(response, key);
|
|
49
|
+
const parsed = operationSchema.safeParse(await response.json().catch(() => null));
|
|
50
|
+
if (!parsed.success || (expected && parsed.data.name !== expected))
|
|
51
|
+
throw invalid("returned an invalid operation");
|
|
52
|
+
const operation = parsed.data;
|
|
53
|
+
expected = operation.name;
|
|
54
|
+
request.continuation?.write(operation.name);
|
|
55
|
+
request.onActivity?.();
|
|
56
|
+
request.signal.throwIfAborted();
|
|
57
|
+
if (operation.done) {
|
|
58
|
+
if (operation.error) {
|
|
59
|
+
const { code, message } = operation.error;
|
|
60
|
+
throw providerError({
|
|
61
|
+
kind: code === 7 || code === 16
|
|
62
|
+
? "auth"
|
|
63
|
+
: code === 8
|
|
64
|
+
? "rate_limit"
|
|
65
|
+
: code === 3 || code === 5 || code === 9
|
|
66
|
+
? "unsupported"
|
|
67
|
+
: "other",
|
|
68
|
+
message: `Inworld async job failed: ${clean(message ?? "audio generation stopped", key)}`,
|
|
69
|
+
});
|
|
70
|
+
}
|
|
71
|
+
if (!operation.response)
|
|
72
|
+
throw invalid("finished without an audio download");
|
|
73
|
+
yield* download(deps.fetch, operation.response.audioUri, request.signal);
|
|
74
|
+
return;
|
|
75
|
+
}
|
|
76
|
+
await deps.clock.sleep(5000, request.signal);
|
|
77
|
+
request.signal.throwIfAborted();
|
|
78
|
+
response = await deps.fetch(`https://api.inworld.ai/lro/v1alpha/${operation.name}`, {
|
|
79
|
+
headers,
|
|
80
|
+
signal: request.signal,
|
|
81
|
+
redirect: "error",
|
|
82
|
+
});
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
async function* download(fetch, uri, signal) {
|
|
86
|
+
const url = URL.parse(uri);
|
|
87
|
+
if (url?.protocol !== "https:" || url.username || url.password)
|
|
88
|
+
throw invalid("returned an invalid audio download URL");
|
|
89
|
+
// This is a signed storage URL, not an Inworld API endpoint. Never forward Basic auth.
|
|
90
|
+
let response;
|
|
91
|
+
try {
|
|
92
|
+
response = await fetch(url.href, { signal, redirect: "error" });
|
|
93
|
+
}
|
|
94
|
+
catch {
|
|
95
|
+
signal.throwIfAborted();
|
|
96
|
+
// Signed query parameters must not become a persisted error or log entry.
|
|
97
|
+
throw invalid("audio download could not be reached; retrying the existing job");
|
|
98
|
+
}
|
|
99
|
+
if (!response.ok)
|
|
100
|
+
throw invalid(`audio download answered ${response.status}; retrying the existing job`);
|
|
101
|
+
if (!response.body)
|
|
102
|
+
throw invalid("answered with no audio");
|
|
103
|
+
const reader = response.body.getReader();
|
|
104
|
+
let heard = false;
|
|
105
|
+
try {
|
|
106
|
+
for (;;) {
|
|
107
|
+
const { value, done } = await reader.read();
|
|
108
|
+
signal.throwIfAborted();
|
|
109
|
+
if (done)
|
|
110
|
+
break;
|
|
111
|
+
if (value.length) {
|
|
112
|
+
heard = true;
|
|
113
|
+
yield value;
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
catch {
|
|
118
|
+
signal.throwIfAborted();
|
|
119
|
+
throw invalid("audio download stopped; retrying the existing job");
|
|
120
|
+
}
|
|
121
|
+
finally {
|
|
122
|
+
await reader.cancel().catch(() => { });
|
|
123
|
+
reader.releaseLock();
|
|
124
|
+
}
|
|
125
|
+
if (!heard)
|
|
126
|
+
throw invalid("answered with no audio");
|
|
127
|
+
}
|
|
128
|
+
async function check(response, key) {
|
|
129
|
+
if (response.ok)
|
|
130
|
+
return;
|
|
131
|
+
const detail = clean(await response.text().catch(() => ""), key);
|
|
132
|
+
const retryAfterMs = retryAfter(response.headers.get("retry-after"));
|
|
133
|
+
throw providerError({
|
|
134
|
+
kind: response.status === 401 || response.status === 403
|
|
135
|
+
? "auth"
|
|
136
|
+
: response.status === 429
|
|
137
|
+
? "rate_limit"
|
|
138
|
+
: response.status === 400 || response.status === 404
|
|
139
|
+
? "unsupported"
|
|
140
|
+
: "other",
|
|
141
|
+
message: `Inworld async answered ${response.status}: ${detail || response.statusText}`,
|
|
142
|
+
...(retryAfterMs === undefined ? {} : { retryAfterMs }),
|
|
143
|
+
});
|
|
144
|
+
}
|
|
145
|
+
function clean(message, key) {
|
|
146
|
+
return redact(message.replaceAll(key, "[redacted]"));
|
|
147
|
+
}
|
|
148
|
+
function invalid(detail) {
|
|
149
|
+
return providerError({ kind: "other", message: `Inworld ${detail}.` });
|
|
150
|
+
}
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
// Inworld's streaming endpoint accepts 4,000 characters. Split inside the
|
|
2
|
+
// adapter so whole-text narration and long intros/outros all remain usable.
|
|
3
|
+
// UTF-16 length is conservative for Unicode; never cut a surrogate pair.
|
|
4
|
+
export function inworldTextParts(text) {
|
|
5
|
+
const parts = [];
|
|
6
|
+
let current = "";
|
|
7
|
+
for (const sentence of new Intl.Segmenter("en", { granularity: "sentence" }).segment(text)) {
|
|
8
|
+
if (current.length + sentence.segment.length <= 4000) {
|
|
9
|
+
current += sentence.segment;
|
|
10
|
+
continue;
|
|
11
|
+
}
|
|
12
|
+
if (current.trim())
|
|
13
|
+
parts.push(current.trim());
|
|
14
|
+
let remaining = sentence.segment;
|
|
15
|
+
while (remaining.length > 4000) {
|
|
16
|
+
const window = remaining.slice(0, 4000);
|
|
17
|
+
const boundary = Math.max(window.lastIndexOf(" "), window.lastIndexOf("\n"), window.lastIndexOf("\t"));
|
|
18
|
+
let cut = boundary > 0 ? boundary : 4000;
|
|
19
|
+
const last = remaining.charCodeAt(cut - 1);
|
|
20
|
+
if (last >= 0xd800 && last <= 0xdbff)
|
|
21
|
+
cut--;
|
|
22
|
+
const piece = remaining.slice(0, cut).trim();
|
|
23
|
+
if (piece)
|
|
24
|
+
parts.push(piece);
|
|
25
|
+
remaining = remaining.slice(cut).trimStart();
|
|
26
|
+
}
|
|
27
|
+
current = remaining;
|
|
28
|
+
}
|
|
29
|
+
if (current.trim())
|
|
30
|
+
parts.push(current.trim());
|
|
31
|
+
return parts;
|
|
32
|
+
}
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
import { redact } from "../../kernel/log.js";
|
|
3
|
+
import { providerError } from "../../kernel/ports/model.js";
|
|
4
|
+
import { retryAfter } from "../retry-after.js";
|
|
5
|
+
import { inworldAsync } from "./inworld-async.js";
|
|
6
|
+
import { inworldTextParts } from "./inworld-text.js";
|
|
7
|
+
export const inworldModels = [
|
|
8
|
+
{ id: "inworld-tts-2", name: "Realtime TTS-2" },
|
|
9
|
+
{ id: "inworld-tts-2-flash", name: "Realtime TTS-2 Flash" },
|
|
10
|
+
];
|
|
11
|
+
const envelope = z.object({
|
|
12
|
+
result: z.object({ audioContent: z.string().optional() }).optional(),
|
|
13
|
+
error: z.object({ code: z.number().optional(), message: z.string().optional() }).optional(),
|
|
14
|
+
});
|
|
15
|
+
export function inworldTts(deps) {
|
|
16
|
+
return {
|
|
17
|
+
id: "inworld",
|
|
18
|
+
capabilities: { streams: true },
|
|
19
|
+
models: async () => inworldModels,
|
|
20
|
+
synthesize: async (request) => {
|
|
21
|
+
const key = deps
|
|
22
|
+
.key()
|
|
23
|
+
?.trim()
|
|
24
|
+
.replace(/^Basic\s+/i, "");
|
|
25
|
+
if (!key)
|
|
26
|
+
throw providerError({ kind: "missing_key", message: "No Inworld API key is stored." });
|
|
27
|
+
const cancel = new AbortController();
|
|
28
|
+
const signal = AbortSignal.any([request.signal, cancel.signal]);
|
|
29
|
+
const iterator = synthesizeParts(deps, { ...request, signal }, key);
|
|
30
|
+
return {
|
|
31
|
+
container: "mp3",
|
|
32
|
+
audio: new ReadableStream({
|
|
33
|
+
async pull(controller) {
|
|
34
|
+
try {
|
|
35
|
+
const next = await iterator.next();
|
|
36
|
+
if (next.done)
|
|
37
|
+
controller.close();
|
|
38
|
+
else
|
|
39
|
+
controller.enqueue(next.value);
|
|
40
|
+
}
|
|
41
|
+
catch (error) {
|
|
42
|
+
controller.error(error);
|
|
43
|
+
}
|
|
44
|
+
},
|
|
45
|
+
async cancel() {
|
|
46
|
+
cancel.abort();
|
|
47
|
+
await iterator.return();
|
|
48
|
+
},
|
|
49
|
+
}, { highWaterMark: 0 }),
|
|
50
|
+
};
|
|
51
|
+
},
|
|
52
|
+
};
|
|
53
|
+
}
|
|
54
|
+
async function* synthesizeParts(deps, request, key) {
|
|
55
|
+
if ((request.model ?? "inworld-tts-2") === "inworld-tts-2" && request.text.length > 4000) {
|
|
56
|
+
yield* inworldAsync(deps, request, key);
|
|
57
|
+
return;
|
|
58
|
+
}
|
|
59
|
+
const parts = inworldTextParts(request.text);
|
|
60
|
+
if (parts.length === 0)
|
|
61
|
+
throw providerError({ kind: "unsupported", message: "Inworld needs text to narrate." });
|
|
62
|
+
for (const text of parts) {
|
|
63
|
+
request.signal.throwIfAborted();
|
|
64
|
+
const response = await deps.fetch("https://api.inworld.ai/tts/v1/voice:stream", {
|
|
65
|
+
method: "POST",
|
|
66
|
+
redirect: "error",
|
|
67
|
+
signal: request.signal,
|
|
68
|
+
headers: { Authorization: `Basic ${key}`, "Content-Type": "application/json" },
|
|
69
|
+
body: JSON.stringify({
|
|
70
|
+
text,
|
|
71
|
+
voiceId: request.voiceId,
|
|
72
|
+
modelId: request.model ?? "inworld-tts-2",
|
|
73
|
+
audioConfig: { audioEncoding: "MP3", sampleRateHertz: 48_000, bitRate: 128_000 },
|
|
74
|
+
}),
|
|
75
|
+
});
|
|
76
|
+
if (!response.ok) {
|
|
77
|
+
const detail = clean(await response.text().catch(() => ""), key);
|
|
78
|
+
const retryAfterMs = retryAfter(response.headers.get("retry-after"));
|
|
79
|
+
throw providerError({
|
|
80
|
+
kind: response.status === 401 || response.status === 403
|
|
81
|
+
? "auth"
|
|
82
|
+
: response.status === 429
|
|
83
|
+
? "rate_limit"
|
|
84
|
+
: response.status === 400 || response.status === 404
|
|
85
|
+
? "unsupported"
|
|
86
|
+
: "other",
|
|
87
|
+
message: clean(`Inworld answered ${response.status} for voice ${request.voiceId}: ${detail || response.statusText}`, key),
|
|
88
|
+
...(retryAfterMs === undefined ? {} : { retryAfterMs }),
|
|
89
|
+
});
|
|
90
|
+
}
|
|
91
|
+
if (response.body === null)
|
|
92
|
+
throw invalid("answered with no audio");
|
|
93
|
+
let heard = false;
|
|
94
|
+
for await (const line of lines(response.body)) {
|
|
95
|
+
request.signal.throwIfAborted();
|
|
96
|
+
let raw;
|
|
97
|
+
try {
|
|
98
|
+
raw = JSON.parse(line);
|
|
99
|
+
}
|
|
100
|
+
catch {
|
|
101
|
+
throw invalid("sent an unreadable audio chunk");
|
|
102
|
+
}
|
|
103
|
+
const parsed = envelope.safeParse(raw);
|
|
104
|
+
if (!parsed.success || (!parsed.data.result && !parsed.data.error))
|
|
105
|
+
throw invalid("sent an unreadable audio chunk");
|
|
106
|
+
if (parsed.data.error) {
|
|
107
|
+
const { code, message } = parsed.data.error;
|
|
108
|
+
throw providerError({
|
|
109
|
+
kind: code === 16 || code === 7
|
|
110
|
+
? "auth"
|
|
111
|
+
: code === 8
|
|
112
|
+
? "rate_limit"
|
|
113
|
+
: code === 3 || code === 5
|
|
114
|
+
? "unsupported"
|
|
115
|
+
: "other",
|
|
116
|
+
message: `Inworld stream failed: ${clean(message ?? "audio generation stopped", key)}`,
|
|
117
|
+
});
|
|
118
|
+
}
|
|
119
|
+
const encoded = parsed.data.result?.audioContent;
|
|
120
|
+
if (!encoded)
|
|
121
|
+
continue;
|
|
122
|
+
if (!/^[A-Za-z0-9+/]*={0,2}$/.test(encoded) || encoded.length % 4 === 1)
|
|
123
|
+
throw invalid("sent invalid encoded audio");
|
|
124
|
+
const audio = Buffer.from(encoded, "base64");
|
|
125
|
+
if (audio.length === 0)
|
|
126
|
+
continue;
|
|
127
|
+
heard = true;
|
|
128
|
+
yield audio;
|
|
129
|
+
}
|
|
130
|
+
if (!heard)
|
|
131
|
+
throw invalid("answered with no audio");
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
async function* lines(body) {
|
|
135
|
+
const reader = body.getReader();
|
|
136
|
+
const decoder = new TextDecoder();
|
|
137
|
+
let pending = "";
|
|
138
|
+
try {
|
|
139
|
+
for (;;) {
|
|
140
|
+
const { done, value } = await reader.read();
|
|
141
|
+
pending += done ? decoder.decode() : decoder.decode(value, { stream: true });
|
|
142
|
+
let end = pending.indexOf("\n");
|
|
143
|
+
while (end !== -1) {
|
|
144
|
+
if (end > 8 * 1024 * 1024)
|
|
145
|
+
throw invalid("sent an oversized audio chunk");
|
|
146
|
+
const line = pending.slice(0, end).trim();
|
|
147
|
+
pending = pending.slice(end + 1);
|
|
148
|
+
if (line)
|
|
149
|
+
yield line;
|
|
150
|
+
end = pending.indexOf("\n");
|
|
151
|
+
}
|
|
152
|
+
if (pending.length > 8 * 1024 * 1024)
|
|
153
|
+
throw invalid("sent an oversized audio chunk");
|
|
154
|
+
if (done)
|
|
155
|
+
break;
|
|
156
|
+
}
|
|
157
|
+
if (pending.trim())
|
|
158
|
+
yield pending.trim();
|
|
159
|
+
}
|
|
160
|
+
finally {
|
|
161
|
+
await reader.cancel().catch(() => { });
|
|
162
|
+
reader.releaseLock();
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
function clean(message, key) {
|
|
166
|
+
return redact(message.replaceAll(key, "[redacted]"));
|
|
167
|
+
}
|
|
168
|
+
function invalid(detail) {
|
|
169
|
+
return providerError({ kind: "other", message: `Inworld ${detail}.` });
|
|
170
|
+
}
|
|
@@ -5,9 +5,18 @@ import { retryAfter } from "../retry-after.js";
|
|
|
5
5
|
// The HTTP gateway adapter for OpenAI's speech endpoint. `fetch` and nothing else; the response
|
|
6
6
|
// body is already the stream the port asks for.
|
|
7
7
|
export const openAiAudioBase = "https://api.openai.com/v1";
|
|
8
|
-
//
|
|
9
|
-
// three per character.
|
|
8
|
+
// Older callers can omit a model; saved project choices always supply one.
|
|
10
9
|
export const openAiTtsModel = "gpt-4o-mini-tts";
|
|
10
|
+
export const openAiTtsModels = [
|
|
11
|
+
{ id: "gpt-4o-mini-tts", name: "GPT-4o mini TTS" },
|
|
12
|
+
{ id: "gpt-4o-mini-tts-2025-12-15", name: "GPT-4o mini TTS (2025-12-15)" },
|
|
13
|
+
{ id: "tts-1", name: "TTS-1" },
|
|
14
|
+
{ id: "tts-1-hd", name: "TTS-1 HD" },
|
|
15
|
+
];
|
|
16
|
+
const modelList = z.object({ data: z.array(z.object({ id: z.string().min(1) })) });
|
|
17
|
+
// /models has no capability metadata. Only the speech endpoint's documented model families
|
|
18
|
+
// and dated snapshots belong here; realtime and transcription models use different APIs.
|
|
19
|
+
const speechModel = /^(?:tts-1(?:-hd)?(?:-\d{4})?|gpt-4o-mini-tts(?:-\d{4}-\d{2}-\d{2})?)$/;
|
|
11
20
|
const errorBody = z.object({
|
|
12
21
|
error: z.object({ message: z.string(), code: z.string().nullish() }),
|
|
13
22
|
});
|
|
@@ -19,6 +28,24 @@ export function openAiTts(deps) {
|
|
|
19
28
|
return {
|
|
20
29
|
id: "openai-tts",
|
|
21
30
|
capabilities: { streams: true },
|
|
31
|
+
models: async () => {
|
|
32
|
+
const response = await deps.fetch(`${openAiAudioBase}/models`, {
|
|
33
|
+
headers: { Authorization: `Bearer ${keyOf(deps)}` },
|
|
34
|
+
signal: AbortSignal.timeout(10_000),
|
|
35
|
+
});
|
|
36
|
+
if (!response.ok)
|
|
37
|
+
throw await failure(response);
|
|
38
|
+
const parsed = modelList.safeParse(safeJson(await response.text()));
|
|
39
|
+
if (!parsed.success) {
|
|
40
|
+
throw providerError({
|
|
41
|
+
kind: "other",
|
|
42
|
+
message: "OpenAI's model list was not in the shape this app can read",
|
|
43
|
+
});
|
|
44
|
+
}
|
|
45
|
+
return parsed.data.data
|
|
46
|
+
.filter((model) => speechModel.test(model.id))
|
|
47
|
+
.map((model) => ({ id: model.id, name: model.id }));
|
|
48
|
+
},
|
|
22
49
|
synthesize: async (req) => {
|
|
23
50
|
const response = await deps.fetch(`${openAiAudioBase}/audio/speech`, {
|
|
24
51
|
method: "POST",
|
|
@@ -31,7 +58,7 @@ export function openAiTts(deps) {
|
|
|
31
58
|
// 400 and that is what the stage shows, so a user who chose Whole text learns the limit
|
|
32
59
|
// from the provider that set it.
|
|
33
60
|
body: JSON.stringify({
|
|
34
|
-
model: openAiTtsModel,
|
|
61
|
+
model: req.model ?? openAiTtsModel,
|
|
35
62
|
input: req.text,
|
|
36
63
|
voice: voiceOf(req.voiceId),
|
|
37
64
|
response_format: "mp3",
|
|
@@ -80,7 +107,7 @@ async function failure(response, voiceId) {
|
|
|
80
107
|
kind: kindOf(response.status),
|
|
81
108
|
// The voice ID is named, so a rejected voice reads differently from
|
|
82
109
|
// a rejected key.
|
|
83
|
-
message: `OpenAI answered ${response.status} for voice ${voiceId}: ${message}`,
|
|
110
|
+
message: `OpenAI answered ${response.status}${voiceId === undefined ? "" : ` for voice ${voiceId}`}: ${message}`,
|
|
84
111
|
...(retryAfterMs === undefined ? {} : { retryAfterMs }),
|
|
85
112
|
});
|
|
86
113
|
}
|