@gentbajko/slopify 0.5.1 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +36 -1
- package/SUBTITLES.md +21 -0
- package/dist/adapter-registry.js +18 -3
- package/dist/adapters/alignment/audio.js +45 -0
- package/dist/adapters/alignment/cache.js +91 -0
- package/dist/adapters/alignment/ctc.js +128 -0
- package/dist/adapters/alignment/index.js +34 -0
- package/dist/adapters/alignment/lock.js +58 -0
- package/dist/adapters/alignment/numbers.js +91 -0
- package/dist/adapters/alignment/protocol.js +23 -0
- package/dist/adapters/alignment/quality.js +17 -0
- package/dist/adapters/alignment/runner.js +76 -0
- package/dist/adapters/alignment/text.js +61 -0
- package/dist/adapters/alignment/vocabulary.js +69 -0
- package/dist/adapters/alignment/worker.js +163 -0
- package/dist/adapters/image/google.js +12 -15
- package/dist/adapters/image/models.js +72 -0
- package/dist/adapters/image/openai.js +3 -5
- package/dist/adapters/llm/catalogue-files.js +23 -0
- package/dist/adapters/llm/claude-code.js +9 -7
- package/dist/adapters/llm/codex-models.js +46 -0
- package/dist/adapters/llm/codex.js +7 -12
- package/dist/adapters/llm/gemini-models.js +85 -0
- package/dist/adapters/llm/gemini-workspace.js +70 -0
- package/dist/adapters/llm/gemini.js +138 -0
- package/dist/adapters/llm/openrouter.js +11 -2
- package/dist/adapters/llm/run-cli.js +6 -9
- package/dist/adapters/tts/cartesia.js +9 -1
- package/dist/adapters/tts/elevenlabs.js +33 -2
- package/dist/adapters/tts/inworld-async.js +150 -0
- package/dist/adapters/tts/inworld-text.js +32 -0
- package/dist/adapters/tts/inworld.js +170 -0
- package/dist/adapters/tts/openai.js +31 -4
- package/dist/assets/fonts/Barlow-Regular.ttf +0 -0
- package/dist/assets/fonts/OFL.txt +93 -0
- package/dist/assets/fonts/SOURCE.txt +7 -0
- package/dist/edge/cli.js +5 -0
- package/dist/edge/events/hub.js +9 -2
- package/dist/edge/events/preview-cache.js +40 -0
- package/dist/edge/http/actions.js +2 -0
- package/dist/edge/http/app.js +30 -1
- package/dist/edge/http/audio-preview.js +46 -0
- package/dist/edge/http/fonts.js +147 -0
- package/dist/edge/http/projects.js +23 -1
- package/dist/edge/http/providers.js +29 -0
- package/dist/edge/http/subtitles.js +92 -0
- package/dist/edge/http/update.js +57 -0
- package/dist/edge/update-worker.js +44 -0
- package/dist/kernel/audio-preview.js +188 -0
- package/dist/kernel/cli-command.js +45 -0
- package/dist/kernel/ports/subtitles.js +1 -0
- package/dist/kernel/runner/providers.js +75 -24
- package/dist/main.js +106 -25
- package/dist/model-catalog.js +37 -0
- package/dist/slices/admission/repo.js +2 -0
- package/dist/slices/admission/rules.js +6 -0
- package/dist/slices/control/index.js +1 -0
- package/dist/slices/control/providers.js +2 -0
- package/dist/slices/fonts/catalog.js +95 -0
- package/dist/slices/fonts/discovery.js +81 -0
- package/dist/slices/fonts/files.js +37 -0
- package/dist/slices/fonts/index.js +5 -0
- package/dist/slices/fonts/model.js +4 -0
- package/dist/slices/fonts/preview.js +14 -0
- package/dist/slices/fonts/sfnt.js +262 -0
- package/dist/slices/fonts/upload.js +40 -0
- package/dist/slices/narration/live.js +21 -0
- package/dist/slices/narration/run.js +9 -9
- package/dist/slices/research/run.js +3 -0
- package/dist/slices/settings/cli-paths.js +98 -0
- package/dist/slices/settings/cli-status.js +21 -12
- package/dist/slices/settings/model.js +11 -0
- package/dist/slices/settings/models.js +89 -0
- package/dist/slices/settings/readiness.js +13 -9
- package/dist/slices/storage/downloads.js +2 -0
- package/dist/slices/storage/layout.js +10 -0
- package/dist/slices/storage/model.js +5 -0
- package/dist/slices/storage/repo.js +1 -0
- package/dist/slices/subtitles/captions.js +91 -0
- package/dist/slices/subtitles/layout.js +19 -0
- package/dist/slices/subtitles/model.js +27 -0
- package/dist/slices/subtitles/prepare.js +167 -0
- package/dist/slices/subtitles/transcript.js +34 -0
- package/dist/slices/video/audio-export.js +3 -0
- package/dist/slices/video/ffmpeg.js +14 -7
- package/dist/slices/video/plan.js +4 -4
- package/dist/slices/video/run.js +5 -2
- package/dist/slices/video/write-export.js +85 -20
- package/dist/updater/candidate.js +28 -0
- package/dist/updater/forward.js +35 -0
- package/dist/updater/install-flow.js +26 -0
- package/dist/updater/install.js +50 -0
- package/dist/updater/model.js +24 -0
- package/dist/updater/plan.js +135 -0
- package/dist/updater/readiness.js +15 -0
- package/dist/updater/registry.js +19 -0
- package/dist/updater/service.js +128 -0
- package/dist/updater/worker.js +136 -0
- package/dist/web/assets/index-6QIQ-TzO.css +1 -0
- package/dist/web/assets/index-CNgWL-ct.js +130 -0
- package/dist/web/index.html +2 -2
- package/package.json +4 -2
- package/dist/web/assets/index-C1fsrIP5.css +0 -1
- package/dist/web/assets/index-DwJ_XTl3.js +0 -81
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
import { fork } from "node:child_process";
|
|
2
|
+
import { workerMessage } from "./protocol.js";
|
|
3
|
+
export async function runAlignmentWorker(input, signal, onProgress, worker = new URL("./worker.js", import.meta.url)) {
|
|
4
|
+
signal.throwIfAborted();
|
|
5
|
+
return new Promise((resolve, reject) => {
|
|
6
|
+
const options = {
|
|
7
|
+
stdio: ["ignore", "ignore", "pipe", "ipc"],
|
|
8
|
+
windowsHide: true,
|
|
9
|
+
execArgv: [],
|
|
10
|
+
serialization: "advanced",
|
|
11
|
+
};
|
|
12
|
+
const child = fork(worker, [], options);
|
|
13
|
+
let finishing = false;
|
|
14
|
+
let failure;
|
|
15
|
+
let output = [];
|
|
16
|
+
let stderr = "";
|
|
17
|
+
const finish = (error, words) => {
|
|
18
|
+
if (finishing)
|
|
19
|
+
return;
|
|
20
|
+
finishing = true;
|
|
21
|
+
failure = error;
|
|
22
|
+
output = words ?? [];
|
|
23
|
+
signal.removeEventListener("abort", abort);
|
|
24
|
+
child.kill("SIGKILL");
|
|
25
|
+
};
|
|
26
|
+
const abort = () => finish(new Error("Subtitle alignment was canceled."));
|
|
27
|
+
signal.addEventListener("abort", abort, { once: true });
|
|
28
|
+
child.stderr?.on("data", (data) => {
|
|
29
|
+
stderr = (stderr + data.toString("utf8")).slice(-2000);
|
|
30
|
+
});
|
|
31
|
+
child.on("error", (error) => finish(error));
|
|
32
|
+
// Wait for close, not only the IPC answer: Windows still owns the decoded audio
|
|
33
|
+
// handle until the worker has actually exited, so early cleanup can fail with EPERM.
|
|
34
|
+
child.on("close", (code) => {
|
|
35
|
+
if (!finishing)
|
|
36
|
+
finish(new Error(`Local subtitle alignment stopped (exit ${String(code)}). ${stderr}`.trim()));
|
|
37
|
+
if (failure !== undefined)
|
|
38
|
+
reject(failure);
|
|
39
|
+
else
|
|
40
|
+
resolve(output);
|
|
41
|
+
});
|
|
42
|
+
child.on("message", (raw) => {
|
|
43
|
+
if (finishing)
|
|
44
|
+
return;
|
|
45
|
+
const parsed = workerMessage.safeParse(raw);
|
|
46
|
+
if (!parsed.success) {
|
|
47
|
+
finish(new Error("The local subtitle worker returned invalid timing data."));
|
|
48
|
+
return;
|
|
49
|
+
}
|
|
50
|
+
const message = parsed.data;
|
|
51
|
+
if (message.type === "progress") {
|
|
52
|
+
try {
|
|
53
|
+
onProgress?.(message.current, message.total);
|
|
54
|
+
}
|
|
55
|
+
catch (error) {
|
|
56
|
+
finish(error instanceof Error ? error : new Error(String(error)));
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
else if (message.type === "error")
|
|
60
|
+
finish(new Error(message.message));
|
|
61
|
+
else
|
|
62
|
+
finish(undefined, message.words.map(({ confidence, ...word }) => ({
|
|
63
|
+
...word,
|
|
64
|
+
...(confidence === undefined ? {} : { confidence }),
|
|
65
|
+
})));
|
|
66
|
+
});
|
|
67
|
+
if (signal.aborted) {
|
|
68
|
+
abort();
|
|
69
|
+
return;
|
|
70
|
+
}
|
|
71
|
+
child.send(input, (error) => {
|
|
72
|
+
if (error !== null)
|
|
73
|
+
finish(error);
|
|
74
|
+
});
|
|
75
|
+
});
|
|
76
|
+
}
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
import { cardinal, numberForms } from "./numbers.js";
|
|
2
|
+
// Original display words remain intact; only acoustic matching uses normalized English.
|
|
3
|
+
export function speechWords(text, observed = "") {
|
|
4
|
+
const words = [];
|
|
5
|
+
for (const textWord of text.trim().split(/\s+/)) {
|
|
6
|
+
const forms = wordForms(textWord);
|
|
7
|
+
const spoken = forms.find((form) => form !== "" && ` ${observed} `.includes(` ${form} `)) ?? forms[0] ?? "";
|
|
8
|
+
if (spoken === "") {
|
|
9
|
+
const previous = words.at(-1);
|
|
10
|
+
if (previous !== undefined)
|
|
11
|
+
previous.text += ` ${textWord}`;
|
|
12
|
+
continue;
|
|
13
|
+
}
|
|
14
|
+
words.push({ text: textWord, spoken });
|
|
15
|
+
}
|
|
16
|
+
if (words.length === 0)
|
|
17
|
+
throw new Error("Local subtitles currently require an English transcript with spoken words.");
|
|
18
|
+
return words;
|
|
19
|
+
}
|
|
20
|
+
function wordForms(raw) {
|
|
21
|
+
const plain = raw.normalize("NFKD").replace(/\p{M}/gu, "").replace(/[‘’]/g, "'");
|
|
22
|
+
if (/\p{L}/u.test(plain.replace(/[A-Za-z]/g, "")))
|
|
23
|
+
throw new Error("Local subtitles currently support English speech and Latin-script names only.");
|
|
24
|
+
const core = plain.replace(/^[^\p{L}\p{N}$£€]+|[^\p{L}\p{N}%]+$/gu, "");
|
|
25
|
+
if (core === "")
|
|
26
|
+
return [];
|
|
27
|
+
const currency = core.startsWith("$")
|
|
28
|
+
? "DOLLARS"
|
|
29
|
+
: core.startsWith("£")
|
|
30
|
+
? "POUNDS"
|
|
31
|
+
: core.startsWith("€")
|
|
32
|
+
? "EUROS"
|
|
33
|
+
: "";
|
|
34
|
+
const percent = core.endsWith("%") ? "PERCENT" : "";
|
|
35
|
+
const clean = core.replace(/^[$£€]/, "").replace(/%$/, "");
|
|
36
|
+
let forms;
|
|
37
|
+
if (/^\d[\d,.]*(?:s|st|nd|rd|th)?$/i.test(clean)) {
|
|
38
|
+
forms = numberForms(clean);
|
|
39
|
+
}
|
|
40
|
+
else {
|
|
41
|
+
const expanded = clean
|
|
42
|
+
.replaceAll("&", " AND ")
|
|
43
|
+
.replaceAll("+", " PLUS ")
|
|
44
|
+
.replace(/\d+/g, (digits) => numberForms(digits)[0] ?? digits);
|
|
45
|
+
const normal = expanded
|
|
46
|
+
.toUpperCase()
|
|
47
|
+
.replace(/[^A-Z']/g, " ")
|
|
48
|
+
.replace(/\s+/g, " ")
|
|
49
|
+
.trim();
|
|
50
|
+
forms = /^[A-Z]{2,8}$/.test(clean) ? [normal, clean.split("").join(" ")] : [normal];
|
|
51
|
+
}
|
|
52
|
+
const result = forms.map((form) => [form, currency, percent].filter(Boolean).join(" "));
|
|
53
|
+
const money = /^(\d[\d,]*)\.(\d{2})$/.exec(clean);
|
|
54
|
+
if (currency !== "" && money !== null) {
|
|
55
|
+
const whole = cardinal(Number((money[1] ?? "").replaceAll(",", "")));
|
|
56
|
+
const fraction = cardinal(Number(money[2]));
|
|
57
|
+
const unit = currency === "POUNDS" ? "PENCE" : "CENTS";
|
|
58
|
+
result.push(`${whole} ${currency} AND ${fraction} ${unit}`, `${whole} ${currency} ${fraction}`, `${whole} ${fraction}`);
|
|
59
|
+
}
|
|
60
|
+
return result;
|
|
61
|
+
}
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
// The vocabulary belongs to the pinned wav2vec2-base-960h model.
|
|
2
|
+
export const vocabulary = {
|
|
3
|
+
"<pad>": 0,
|
|
4
|
+
"<s>": 1,
|
|
5
|
+
"</s>": 2,
|
|
6
|
+
"<unk>": 3,
|
|
7
|
+
"|": 4,
|
|
8
|
+
E: 5,
|
|
9
|
+
T: 6,
|
|
10
|
+
A: 7,
|
|
11
|
+
O: 8,
|
|
12
|
+
N: 9,
|
|
13
|
+
I: 10,
|
|
14
|
+
H: 11,
|
|
15
|
+
S: 12,
|
|
16
|
+
R: 13,
|
|
17
|
+
D: 14,
|
|
18
|
+
L: 15,
|
|
19
|
+
U: 16,
|
|
20
|
+
M: 17,
|
|
21
|
+
W: 18,
|
|
22
|
+
C: 19,
|
|
23
|
+
F: 20,
|
|
24
|
+
G: 21,
|
|
25
|
+
Y: 22,
|
|
26
|
+
P: 23,
|
|
27
|
+
B: 24,
|
|
28
|
+
V: 25,
|
|
29
|
+
K: 26,
|
|
30
|
+
"'": 27,
|
|
31
|
+
X: 28,
|
|
32
|
+
J: 29,
|
|
33
|
+
Q: 30,
|
|
34
|
+
Z: 31,
|
|
35
|
+
};
|
|
36
|
+
export const letters = [
|
|
37
|
+
"",
|
|
38
|
+
"",
|
|
39
|
+
"",
|
|
40
|
+
"",
|
|
41
|
+
" ",
|
|
42
|
+
"E",
|
|
43
|
+
"T",
|
|
44
|
+
"A",
|
|
45
|
+
"O",
|
|
46
|
+
"N",
|
|
47
|
+
"I",
|
|
48
|
+
"H",
|
|
49
|
+
"S",
|
|
50
|
+
"R",
|
|
51
|
+
"D",
|
|
52
|
+
"L",
|
|
53
|
+
"U",
|
|
54
|
+
"M",
|
|
55
|
+
"W",
|
|
56
|
+
"C",
|
|
57
|
+
"F",
|
|
58
|
+
"G",
|
|
59
|
+
"Y",
|
|
60
|
+
"P",
|
|
61
|
+
"B",
|
|
62
|
+
"V",
|
|
63
|
+
"K",
|
|
64
|
+
"'",
|
|
65
|
+
"X",
|
|
66
|
+
"J",
|
|
67
|
+
"Q",
|
|
68
|
+
"Z",
|
|
69
|
+
];
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
import { open, readFile, stat } from "node:fs/promises";
|
|
2
|
+
import * as ort from "onnxruntime-web/wasm";
|
|
3
|
+
import { alignWindow, frameSeconds, mismatch } from "./ctc.js";
|
|
4
|
+
import { workerInput } from "./protocol.js";
|
|
5
|
+
import { agreesWithSpeech } from "./quality.js";
|
|
6
|
+
import { speechWords } from "./text.js";
|
|
7
|
+
import { letters } from "./vocabulary.js";
|
|
8
|
+
const sampleRate = 16000;
|
|
9
|
+
const windowSeconds = 12;
|
|
10
|
+
const overlapSeconds = 2;
|
|
11
|
+
process.once("message", (raw) => {
|
|
12
|
+
const input = workerInput.safeParse(raw);
|
|
13
|
+
if (!input.success) {
|
|
14
|
+
send({ type: "error", message: "The subtitle worker received an invalid request." });
|
|
15
|
+
return;
|
|
16
|
+
}
|
|
17
|
+
run(input.data).then((words) => send({ type: "done", words }), (error) => send({ type: "error", message: error instanceof Error ? error.message : String(error) }));
|
|
18
|
+
});
|
|
19
|
+
function send(message) {
|
|
20
|
+
process.send?.(message);
|
|
21
|
+
}
|
|
22
|
+
async function run(input) {
|
|
23
|
+
ort.env.wasm.numThreads = 1;
|
|
24
|
+
const model = new Uint8Array(await readFile(input.modelPath));
|
|
25
|
+
const session = await ort.InferenceSession.create(model, { executionProviders: ["wasm"] });
|
|
26
|
+
const file = await open(input.pcmPath, "r");
|
|
27
|
+
try {
|
|
28
|
+
const totalSamples = (await stat(input.pcmPath)).size / 4;
|
|
29
|
+
if (!Number.isInteger(totalSamples) || totalSamples < sampleRate / 10)
|
|
30
|
+
throw new Error("The narration is too short to align subtitles.");
|
|
31
|
+
const source = speechWords(input.text);
|
|
32
|
+
const output = [];
|
|
33
|
+
let cursor = 0;
|
|
34
|
+
let sampleAt = 0;
|
|
35
|
+
while (sampleAt < totalSamples && cursor < source.length) {
|
|
36
|
+
const count = Math.min(windowSeconds * sampleRate, totalSamples - sampleAt);
|
|
37
|
+
const buffer = Buffer.alloc(count * 4);
|
|
38
|
+
await file.read(buffer, 0, buffer.length, sampleAt * 4);
|
|
39
|
+
const audio = new Float32Array(buffer.buffer, buffer.byteOffset, count);
|
|
40
|
+
const normalized = normalize(audio);
|
|
41
|
+
if (normalized === undefined) {
|
|
42
|
+
sampleAt += count;
|
|
43
|
+
continue;
|
|
44
|
+
}
|
|
45
|
+
const result = await session.run({
|
|
46
|
+
input_values: new ort.Tensor("float32", normalized, [1, count]),
|
|
47
|
+
});
|
|
48
|
+
const logits = result.logits;
|
|
49
|
+
if (logits === undefined ||
|
|
50
|
+
!(logits.data instanceof Float32Array) ||
|
|
51
|
+
logits.dims.length !== 3 ||
|
|
52
|
+
logits.dims[2] !== 32)
|
|
53
|
+
throw new Error("The local subtitle model returned an unsupported result.");
|
|
54
|
+
const frames = logits.dims[1] ?? 0;
|
|
55
|
+
const observed = greedy(logits.data, frames);
|
|
56
|
+
if (observed.replace(/[^A-Z]/g, "").length === 0) {
|
|
57
|
+
sampleAt += count;
|
|
58
|
+
for (const tensor of Object.values(result))
|
|
59
|
+
tensor.dispose();
|
|
60
|
+
continue;
|
|
61
|
+
}
|
|
62
|
+
const candidate = candidates(source, cursor, observed);
|
|
63
|
+
const finalWindow = sampleAt + count >= totalSamples;
|
|
64
|
+
const complete = finalWindow && cursor + candidate.length === source.length;
|
|
65
|
+
const aligned = alignWindow(logits.data, frames, candidate, complete).words;
|
|
66
|
+
const cutoff = finalWindow ? count / sampleRate : count / sampleRate - overlapSeconds;
|
|
67
|
+
const accepted = aligned.filter((word) => word.end <= cutoff);
|
|
68
|
+
const last = accepted.at(-1);
|
|
69
|
+
if (last === undefined)
|
|
70
|
+
throw new Error(mismatch);
|
|
71
|
+
const heard = greedy(logits.data, Math.min(frames, Math.ceil(last.end / frameSeconds)));
|
|
72
|
+
if (!agreesWithSpeech(candidate
|
|
73
|
+
.slice(0, accepted.length)
|
|
74
|
+
.map((word) => word.spoken)
|
|
75
|
+
.join(" "), heard))
|
|
76
|
+
throw new Error(mismatch);
|
|
77
|
+
const offset = sampleAt / sampleRate;
|
|
78
|
+
output.push(...accepted.map((word) => ({
|
|
79
|
+
...word,
|
|
80
|
+
start: offset + word.start,
|
|
81
|
+
end: offset + word.end,
|
|
82
|
+
})));
|
|
83
|
+
cursor += accepted.length;
|
|
84
|
+
send({ type: "progress", current: cursor, total: source.length });
|
|
85
|
+
// Keep a small leading silence margin but never re-use speech already captioned.
|
|
86
|
+
sampleAt += Math.max(1, Math.floor(last.end * sampleRate));
|
|
87
|
+
for (const tensor of Object.values(result))
|
|
88
|
+
tensor.dispose();
|
|
89
|
+
}
|
|
90
|
+
if (cursor !== source.length || output.length === 0)
|
|
91
|
+
throw new Error(mismatch);
|
|
92
|
+
// A substantial spoken tail absent from the transcript is a mismatch too.
|
|
93
|
+
while (sampleAt + sampleRate < totalSamples) {
|
|
94
|
+
const count = Math.min(windowSeconds * sampleRate, totalSamples - sampleAt);
|
|
95
|
+
const buffer = Buffer.alloc(count * 4);
|
|
96
|
+
await file.read(buffer, 0, buffer.length, sampleAt * 4);
|
|
97
|
+
const audio = normalize(new Float32Array(buffer.buffer, buffer.byteOffset, count));
|
|
98
|
+
if (audio !== undefined) {
|
|
99
|
+
const result = await session.run({
|
|
100
|
+
input_values: new ort.Tensor("float32", audio, [1, count]),
|
|
101
|
+
});
|
|
102
|
+
const logits = result.logits;
|
|
103
|
+
if (logits !== undefined &&
|
|
104
|
+
logits.data instanceof Float32Array &&
|
|
105
|
+
greedy(logits.data, logits.dims[1] ?? 0).replace(/[^A-Z]/g, "").length > 8)
|
|
106
|
+
throw new Error(mismatch);
|
|
107
|
+
for (const tensor of Object.values(result))
|
|
108
|
+
tensor.dispose();
|
|
109
|
+
}
|
|
110
|
+
sampleAt += count;
|
|
111
|
+
}
|
|
112
|
+
return output;
|
|
113
|
+
}
|
|
114
|
+
finally {
|
|
115
|
+
await file.close();
|
|
116
|
+
await session.release();
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
function candidates(source, cursor, observed) {
|
|
120
|
+
const selected = [];
|
|
121
|
+
let length = 0;
|
|
122
|
+
for (let index = cursor; index < source.length; index += 1) {
|
|
123
|
+
const word = source[index];
|
|
124
|
+
if (word === undefined)
|
|
125
|
+
break;
|
|
126
|
+
const normalized = speechWords(word.text, observed)[0];
|
|
127
|
+
if (normalized === undefined)
|
|
128
|
+
continue;
|
|
129
|
+
if (length + normalized.spoken.length + 1 > 900)
|
|
130
|
+
break;
|
|
131
|
+
selected.push(normalized);
|
|
132
|
+
length += normalized.spoken.length + 1;
|
|
133
|
+
}
|
|
134
|
+
return selected;
|
|
135
|
+
}
|
|
136
|
+
function normalize(audio) {
|
|
137
|
+
let mean = 0;
|
|
138
|
+
for (const value of audio)
|
|
139
|
+
mean += value;
|
|
140
|
+
mean /= audio.length;
|
|
141
|
+
let variance = 0;
|
|
142
|
+
for (const value of audio)
|
|
143
|
+
variance += (value - mean) ** 2;
|
|
144
|
+
variance /= audio.length;
|
|
145
|
+
if (variance < 1e-10)
|
|
146
|
+
return undefined;
|
|
147
|
+
const divisor = Math.sqrt(variance + 1e-7);
|
|
148
|
+
return Float32Array.from(audio, (value) => (value - mean) / divisor);
|
|
149
|
+
}
|
|
150
|
+
function greedy(logits, frames) {
|
|
151
|
+
let previous = -1;
|
|
152
|
+
let text = "";
|
|
153
|
+
for (let frame = 0; frame < frames; frame += 1) {
|
|
154
|
+
let best = 0;
|
|
155
|
+
for (let label = 1; label < 32; label += 1)
|
|
156
|
+
if ((logits[frame * 32 + label] ?? -Infinity) > (logits[frame * 32 + best] ?? -Infinity))
|
|
157
|
+
best = label;
|
|
158
|
+
if (best !== previous && best !== 0)
|
|
159
|
+
text += letters[best] ?? "";
|
|
160
|
+
previous = best;
|
|
161
|
+
}
|
|
162
|
+
return text.trim().replace(/\s+/g, " ");
|
|
163
|
+
}
|
|
@@ -3,28 +3,21 @@ import { redact } from "../../kernel/log.js";
|
|
|
3
3
|
import { providerError } from "../../kernel/ports/model.js";
|
|
4
4
|
import { retryAfter } from "../retry-after.js";
|
|
5
5
|
import { describeBytes, sniffImage } from "./bytes.js";
|
|
6
|
+
import { discoverGoogleImages } from "./models.js";
|
|
6
7
|
// The HTTP gateway adapter for Google's own image generation, billed to a Gemini API key
|
|
7
8
|
// rather than to a host reselling the same models. Like the OpenAI one it hands back the
|
|
8
|
-
// bytes with the call, so there is no link to follow
|
|
9
|
-
// one `input` string and a `response_format`, with no per-model size table to keep.
|
|
9
|
+
// bytes with the call, so there is no link to follow.
|
|
10
10
|
export const googleImagesBase = "https://generativelanguage.googleapis.com/v1beta";
|
|
11
|
-
//
|
|
12
|
-
// text and embedding model too, so the image shortlist is this adapter's own data. Named as
|
|
13
|
-
// Google markets them: "Nano Banana" is the family, `gemini-*-image` is what the API answers
|
|
14
|
-
// to, and the picker shows the name a user would recognise. Newest first.
|
|
15
|
-
//
|
|
16
|
-
// A Gemini model without the `-image` suffix returns text and cannot be used here, whatever
|
|
17
|
-
// its version number: `gemini-3.8-flash` is newer than all of these and generates no images.
|
|
11
|
+
// Offline choices only; the picker normally loads the provider catalogue.
|
|
18
12
|
export const googleImageModels = [
|
|
19
13
|
{ id: "gemini-3.1-flash-image", name: "Nano Banana 2" },
|
|
20
14
|
{ id: "gemini-3.1-flash-lite-image", name: "Nano Banana 2 Lite" },
|
|
21
15
|
{ id: "gemini-3-pro-image", name: "Nano Banana Pro" },
|
|
22
16
|
{ id: "gemini-2.5-flash-image", name: "Nano Banana" },
|
|
23
17
|
];
|
|
24
|
-
//
|
|
25
|
-
//
|
|
26
|
-
|
|
27
|
-
const imageSize = "2K";
|
|
18
|
+
// 2.5 and Flash Lite cannot generate 2K images. Unknown models keep their own default
|
|
19
|
+
// resolution until their capabilities are known; discovery alone does not describe sizes.
|
|
20
|
+
const highResolutionModel = /^gemini-(?:3(?:\.1)?-pro|3\.1-flash)-image(?:-preview(?:-\d{2}-\d{2})?)?$/;
|
|
28
21
|
// A wire payload is narrowed, never cast. The image arrives as one content part of a
|
|
29
22
|
// `model_output` step, beside any text the model chose to write, so both the step and the
|
|
30
23
|
// part are searched for by type rather than read off a fixed index.
|
|
@@ -58,7 +51,7 @@ const refusalWords = /\b(safety|blocked|content polic|prohibited|violat)/i;
|
|
|
58
51
|
export function googleImage(deps) {
|
|
59
52
|
return {
|
|
60
53
|
id: "google-image",
|
|
61
|
-
models: () =>
|
|
54
|
+
models: () => discoverGoogleImages(deps),
|
|
62
55
|
generate: async (req) => {
|
|
63
56
|
const response = await deps.fetch(`${googleImagesBase}/interactions`, {
|
|
64
57
|
method: "POST",
|
|
@@ -72,7 +65,11 @@ export function googleImage(deps) {
|
|
|
72
65
|
// The stage sends Number as that many independent calls, one piece each, so one
|
|
73
66
|
// image per request is what it asks for. Nothing about style is set: the stage
|
|
74
67
|
// asks for the provider's own.
|
|
75
|
-
response_format: {
|
|
68
|
+
response_format: {
|
|
69
|
+
type: "image",
|
|
70
|
+
aspect_ratio: req.aspect,
|
|
71
|
+
...(highResolutionModel.test(req.model) ? { image_size: "2K" } : {}),
|
|
72
|
+
},
|
|
76
73
|
}),
|
|
77
74
|
});
|
|
78
75
|
if (!response.ok) {
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
import { providerError } from "../../kernel/ports/model.js";
|
|
3
|
+
const openAiModels = z.object({ data: z.array(z.object({ id: z.string() })) });
|
|
4
|
+
const googleModels = z.object({
|
|
5
|
+
models: z
|
|
6
|
+
.array(z.object({
|
|
7
|
+
name: z.string(),
|
|
8
|
+
displayName: z.string().optional(),
|
|
9
|
+
supportedGenerationMethods: z.array(z.string()).optional(),
|
|
10
|
+
}))
|
|
11
|
+
.default([]),
|
|
12
|
+
nextPageToken: z.string().optional(),
|
|
13
|
+
});
|
|
14
|
+
export async function discoverOpenAiImages(deps) {
|
|
15
|
+
const response = await deps.fetch("https://api.openai.com/v1/models", {
|
|
16
|
+
headers: { Authorization: `Bearer ${requireKey(deps)}` },
|
|
17
|
+
signal: AbortSignal.timeout(10_000),
|
|
18
|
+
});
|
|
19
|
+
if (!response.ok)
|
|
20
|
+
throw unavailable();
|
|
21
|
+
const parsed = openAiModels.safeParse(await response.json());
|
|
22
|
+
if (!parsed.success)
|
|
23
|
+
throw unavailable();
|
|
24
|
+
return parsed.data.data
|
|
25
|
+
.filter(({ id }) => id.startsWith("gpt-image-"))
|
|
26
|
+
.map(({ id }) => ({ id, name: id }));
|
|
27
|
+
}
|
|
28
|
+
export async function discoverGoogleImages(deps) {
|
|
29
|
+
const headers = { "x-goog-api-key": requireKey(deps) };
|
|
30
|
+
const signal = AbortSignal.timeout(10_000);
|
|
31
|
+
const result = [];
|
|
32
|
+
const tokens = new Set();
|
|
33
|
+
let page = "";
|
|
34
|
+
// Bound a malformed provider's pagination while allowing 20,000 model records.
|
|
35
|
+
for (let count = 0; count < 20; count++) {
|
|
36
|
+
const url = new URL("https://generativelanguage.googleapis.com/v1beta/models");
|
|
37
|
+
url.searchParams.set("pageSize", "1000");
|
|
38
|
+
if (page !== "")
|
|
39
|
+
url.searchParams.set("pageToken", page);
|
|
40
|
+
const response = await deps.fetch(url, { headers, signal });
|
|
41
|
+
if (!response.ok)
|
|
42
|
+
throw unavailable();
|
|
43
|
+
const parsed = googleModels.safeParse(await response.json());
|
|
44
|
+
if (!parsed.success)
|
|
45
|
+
throw unavailable();
|
|
46
|
+
for (const model of parsed.data.models) {
|
|
47
|
+
const id = model.name.replace(/^models\//, "");
|
|
48
|
+
// Imagen uses a different generation API. Only Gemini image models share this adapter.
|
|
49
|
+
if (id.startsWith("gemini-") &&
|
|
50
|
+
/-image(?:-|$)/.test(id) &&
|
|
51
|
+
model.supportedGenerationMethods?.includes("generateContent") === true) {
|
|
52
|
+
result.push({ id, name: model.displayName ?? id });
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
page = parsed.data.nextPageToken ?? "";
|
|
56
|
+
if (page === "")
|
|
57
|
+
return result;
|
|
58
|
+
if (tokens.has(page))
|
|
59
|
+
throw unavailable();
|
|
60
|
+
tokens.add(page);
|
|
61
|
+
}
|
|
62
|
+
throw unavailable();
|
|
63
|
+
}
|
|
64
|
+
function requireKey(deps) {
|
|
65
|
+
const key = deps.key();
|
|
66
|
+
if (key === undefined || key === "")
|
|
67
|
+
throw providerError({ kind: "missing_key", message: "Save an API key to load models." });
|
|
68
|
+
return key;
|
|
69
|
+
}
|
|
70
|
+
function unavailable() {
|
|
71
|
+
return providerError({ kind: "other", message: "The provider model list could not be loaded." });
|
|
72
|
+
}
|
|
@@ -3,15 +3,13 @@ import { redact } from "../../kernel/log.js";
|
|
|
3
3
|
import { providerError } from "../../kernel/ports/model.js";
|
|
4
4
|
import { retryAfter } from "../retry-after.js";
|
|
5
5
|
import { describeBytes, sniffImage } from "./bytes.js";
|
|
6
|
+
import { discoverOpenAiImages } from "./models.js";
|
|
6
7
|
// The HTTP gateway adapter for OpenAI's images endpoint: the platform's own `fetch` and
|
|
7
8
|
// nothing else, because the whole call is one request. Unlike fal and Replicate this one
|
|
8
9
|
// hands back the image itself - a GPT image model always answers with base64, never a URL -
|
|
9
10
|
// so there is no link to follow.
|
|
10
11
|
export const openAiImagesBase = "https://api.openai.com/v1";
|
|
11
|
-
//
|
|
12
|
-
// lists every model on the account, chat and embeddings among them, so the image
|
|
13
|
-
// shortlist is this adapter's own data. These are the four GPT image models OpenAI
|
|
14
|
-
// documents; adding the next one is a line here and no code change anywhere else.
|
|
12
|
+
// Offline choices only; the picker normally loads the provider catalogue.
|
|
15
13
|
export const openAiImageModels = [
|
|
16
14
|
{ id: "gpt-image-2", name: "GPT Image 2" },
|
|
17
15
|
{ id: "gpt-image-1.5", name: "GPT Image 1.5" },
|
|
@@ -52,7 +50,7 @@ export function sizeFor(model, aspect) {
|
|
|
52
50
|
export function openAiImage(deps) {
|
|
53
51
|
return {
|
|
54
52
|
id: "openai-image",
|
|
55
|
-
models: () =>
|
|
53
|
+
models: () => discoverOpenAiImages(deps),
|
|
56
54
|
generate: async (req) => {
|
|
57
55
|
const response = await deps.fetch(`${openAiImagesBase}/images/generations`, {
|
|
58
56
|
method: "POST",
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
import { open } from "node:fs/promises";
|
|
2
|
+
// Bound both the initial read and a file that grows between stat and read. Model
|
|
3
|
+
// metadata is data only: none of the installed CLI's modules are evaluated.
|
|
4
|
+
export async function readCatalogueFile(path, maxBytes) {
|
|
5
|
+
const file = await open(path, "r");
|
|
6
|
+
try {
|
|
7
|
+
const info = await file.stat();
|
|
8
|
+
if (!info.isFile() || info.size > maxBytes)
|
|
9
|
+
throw new Error("Invalid model metadata file");
|
|
10
|
+
const bytes = Buffer.alloc(maxBytes + 1);
|
|
11
|
+
let size = 0;
|
|
12
|
+
while (size <= maxBytes) {
|
|
13
|
+
const chunk = await file.read(bytes, size, bytes.length - size, null);
|
|
14
|
+
if (chunk.bytesRead === 0)
|
|
15
|
+
return bytes.toString("utf8", 0, size);
|
|
16
|
+
size += chunk.bytesRead;
|
|
17
|
+
}
|
|
18
|
+
throw new Error("Model metadata file is too large");
|
|
19
|
+
}
|
|
20
|
+
finally {
|
|
21
|
+
await file.close();
|
|
22
|
+
}
|
|
23
|
+
}
|
|
@@ -8,10 +8,9 @@ import { lines } from "./sse-lines.js";
|
|
|
8
8
|
// may not import `slices/**`, and `slices/settings/cli-status.ts` already probes the binary per
|
|
9
9
|
// request. The registry `main.ts` builds is where this adapter and that probe meet.
|
|
10
10
|
export const claudeCodeBinary = "claude";
|
|
11
|
-
//
|
|
12
|
-
//
|
|
13
|
-
//
|
|
14
|
-
// reading the list off the CLI is the upgrade when it can print one.
|
|
11
|
+
// Official stable family aliases resolve to the latest model available to the
|
|
12
|
+
// installed CLI/account. Full model IDs remain available through custom entry.
|
|
13
|
+
// https://code.claude.com/docs/en/model-config
|
|
15
14
|
export const claudeCodeModels = [
|
|
16
15
|
{ id: "fable", name: "Claude Fable (latest)" },
|
|
17
16
|
{ id: "opus", name: "Claude Opus (latest)" },
|
|
@@ -37,6 +36,7 @@ export function claudeCodeArgs(req) {
|
|
|
37
36
|
"stream-json",
|
|
38
37
|
// stream-json output is refused without it.
|
|
39
38
|
"--verbose",
|
|
39
|
+
"--include-partial-messages",
|
|
40
40
|
// `claude --help` 2.1.263: safe mode excludes CLAUDE.md, output styles, skills and
|
|
41
41
|
// hooks while retaining login and managed policy. --bare would discard OAuth.
|
|
42
42
|
"--safe-mode",
|
|
@@ -82,6 +82,10 @@ export function claudeCodeLlm(deps) {
|
|
|
82
82
|
continue;
|
|
83
83
|
}
|
|
84
84
|
const event = cliEvent(binary, line);
|
|
85
|
+
// Partial text/thinking and tool events are a heartbeat. Full assistant
|
|
86
|
+
// messages remain the sole prose source so text is never appended twice.
|
|
87
|
+
yield { type: "activity" };
|
|
88
|
+
req.signal.throwIfAborted();
|
|
85
89
|
if (event.type === "assistant") {
|
|
86
90
|
for (const block of cliShaped(binary, assistantEvent, event.value).message.content) {
|
|
87
91
|
// A turn also carries `thinking` and `tool_use` blocks; only the prose is the
|
|
@@ -133,9 +137,7 @@ export function claudeCodeLlm(deps) {
|
|
|
133
137
|
}
|
|
134
138
|
return {
|
|
135
139
|
id: "claude-code",
|
|
136
|
-
//
|
|
137
|
-
// see life on the stream. ceiling: `--include-partial-messages` would give per-token
|
|
138
|
-
// deltas for the streamed article; it is the upgrade when the page needs finer text.
|
|
140
|
+
// Partial messages keep the deadline alive; complete assistant turns supply prose.
|
|
139
141
|
capabilities: { streams: true, reportsUsage: true, webSearch: true },
|
|
140
142
|
models: () => Promise.resolve(claudeCodeModels),
|
|
141
143
|
complete,
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { homedir } from "node:os";
|
|
2
|
+
import { join } from "node:path";
|
|
3
|
+
import { z } from "zod";
|
|
4
|
+
import { readCatalogueFile } from "./catalogue-files.js";
|
|
5
|
+
const safeText = z
|
|
6
|
+
.string()
|
|
7
|
+
.trim()
|
|
8
|
+
.min(1)
|
|
9
|
+
.max(256)
|
|
10
|
+
.refine((text) => [...text].every((character) => character.charCodeAt(0) >= 32 && character.charCodeAt(0) !== 127));
|
|
11
|
+
// Deliberately omit model instructions, capabilities and every authentication
|
|
12
|
+
// file. Visibility refers to the CLI picker, not supported_in_api: some visible
|
|
13
|
+
// CLI-only models cannot be called through the public API.
|
|
14
|
+
const cache = z.object({ models: z.array(z.unknown()).max(1000) });
|
|
15
|
+
const model = z.object({
|
|
16
|
+
slug: safeText.refine((id) => !/\s/.test(id)),
|
|
17
|
+
display_name: safeText.optional(),
|
|
18
|
+
visibility: z.literal("list"),
|
|
19
|
+
priority: z.number().finite().optional(),
|
|
20
|
+
});
|
|
21
|
+
export async function nodeCodexModels(env = process.env) {
|
|
22
|
+
try {
|
|
23
|
+
const directory = env.CODEX_HOME?.trim() || join(homedir(), ".codex");
|
|
24
|
+
const source = await readCatalogueFile(join(directory, "models_cache.json"), 8 * 1024 * 1024);
|
|
25
|
+
const parsed = cache.parse(JSON.parse(source));
|
|
26
|
+
const models = parsed.models
|
|
27
|
+
.flatMap((entry) => {
|
|
28
|
+
const parsed = model.safeParse(entry);
|
|
29
|
+
return parsed.success ? [parsed.data] : [];
|
|
30
|
+
})
|
|
31
|
+
.sort((left, right) => (left.priority ?? Infinity) - (right.priority ?? Infinity));
|
|
32
|
+
const unique = new Map();
|
|
33
|
+
for (const item of models) {
|
|
34
|
+
if (!unique.has(item.slug)) {
|
|
35
|
+
unique.set(item.slug, { id: item.slug, name: item.display_name ?? item.slug });
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
if (unique.size === 0)
|
|
39
|
+
throw new Error("No visible models");
|
|
40
|
+
return [...unique.values()];
|
|
41
|
+
}
|
|
42
|
+
catch {
|
|
43
|
+
// Do not expose cache contents, home paths or raw JSON errors to the browser.
|
|
44
|
+
throw new Error("Codex model metadata is unavailable. Open Codex once to refresh its model list, or enter a custom model ID.");
|
|
45
|
+
}
|
|
46
|
+
}
|