nixamp 0.25.1 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +108 -7
- package/dist/captions.d.ts +7 -0
- package/dist/captions.js +92 -28
- package/dist/live-voice.d.ts +70 -0
- package/dist/live-voice.js +299 -0
- package/dist/server.d.ts +2 -0
- package/dist/server.js +154 -5
- package/dist/speaker-turns.d.ts +32 -0
- package/dist/speaker-turns.js +30 -0
- package/dist/speech.d.ts +47 -0
- package/dist/speech.js +49 -4
- package/dist/transcript-client.d.ts +2 -2
- package/dist/transcript-client.js +5 -2
- package/dist/transcripts.d.ts +5 -0
- package/dist/transcripts.js +8 -3
- package/dist/translate-jobs.js +13 -7
- package/dist/translate.d.ts +2 -0
- package/dist/translate.js +9 -0
- package/dist/voice-profile.d.ts +7 -0
- package/dist/voice-profile.js +53 -0
- package/dist/warm.js +3 -3
- package/package.json +1 -1
- package/src/captions.ts +86 -25
- package/src/live-voice.ts +255 -0
- package/src/server.ts +104 -5
- package/src/speaker-turns.ts +34 -0
- package/src/speech.ts +46 -7
- package/src/transcript-client.ts +4 -0
- package/src/transcripts.ts +13 -3
- package/src/translate-jobs.ts +13 -7
- package/src/translate.ts +7 -1
- package/src/voice-profile.ts +46 -0
- package/src/warm.ts +3 -3
- package/web/dist/assets/{hls-3VKVEQE3-GlzSYi0P.js → hls-3VKVEQE3-vgax_tk1.js} +1 -1
- package/web/dist/assets/index-BzjrTOLf.js +1 -0
- package/web/dist/assets/index-D2Iy07pG.css +1 -0
- package/web/dist/assets/{mpegts-KIbyW_RL.js → mpegts-DmcUOiHq.js} +1 -1
- package/web/dist/assets/{mpegts-LO6RVLD6-Fs_vbGcB.js → mpegts-LO6RVLD6-CH3EQi6L.js} +1 -1
- package/web/dist/index.html +42 -22
- package/web/dist/sw.js +6 -6
- package/web/dist/assets/index-DJHHB_63.js +0 -1
- package/web/dist/assets/index-oyp61Kly.css +0 -1
|
@@ -0,0 +1,299 @@
|
|
|
1
|
+
/** ElevenLabs Flash for live captions. The key stays on the account server. */
|
|
2
|
+
import { createHash, randomBytes } from "node:crypto";
|
|
3
|
+
import { SpeechError, decodeWav, encodeWav, quietSamples } from "./speech.js";
|
|
4
|
+
import { speakerTurns } from "./speaker-turns.js";
|
|
5
|
+
import { Guard } from "./guard.js";
|
|
6
|
+
export const LIVE_VOICE_MODEL = "eleven_flash_v2_5";
|
|
7
|
+
export const LIVE_VOICE_RATE = 16_000;
|
|
8
|
+
export const LIVE_VOICE_LANGUAGES = new Set("en ja zh de hi fr ko pt it es id nl tr fil pl sv bg ro ar cs el fi hr ms sk da ta uk ru hu no vi".split(" "));
|
|
9
|
+
const HEADERS = { "content-type": "audio/pcm", "cache-control": "no-store", "x-audio-sample-rate": String(LIVE_VOICE_RATE) };
|
|
10
|
+
const budget = (value, fallback) => Number.isFinite(value) && value >= 0 ? Math.floor(value) : fallback;
|
|
11
|
+
export class LiveVoice {
|
|
12
|
+
key;
|
|
13
|
+
fetcher;
|
|
14
|
+
now;
|
|
15
|
+
catalog = null;
|
|
16
|
+
catalogUntil = 0;
|
|
17
|
+
cache = new Map();
|
|
18
|
+
pending = new Map();
|
|
19
|
+
usage = new Map();
|
|
20
|
+
charsPerMinute;
|
|
21
|
+
requests;
|
|
22
|
+
grants = new Map();
|
|
23
|
+
activeBy = new Map();
|
|
24
|
+
db;
|
|
25
|
+
schema = null;
|
|
26
|
+
dailyChars;
|
|
27
|
+
cleanupAt = 0;
|
|
28
|
+
hearing = new Set();
|
|
29
|
+
dailyAudioSeconds;
|
|
30
|
+
userDailyChars;
|
|
31
|
+
userDailyAudioSeconds;
|
|
32
|
+
constructor(options = {}) {
|
|
33
|
+
this.key = options.apiKey ?? process.env["ELEVENLABS_API_KEY"] ?? "";
|
|
34
|
+
this.fetcher = options.fetcher ?? fetch;
|
|
35
|
+
this.now = options.now ?? Date.now;
|
|
36
|
+
this.charsPerMinute = budget(options.charsPerMinute, 3000);
|
|
37
|
+
this.dailyChars = budget(options.dailyChars, 200_000);
|
|
38
|
+
this.dailyAudioSeconds = budget(options.dailyAudioSeconds, 86_400);
|
|
39
|
+
this.userDailyChars = budget(options.userDailyChars, 120_000);
|
|
40
|
+
this.userDailyAudioSeconds = budget(options.userDailyAudioSeconds, 43_200);
|
|
41
|
+
this.requests = new Guard(this.now);
|
|
42
|
+
this.db = options.db;
|
|
43
|
+
}
|
|
44
|
+
available() { return this.key !== ""; }
|
|
45
|
+
/** Optional diarization, billed only while a signed-in listener requests it.
|
|
46
|
+
* Rolling audio is bounded to 15 seconds, including overlap. Every second
|
|
47
|
+
* submitted (also repeated context) consumes the persistent provider budget. */
|
|
48
|
+
async hear(bytes, by, signal) {
|
|
49
|
+
if (!this.available())
|
|
50
|
+
throw new SpeechError("speaker voices are unavailable", 503);
|
|
51
|
+
const wav = decodeWav(bytes);
|
|
52
|
+
const seconds = wav.samples.length / wav.rate;
|
|
53
|
+
if (wav.rate !== 16000 || wav.channels !== 1 || seconds < 0.2 || seconds > 15.1)
|
|
54
|
+
throw new SpeechError("send up to 15 seconds of mono 16 kHz WAV", 400);
|
|
55
|
+
if (wav.samples.some(sample => !Number.isFinite(sample)))
|
|
56
|
+
throw new SpeechError("invalid audio samples", 400);
|
|
57
|
+
if (!this.requests.check(`hear:${by}`, { allowed: 20, windowMs: 60_000 }).ok)
|
|
58
|
+
throw new SpeechError("too many speaker transcription requests", 429);
|
|
59
|
+
if (quietSamples(wav.samples))
|
|
60
|
+
return { language: "", seconds, turns: [] };
|
|
61
|
+
if (this.hearing.has(by) || this.hearing.size >= 4)
|
|
62
|
+
throw new SpeechError("speaker transcription is busy", 429);
|
|
63
|
+
this.hearing.add(by);
|
|
64
|
+
try {
|
|
65
|
+
const billed = Math.ceil(seconds);
|
|
66
|
+
await this.reserve(`scribe:user:${by}`, billed, 300, 60_000);
|
|
67
|
+
await this.reserve("scribe:server", billed, 600, 60_000);
|
|
68
|
+
await this.reserve(`scribe:user:${by}`, billed, this.userDailyAudioSeconds, 86_400_000);
|
|
69
|
+
await this.reserve("scribe:server", billed, this.dailyAudioSeconds, 86_400_000);
|
|
70
|
+
signal?.throwIfAborted();
|
|
71
|
+
const form = new FormData();
|
|
72
|
+
// Canonical PCM prevents a crafted container from billing more audio
|
|
73
|
+
// than the duration we validated, and removes uploaded metadata.
|
|
74
|
+
form.set("file", new Blob([new Uint8Array(encodeWav(wav.samples))], { type: "audio/wav" }), "listening.wav");
|
|
75
|
+
form.set("model_id", "scribe_v2");
|
|
76
|
+
form.set("diarize", "true");
|
|
77
|
+
form.set("tag_audio_events", "false");
|
|
78
|
+
form.set("timestamps_granularity", "word");
|
|
79
|
+
// No language_code: preserve the source language, including Spanish.
|
|
80
|
+
const answer = await this.fetcher("https://api.elevenlabs.io/v1/speech-to-text", {
|
|
81
|
+
method: "POST", headers: { "xi-api-key": this.key }, body: form,
|
|
82
|
+
signal: signal ? AbortSignal.any([signal, AbortSignal.timeout(10_000)]) : AbortSignal.timeout(10_000),
|
|
83
|
+
});
|
|
84
|
+
if (!answer.ok)
|
|
85
|
+
throw new SpeechError("speaker transcription could not run; check provider quota and permissions", answer.status === 429 ? 429 : 502);
|
|
86
|
+
return speakerTurns(await answer.json(), wav);
|
|
87
|
+
}
|
|
88
|
+
finally {
|
|
89
|
+
this.hearing.delete(by);
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
async voices() {
|
|
93
|
+
if (!this.available())
|
|
94
|
+
throw new SpeechError("translated audio needs ELEVENLABS_API_KEY on the account server", 503);
|
|
95
|
+
if (!this.catalog || this.now() >= this.catalogUntil) {
|
|
96
|
+
this.catalogUntil = this.now() + 3600_000;
|
|
97
|
+
this.catalog = this.fetcher("https://api.elevenlabs.io/v2/voices?page_size=100&voice_type=default", {
|
|
98
|
+
headers: { "xi-api-key": this.key }, signal: AbortSignal.timeout(8000),
|
|
99
|
+
}).then(async (response) => {
|
|
100
|
+
if (!response.ok)
|
|
101
|
+
throw new SpeechError("ElevenLabs could not list voices; check the server key and its voice permissions", 503);
|
|
102
|
+
const body = await response.json();
|
|
103
|
+
return body.voices.filter(voice => /^[a-zA-Z0-9_-]+$/.test(voice.voice_id)).map(voice => ({
|
|
104
|
+
id: voice.voice_id, name: voice.name, gender: voice.labels?.["gender"] ?? "neutral", language: voice.labels?.["language"] ?? "",
|
|
105
|
+
}));
|
|
106
|
+
}).catch(error => { this.catalog = null; throw error; });
|
|
107
|
+
}
|
|
108
|
+
return this.catalog;
|
|
109
|
+
}
|
|
110
|
+
/** A 90-second capability for one channel, never the viewer's account credential. */
|
|
111
|
+
async grant(by, channel) {
|
|
112
|
+
if (!/^[\w-]{1,80}$/.test(channel))
|
|
113
|
+
throw new SpeechError("choose a playback session or live channel", 400);
|
|
114
|
+
if (!this.requests.check(`grant:${by}`, { allowed: 10, windowMs: 60_000 }).ok)
|
|
115
|
+
throw new SpeechError("too many audio authorization requests", 429);
|
|
116
|
+
for (const [token, grant] of this.grants)
|
|
117
|
+
if (grant.expires <= this.now())
|
|
118
|
+
this.grants.delete(token);
|
|
119
|
+
if (this.grants.size >= 5000)
|
|
120
|
+
throw new SpeechError("audio authorization is busy", 503);
|
|
121
|
+
const token = `nxd_${randomBytes(32).toString("base64url")}`;
|
|
122
|
+
const expires = this.now() + 90_000;
|
|
123
|
+
if (this.db) {
|
|
124
|
+
await this.ensure();
|
|
125
|
+
await this.db.query("INSERT INTO live_voice_grants (token_hash, by_account, channel, expires_at, remaining) VALUES ($1, $2, $3, $4, 2000)", [createHash("sha256").update(token).digest("hex"), by, channel, new Date(expires)]);
|
|
126
|
+
}
|
|
127
|
+
else
|
|
128
|
+
this.grants.set(token, { by, channel, expires, remaining: 2000 });
|
|
129
|
+
return { token, expires };
|
|
130
|
+
}
|
|
131
|
+
async authorize(token, channel, chars) {
|
|
132
|
+
if (!/^nxd_[A-Za-z0-9_-]{43}$/.test(token))
|
|
133
|
+
throw new SpeechError("sign in to enable translated audio", 401);
|
|
134
|
+
if (!Number.isFinite(chars) || chars < 1 || chars > 600)
|
|
135
|
+
throw new SpeechError("invalid caption length", 400);
|
|
136
|
+
if (this.db) {
|
|
137
|
+
await this.ensure();
|
|
138
|
+
const result = await this.db.query(`UPDATE live_voice_grants SET remaining = remaining - $3
|
|
139
|
+
WHERE token_hash = $1 AND channel = $2 AND expires_at > $4 AND remaining >= $3 RETURNING by_account`, [createHash("sha256").update(token).digest("hex"), channel, chars, new Date(this.now())]);
|
|
140
|
+
if (!result.rows.length)
|
|
141
|
+
throw new SpeechError("audio authorization expired or reached its limit; enable translated audio again", 401);
|
|
142
|
+
return String(result.rows[0]?.["by_account"]);
|
|
143
|
+
}
|
|
144
|
+
const grant = this.grants.get(token);
|
|
145
|
+
if (!grant || grant.expires <= this.now() || grant.channel !== channel)
|
|
146
|
+
throw new SpeechError("sign in to renew translated audio", 401);
|
|
147
|
+
if (grant.remaining < chars)
|
|
148
|
+
throw new SpeechError("this audio authorization reached its character limit", 429);
|
|
149
|
+
grant.remaining -= chars;
|
|
150
|
+
return grant.by;
|
|
151
|
+
}
|
|
152
|
+
async ensure() {
|
|
153
|
+
if (!this.db)
|
|
154
|
+
return;
|
|
155
|
+
this.schema ??= (async () => {
|
|
156
|
+
await this.db.query("CREATE TABLE IF NOT EXISTS live_voice_usage (bucket TEXT PRIMARY KEY, chars BIGINT NOT NULL, expires_at TIMESTAMPTZ NOT NULL)");
|
|
157
|
+
await this.db.query("CREATE TABLE IF NOT EXISTS live_voice_grants (token_hash TEXT PRIMARY KEY, by_account TEXT NOT NULL, channel TEXT NOT NULL, expires_at TIMESTAMPTZ NOT NULL, remaining INTEGER NOT NULL)");
|
|
158
|
+
})().catch(error => { this.schema = null; throw error; });
|
|
159
|
+
await this.schema;
|
|
160
|
+
if (this.now() >= this.cleanupAt) {
|
|
161
|
+
this.cleanupAt = this.now() + 60_000;
|
|
162
|
+
// Keep only live windows; accounting survives restarts and is shared by replicas.
|
|
163
|
+
await this.db.query("DELETE FROM live_voice_usage WHERE expires_at <= $1", [new Date(this.now())]);
|
|
164
|
+
await this.db.query("DELETE FROM live_voice_grants WHERE expires_at <= $1", [new Date(this.now())]);
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
async reserve(bucket, chars, limit, windowMs) {
|
|
168
|
+
if (chars > limit)
|
|
169
|
+
throw new SpeechError("translated audio reached its character budget; try again later", 429);
|
|
170
|
+
const window = Math.floor(this.now() / windowMs);
|
|
171
|
+
const key = `${bucket}:${windowMs}:${window}`;
|
|
172
|
+
if (this.db) {
|
|
173
|
+
await this.ensure();
|
|
174
|
+
const result = await this.db.query(`INSERT INTO live_voice_usage (bucket, chars, expires_at) VALUES ($1, $2, $4)
|
|
175
|
+
ON CONFLICT (bucket) DO UPDATE SET chars = live_voice_usage.chars + EXCLUDED.chars
|
|
176
|
+
WHERE live_voice_usage.chars + EXCLUDED.chars <= $3 RETURNING chars`, [key, chars, limit, new Date((window + 1) * windowMs)]);
|
|
177
|
+
if (!result.rows.length)
|
|
178
|
+
throw new SpeechError("translated audio reached its character budget; try again later", 429);
|
|
179
|
+
return;
|
|
180
|
+
}
|
|
181
|
+
// Local/single-process installations without a database. Account servers pass their pool.
|
|
182
|
+
const record = this.usage.get(key) ?? { minute: (window + 1) * windowMs, chars: 0 };
|
|
183
|
+
if (record.chars + chars > limit)
|
|
184
|
+
throw new SpeechError("translated audio reached its character budget; try again later", 429);
|
|
185
|
+
record.chars += chars;
|
|
186
|
+
this.usage.set(key, record);
|
|
187
|
+
}
|
|
188
|
+
async charge(by, channel, chars) {
|
|
189
|
+
for (const [key, record] of this.usage)
|
|
190
|
+
if (record.minute <= this.now())
|
|
191
|
+
this.usage.delete(key);
|
|
192
|
+
await this.reserve(`user:${by}`, chars, this.charsPerMinute, 60_000);
|
|
193
|
+
await this.reserve(`channel:${channel}`, chars, 6000, 60_000);
|
|
194
|
+
await this.reserve("server", chars, 12_000, 60_000);
|
|
195
|
+
await this.reserve(`user:${by}`, chars, this.userDailyChars, 86400_000);
|
|
196
|
+
await this.reserve("server", chars, this.dailyChars, 86400_000);
|
|
197
|
+
}
|
|
198
|
+
checkRequest(by) {
|
|
199
|
+
if (!this.requests.check(`audio:${by}`, { allowed: 60, windowMs: 60_000 }).ok)
|
|
200
|
+
throw new SpeechError("too many translated audio requests", 429);
|
|
201
|
+
}
|
|
202
|
+
/** Stream the first request immediately; concurrent listeners share its cached result. */
|
|
203
|
+
async stream(ask, by, signal) {
|
|
204
|
+
this.checkRequest(by);
|
|
205
|
+
const text = typeof ask.text === "string" ? ask.text.trim() : "";
|
|
206
|
+
if (!text || text.length > 600)
|
|
207
|
+
throw new SpeechError("translated audio needs a caption of 1–600 characters", 400);
|
|
208
|
+
if (!LIVE_VOICE_LANGUAGES.has(ask.language))
|
|
209
|
+
throw new SpeechError("this language is not supported by Flash voices", 400);
|
|
210
|
+
const voices = await this.voices();
|
|
211
|
+
signal?.throwIfAborted();
|
|
212
|
+
const gender = ask.profile === "lower" ? "male" : ask.profile === "higher" ? "female" : "neutral";
|
|
213
|
+
const voice = ask.voice && ask.voice !== "auto"
|
|
214
|
+
? voices.find(voice => voice.id === ask.voice)
|
|
215
|
+
: voices.find(voice => voice.gender === gender) ?? voices[0];
|
|
216
|
+
if (!voice)
|
|
217
|
+
throw new SpeechError("choose an available voice", 400);
|
|
218
|
+
const id = createHash("sha256").update(JSON.stringify([LIVE_VOICE_MODEL, voice.id, ask.language, text])).digest("hex");
|
|
219
|
+
for (const [key, item] of this.cache)
|
|
220
|
+
if (item.until < this.now())
|
|
221
|
+
this.cache.delete(key);
|
|
222
|
+
const cached = this.cache.get(id);
|
|
223
|
+
if (cached)
|
|
224
|
+
return new Response(new Uint8Array(cached.bytes), { headers: HEADERS });
|
|
225
|
+
const pending = this.pending.get(id);
|
|
226
|
+
if (pending)
|
|
227
|
+
return new Response(new Uint8Array(await pending), { headers: HEADERS });
|
|
228
|
+
if (this.pending.size >= 4)
|
|
229
|
+
throw new SpeechError("translated audio is busy; waiting for the next caption", 429);
|
|
230
|
+
if ((this.activeBy.get(by) ?? 0) >= 2)
|
|
231
|
+
throw new SpeechError("two voice requests are already active for this account", 429);
|
|
232
|
+
// Register before the first provider await, so identical requests cannot both bill.
|
|
233
|
+
let done;
|
|
234
|
+
let fail;
|
|
235
|
+
const finished = new Promise((resolve, reject) => { done = resolve; fail = reject; });
|
|
236
|
+
this.pending.set(id, finished);
|
|
237
|
+
this.activeBy.set(by, (this.activeBy.get(by) ?? 0) + 1);
|
|
238
|
+
const release = () => { this.pending.delete(id); this.activeBy.set(by, Math.max(0, (this.activeBy.get(by) ?? 1) - 1)); if (!this.activeBy.get(by))
|
|
239
|
+
this.activeBy.delete(by); };
|
|
240
|
+
void finished.catch(() => undefined);
|
|
241
|
+
try {
|
|
242
|
+
await this.charge(by, ask.channel ?? "direct", text.length);
|
|
243
|
+
signal?.throwIfAborted();
|
|
244
|
+
const response = await this.fetcher(`https://api.elevenlabs.io/v1/text-to-speech/${voice.id}/stream?output_format=pcm_16000`, {
|
|
245
|
+
method: "POST",
|
|
246
|
+
headers: { "xi-api-key": this.key, "content-type": "application/json" },
|
|
247
|
+
body: JSON.stringify({ text, model_id: LIVE_VOICE_MODEL, language_code: ask.language }),
|
|
248
|
+
signal: signal ? AbortSignal.any([signal, AbortSignal.timeout(10_000)]) : AbortSignal.timeout(10_000),
|
|
249
|
+
});
|
|
250
|
+
if (!response.ok || !response.body)
|
|
251
|
+
throw new SpeechError(response.status === 429 ? "ElevenLabs audio quota is temporarily exhausted" : "ElevenLabs could not generate audio; check the server key and quota", response.status === 429 ? 429 : 502);
|
|
252
|
+
const [play, keep] = response.body.tee();
|
|
253
|
+
void (async () => {
|
|
254
|
+
const reader = keep.getReader();
|
|
255
|
+
const chunks = [];
|
|
256
|
+
let size = 0;
|
|
257
|
+
try {
|
|
258
|
+
while (true) {
|
|
259
|
+
const { done: ended, value } = await reader.read();
|
|
260
|
+
if (ended)
|
|
261
|
+
break;
|
|
262
|
+
size += value.length;
|
|
263
|
+
if (size > 2 * 1024 * 1024) {
|
|
264
|
+
await reader.cancel();
|
|
265
|
+
throw new SpeechError("voice response was too long", 502);
|
|
266
|
+
}
|
|
267
|
+
chunks.push(value);
|
|
268
|
+
}
|
|
269
|
+
if (!size || size % 2)
|
|
270
|
+
throw new SpeechError("voice response contained incomplete audio", 502);
|
|
271
|
+
const bytes = new Uint8Array(size);
|
|
272
|
+
let at = 0;
|
|
273
|
+
for (const chunk of chunks) {
|
|
274
|
+
bytes.set(chunk, at);
|
|
275
|
+
at += chunk.length;
|
|
276
|
+
}
|
|
277
|
+
this.cache.set(id, { until: this.now() + 60_000, bytes });
|
|
278
|
+
// At most a few MB for recent lines, never a recording archive.
|
|
279
|
+
while (this.cache.size > 32)
|
|
280
|
+
this.cache.delete(this.cache.keys().next().value);
|
|
281
|
+
done(bytes);
|
|
282
|
+
}
|
|
283
|
+
catch (error) {
|
|
284
|
+
fail(error);
|
|
285
|
+
}
|
|
286
|
+
finally {
|
|
287
|
+
reader.releaseLock();
|
|
288
|
+
release();
|
|
289
|
+
}
|
|
290
|
+
})();
|
|
291
|
+
return new Response(play, { headers: HEADERS });
|
|
292
|
+
}
|
|
293
|
+
catch (error) {
|
|
294
|
+
fail(error);
|
|
295
|
+
release();
|
|
296
|
+
throw error;
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
}
|
package/dist/server.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { type IncomingMessage, type Server, type ServerResponse } from "node:http";
|
|
2
|
+
import { LiveVoice } from "./live-voice.ts";
|
|
2
3
|
import { Connections } from "./connections.ts";
|
|
3
4
|
import { Broadcaster, type Destination, type EncoderSettings } from "./broadcast.ts";
|
|
4
5
|
import { Ingest } from "./ingest.ts";
|
|
@@ -703,6 +704,7 @@ export interface HandlerOptions {
|
|
|
703
704
|
trollbox?: Trollbox;
|
|
704
705
|
/** Speech to text: a line said out loud, heard here. Needs the optional model. */
|
|
705
706
|
speech?: Speech;
|
|
707
|
+
liveVoice?: LiveVoice;
|
|
706
708
|
/** Translation: texts in another language, by a model here. The same optional library. */
|
|
707
709
|
translator?: Translator;
|
|
708
710
|
/** The transcript store: what was heard, kept under the media's identity. Where the accounts are. */
|
package/dist/server.js
CHANGED
|
@@ -16,6 +16,9 @@ import { randomBytes } from "node:crypto";
|
|
|
16
16
|
import { hostname } from "node:os";
|
|
17
17
|
import { spawn, spawnSync } from "node:child_process";
|
|
18
18
|
import { readFileSync } from "node:fs";
|
|
19
|
+
import { Readable } from "node:stream";
|
|
20
|
+
import { pipeline as pipeStream } from "node:stream/promises";
|
|
21
|
+
import { LiveVoice, LIVE_VOICE_LANGUAGES, LIVE_VOICE_MODEL } from "./live-voice.js";
|
|
19
22
|
import { Connections } from "./connections.js";
|
|
20
23
|
import { Broadcaster, DEFAULT_ENCODER, PRESETS, redact, } from "./broadcast.js";
|
|
21
24
|
import { Ingest, normaliseFormat } from "./ingest.js";
|
|
@@ -65,7 +68,7 @@ import { Tickets, needsTicket, ticketFrom, ticketsFromEnv } from "./tickets.js";
|
|
|
65
68
|
import { Layouts } from "./layouts.js";
|
|
66
69
|
import { Rooms } from "./rooms.js";
|
|
67
70
|
import { Trollbox, TrollboxError, fallbackHandle, roomFor } from "./trollbox.js";
|
|
68
|
-
import { MAX_BYTES as SPEECH_BYTES, Speech, SpeechError, isWav, languageOf } from "./speech.js";
|
|
71
|
+
import { MAX_BYTES as SPEECH_BYTES, NATIVE_REVISION, Speech, SpeechError, isWav, languageOf } from "./speech.js";
|
|
69
72
|
import { Captions } from "./captions.js";
|
|
70
73
|
import { LANGUAGES, Translator } from "./translate.js";
|
|
71
74
|
import { StoredTranslations } from "./translate-jobs.js";
|
|
@@ -1106,7 +1109,7 @@ const CORS = {
|
|
|
1106
1109
|
// paths and takes six commands; binding to 127.0.0.1 is what keeps it shut.
|
|
1107
1110
|
"access-control-allow-origin": "*",
|
|
1108
1111
|
"access-control-allow-methods": "GET, POST, PUT, DELETE, OPTIONS",
|
|
1109
|
-
"access-control-allow-headers": "content-type",
|
|
1112
|
+
"access-control-allow-headers": "content-type, authorization",
|
|
1110
1113
|
"access-control-max-age": "86400",
|
|
1111
1114
|
};
|
|
1112
1115
|
/**
|
|
@@ -1278,6 +1281,8 @@ export function joinDocument(shell, subject, site) {
|
|
|
1278
1281
|
.replace("</head>", `${card}<meta name="nixamp-shell-title" content="${htmlText(shellTitle)}" />\n </head>`);
|
|
1279
1282
|
}
|
|
1280
1283
|
function json(response, code, body) {
|
|
1284
|
+
if (code === 429)
|
|
1285
|
+
response.setHeader("retry-after", "60");
|
|
1281
1286
|
const text = JSON.stringify(body);
|
|
1282
1287
|
response.writeHead(code, {
|
|
1283
1288
|
...CORS,
|
|
@@ -1287,6 +1292,18 @@ function json(response, code, body) {
|
|
|
1287
1292
|
});
|
|
1288
1293
|
response.end(text);
|
|
1289
1294
|
}
|
|
1295
|
+
/** Stream PCM without waiting for a complete utterance, respecting browser backpressure. */
|
|
1296
|
+
async function voiceResponse(response, upstream) {
|
|
1297
|
+
response.writeHead(upstream.status, {
|
|
1298
|
+
...CORS, "content-type": upstream.headers.get("content-type") ?? "application/json",
|
|
1299
|
+
"cache-control": "no-store", "x-accel-buffering": "no",
|
|
1300
|
+
});
|
|
1301
|
+
if (!upstream.body) {
|
|
1302
|
+
response.end();
|
|
1303
|
+
return;
|
|
1304
|
+
}
|
|
1305
|
+
await pipeStream(Readable.fromWeb(upstream.body), response);
|
|
1306
|
+
}
|
|
1290
1307
|
async function readBody(request, limit = 64 * 1024) {
|
|
1291
1308
|
const chunks = [];
|
|
1292
1309
|
let size = 0;
|
|
@@ -2314,6 +2331,80 @@ export function createHandler(engine, options) {
|
|
|
2314
2331
|
* room, the words go straight into that room's trollbox as a line by
|
|
2315
2332
|
* whoever spoke them. The page, the CLI and the MCP tools all come here.
|
|
2316
2333
|
*/
|
|
2334
|
+
if (["/api/v1/speech/voices", "/api/v1/speech/grant", "/api/v1/speech/synthesize", "/api/v1/speech/speakers"].includes(path) && options.accounts) {
|
|
2335
|
+
const controller = new AbortController();
|
|
2336
|
+
response.once("close", () => controller.abort());
|
|
2337
|
+
try {
|
|
2338
|
+
const caller = callerOf(request.headers, request.socket.remoteAddress, options.behindProxy ?? false);
|
|
2339
|
+
if (!guard.check(`voice-ip:${caller}`, { allowed: 600, windowMs: 60_000 }).ok) {
|
|
2340
|
+
json(response, 429, { error: "too many voice requests" });
|
|
2341
|
+
return;
|
|
2342
|
+
}
|
|
2343
|
+
const service = options.liveVoice;
|
|
2344
|
+
if (!service?.available()) {
|
|
2345
|
+
json(response, 503, { error: "translated audio needs ELEVENLABS_API_KEY on the account server" });
|
|
2346
|
+
return;
|
|
2347
|
+
}
|
|
2348
|
+
if (path.endsWith("/synthesize") && request.method === "POST") {
|
|
2349
|
+
let body;
|
|
2350
|
+
try {
|
|
2351
|
+
body = JSON.parse(await readBody(request, 4096));
|
|
2352
|
+
}
|
|
2353
|
+
catch {
|
|
2354
|
+
json(response, 400, { error: "send a caption as JSON" });
|
|
2355
|
+
return;
|
|
2356
|
+
}
|
|
2357
|
+
if (!body || typeof body !== "object") {
|
|
2358
|
+
json(response, 400, { error: "send a caption as JSON" });
|
|
2359
|
+
return;
|
|
2360
|
+
}
|
|
2361
|
+
const by = await service.authorize(tokenFrom(request.headers), body.channel ?? "", typeof body.text === "string" ? body.text.length : 0);
|
|
2362
|
+
await voiceResponse(response, await service.stream(body, by, controller.signal));
|
|
2363
|
+
}
|
|
2364
|
+
else {
|
|
2365
|
+
const who = await options.accounts.whoIs(tokenFrom(request.headers));
|
|
2366
|
+
if (!who) {
|
|
2367
|
+
json(response, 401, { error: "sign in to use translated audio" });
|
|
2368
|
+
return;
|
|
2369
|
+
}
|
|
2370
|
+
if (path.endsWith("/voices") && request.method === "GET") {
|
|
2371
|
+
json(response, 200, { voices: await service.voices(), model: LIVE_VOICE_MODEL, languages: [...LIVE_VOICE_LANGUAGES] });
|
|
2372
|
+
}
|
|
2373
|
+
else if (path.endsWith("/speakers") && request.method === "POST") {
|
|
2374
|
+
if (!request.headers["content-type"]?.startsWith("audio/wav")) {
|
|
2375
|
+
json(response, 415, { error: "send a mono 16 kHz WAV" });
|
|
2376
|
+
return;
|
|
2377
|
+
}
|
|
2378
|
+
const bytes = await readBytes(request, 484_000);
|
|
2379
|
+
json(response, 200, await service.hear(bytes, who.id, controller.signal));
|
|
2380
|
+
}
|
|
2381
|
+
else if (path.endsWith("/grant") && request.method === "POST") {
|
|
2382
|
+
if (!request.headers["content-type"]?.startsWith("application/json")) {
|
|
2383
|
+
json(response, 415, { error: "send JSON" });
|
|
2384
|
+
return;
|
|
2385
|
+
}
|
|
2386
|
+
let body;
|
|
2387
|
+
try {
|
|
2388
|
+
body = JSON.parse(await readBody(request, 1024));
|
|
2389
|
+
}
|
|
2390
|
+
catch {
|
|
2391
|
+
json(response, 400, { error: "send a channel as JSON" });
|
|
2392
|
+
return;
|
|
2393
|
+
}
|
|
2394
|
+
json(response, 200, await service.grant(who.id, typeof body?.channel === "string" ? body.channel : ""));
|
|
2395
|
+
}
|
|
2396
|
+
else
|
|
2397
|
+
json(response, 405, { error: "GET voices or POST an audio request" });
|
|
2398
|
+
}
|
|
2399
|
+
}
|
|
2400
|
+
catch (error) {
|
|
2401
|
+
if (response.headersSent)
|
|
2402
|
+
response.destroy();
|
|
2403
|
+
else
|
|
2404
|
+
json(response, error instanceof SpeechError ? error.status : 502, { error: error instanceof SpeechError ? error.message : "translated audio is unavailable" });
|
|
2405
|
+
}
|
|
2406
|
+
return;
|
|
2407
|
+
}
|
|
2317
2408
|
if (path === "/api/v1/speech/transcribe" && options.speech && options.accounts) {
|
|
2318
2409
|
const speech = options.speech;
|
|
2319
2410
|
try {
|
|
@@ -2343,6 +2434,7 @@ export function createHandler(engine, options) {
|
|
|
2343
2434
|
// Pieces with their timing, for a whole file being written down.
|
|
2344
2435
|
timestamps: url.searchParams.get("timestamps") === "1",
|
|
2345
2436
|
by: who.id,
|
|
2437
|
+
...(url.searchParams.get("live") === "1" ? { deadline: Date.now() + 10_000 } : {}),
|
|
2346
2438
|
});
|
|
2347
2439
|
// The language travels back: as told, or as the ear guessed it, so
|
|
2348
2440
|
// a captioner can say it next time and a transcript can be kept as it.
|
|
@@ -2563,7 +2655,7 @@ export function createHandler(engine, options) {
|
|
|
2563
2655
|
return;
|
|
2564
2656
|
}
|
|
2565
2657
|
try {
|
|
2566
|
-
json(response, 200, await translator.translate(texts, from, to, { by: who.id }));
|
|
2658
|
+
json(response, 200, await translator.translate(texts, from, to, { by: who.id, ...(url.searchParams.get("live") === "1" ? { deadline: Date.now() + 10_000 } : {}) }));
|
|
2567
2659
|
}
|
|
2568
2660
|
catch (error) {
|
|
2569
2661
|
if (error instanceof SpeechError)
|
|
@@ -2620,6 +2712,12 @@ export function createHandler(engine, options) {
|
|
|
2620
2712
|
json(response, 400, { error: "format is json, srt, vtt or txt" });
|
|
2621
2713
|
return;
|
|
2622
2714
|
}
|
|
2715
|
+
// Joining a live must not start translating an entire old transcript.
|
|
2716
|
+
if (url.searchParams.get("cached") === "1") {
|
|
2717
|
+
const saved = await store.get(id, language);
|
|
2718
|
+
json(response, saved ? 200 : 404, saved ? wire(saved) : { error: "no cached transcript" });
|
|
2719
|
+
return;
|
|
2720
|
+
}
|
|
2623
2721
|
const answer = await options.translations.get(id, language, who.id);
|
|
2624
2722
|
if (answer.status !== 200 && answer.status !== 202) {
|
|
2625
2723
|
json(response, answer.status, { error: answer.error });
|
|
@@ -3930,6 +4028,44 @@ export function createHandler(engine, options) {
|
|
|
3930
4028
|
* started by the first person asking. `captions` is the live stream
|
|
3931
4029
|
* of lines; `transcript` is the recent ones as JSON, for a poll.
|
|
3932
4030
|
*/
|
|
4031
|
+
if ((action === "voice" || action === "voice-options") && request.method === "GET") {
|
|
4032
|
+
const controller = new AbortController();
|
|
4033
|
+
response.once("close", () => controller.abort());
|
|
4034
|
+
const timeout = setTimeout(() => controller.abort(), 12_000);
|
|
4035
|
+
try {
|
|
4036
|
+
const caller = callerOf(request.headers, request.socket.remoteAddress, options.behindProxy ?? false);
|
|
4037
|
+
if (!guard.check(`channel-voice-ip:${caller}`, { allowed: 60, windowMs: 60_000 }).ok || !guard.check(`channel-voice:${id}`, { allowed: 240, windowMs: 60_000 }).ok) {
|
|
4038
|
+
json(response, 429, { error: "too many translated audio requests" });
|
|
4039
|
+
return;
|
|
4040
|
+
}
|
|
4041
|
+
if (!channels.has(id)) {
|
|
4042
|
+
json(response, 404, { error: "nothing is playing on that channel" });
|
|
4043
|
+
return;
|
|
4044
|
+
}
|
|
4045
|
+
if (!options.captions) {
|
|
4046
|
+
json(response, 503, { error: "this server cannot caption" });
|
|
4047
|
+
return;
|
|
4048
|
+
}
|
|
4049
|
+
const at = action === "voice-options" ? null : Number(url.searchParams.get("at"));
|
|
4050
|
+
const language = languageCode(url.searchParams.get("language"));
|
|
4051
|
+
if ((at !== null && (!url.searchParams.has("at") || !Number.isFinite(at))) || language === null) {
|
|
4052
|
+
json(response, 400, { error: "audio needs a caption time and language" });
|
|
4053
|
+
return;
|
|
4054
|
+
}
|
|
4055
|
+
const upstream = await options.captions.voiceRequest(id, at, language, url.searchParams.get("voice") ?? "auto", controller.signal, tokenFrom(request.headers));
|
|
4056
|
+
await voiceResponse(response, upstream);
|
|
4057
|
+
}
|
|
4058
|
+
catch (error) {
|
|
4059
|
+
if (response.headersSent)
|
|
4060
|
+
response.destroy();
|
|
4061
|
+
else
|
|
4062
|
+
json(response, error instanceof SpeechError ? error.status : 502, { error: error instanceof SpeechError ? error.message : "translated audio is unavailable" });
|
|
4063
|
+
}
|
|
4064
|
+
finally {
|
|
4065
|
+
clearTimeout(timeout);
|
|
4066
|
+
}
|
|
4067
|
+
return;
|
|
4068
|
+
}
|
|
3933
4069
|
if ((action === "captions" || action === "transcript") && request.method === "GET") {
|
|
3934
4070
|
const captions = options.captions;
|
|
3935
4071
|
if (!captions) {
|
|
@@ -3940,6 +4076,11 @@ export function createHandler(engine, options) {
|
|
|
3940
4076
|
json(response, 503, { error: "this server is not signed in to nixamp.com, so it cannot caption; run `nixamp login` on it" });
|
|
3941
4077
|
return;
|
|
3942
4078
|
}
|
|
4079
|
+
const caller = callerOf(request.headers, request.socket.remoteAddress, options.behindProxy ?? false);
|
|
4080
|
+
if (!guard.check(`captions:${caller}`, { allowed: 120, windowMs: 60_000 }).ok || !captions.capacity(id)) {
|
|
4081
|
+
json(response, 429, { error: "captioning is at capacity; try again in a moment" });
|
|
4082
|
+
return;
|
|
4083
|
+
}
|
|
3943
4084
|
if (!channels.has(id)) {
|
|
3944
4085
|
json(response, 404, { error: "nothing is playing on that channel" });
|
|
3945
4086
|
return;
|
|
@@ -3958,7 +4099,7 @@ export function createHandler(engine, options) {
|
|
|
3958
4099
|
// Asking keeps the captioner up: it stops a minute after the last ask.
|
|
3959
4100
|
captions.subscribe(id, () => undefined, wanted)?.();
|
|
3960
4101
|
json(response, 200, {
|
|
3961
|
-
channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted),
|
|
4102
|
+
channel: id, recognitionRevision: NATIVE_REVISION, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted),
|
|
3962
4103
|
});
|
|
3963
4104
|
return;
|
|
3964
4105
|
}
|
|
@@ -3979,7 +4120,7 @@ export function createHandler(engine, options) {
|
|
|
3979
4120
|
}
|
|
3980
4121
|
// How far behind the live edge a newcomer's playback starts, so the
|
|
3981
4122
|
// page can hold each line until its own sound gets there.
|
|
3982
|
-
write("hello", { channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted) });
|
|
4123
|
+
write("hello", { channel: id, recognitionRevision: NATIVE_REVISION, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted) });
|
|
3983
4124
|
const beat = setInterval(() => response.write(": beat\n\n"), 20_000);
|
|
3984
4125
|
beat.unref?.();
|
|
3985
4126
|
const done = () => {
|
|
@@ -5276,6 +5417,13 @@ export async function serve(argv, version = "0.1.0") {
|
|
|
5276
5417
|
// the first person to speak does not wait for the model to arrive; a box
|
|
5277
5418
|
// without the optional model never says it is ready, and answers 503.
|
|
5278
5419
|
const speech = accounts && process.env["NIXAMP_STT"] !== "off" ? new Speech() : undefined;
|
|
5420
|
+
const liveVoice = accounts && process.env["NIXAMP_DUBBING"] !== "off" ? new LiveVoice({
|
|
5421
|
+
...(pool ? { db: pool } : {}),
|
|
5422
|
+
dailyChars: Number(process.env["NIXAMP_DUB_DAILY_CHARS"] ?? 200_000),
|
|
5423
|
+
dailyAudioSeconds: Number(process.env["NIXAMP_DUB_DAILY_AUDIO_SECONDS"] ?? 86_400),
|
|
5424
|
+
userDailyChars: Number(process.env["NIXAMP_DUB_USER_DAILY_CHARS"] ?? 120_000),
|
|
5425
|
+
userDailyAudioSeconds: Number(process.env["NIXAMP_DUB_USER_DAILY_AUDIO_SECONDS"] ?? 43_200),
|
|
5426
|
+
}) : undefined;
|
|
5279
5427
|
if (speech) {
|
|
5280
5428
|
void speech.warm().then((ready) => {
|
|
5281
5429
|
console.error(ready ? `nixamp: hearing with ${speech.model}` : `nixamp: not hearing: ${speech.lastFailure}`);
|
|
@@ -5759,6 +5907,7 @@ export async function serve(argv, version = "0.1.0") {
|
|
|
5759
5907
|
...(rooms ? { rooms } : {}),
|
|
5760
5908
|
...(trollbox ? { trollbox } : {}),
|
|
5761
5909
|
...(speech ? { speech } : {}),
|
|
5910
|
+
...(liveVoice ? { liveVoice } : {}),
|
|
5762
5911
|
...(translator ? { translator } : {}),
|
|
5763
5912
|
...(transcripts ? { transcripts } : {}),
|
|
5764
5913
|
...(translations ? { translations } : {}),
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
/** Provider speaker IDs belong to a clip. Clients reconcile the overlapping
|
|
2
|
+
* word timestamps before assigning a voice; these are not people's identities. */
|
|
3
|
+
import { type Wav } from "./speech.ts";
|
|
4
|
+
import { type VoiceProfile } from "./voice-profile.ts";
|
|
5
|
+
export interface SpeakerTurn {
|
|
6
|
+
start: number;
|
|
7
|
+
end: number;
|
|
8
|
+
text: string;
|
|
9
|
+
speaker: string;
|
|
10
|
+
profile: VoiceProfile;
|
|
11
|
+
words: {
|
|
12
|
+
text: string;
|
|
13
|
+
start: number;
|
|
14
|
+
end: number;
|
|
15
|
+
}[];
|
|
16
|
+
}
|
|
17
|
+
export interface SpeakerTranscript {
|
|
18
|
+
language: string;
|
|
19
|
+
seconds: number;
|
|
20
|
+
turns: SpeakerTurn[];
|
|
21
|
+
}
|
|
22
|
+
export interface ScribeResult {
|
|
23
|
+
language_code?: string;
|
|
24
|
+
words?: {
|
|
25
|
+
text: string;
|
|
26
|
+
start: number;
|
|
27
|
+
end: number;
|
|
28
|
+
type: string;
|
|
29
|
+
speaker_id?: string | null;
|
|
30
|
+
}[];
|
|
31
|
+
}
|
|
32
|
+
export declare function speakerTurns(body: ScribeResult, wav: Wav): SpeakerTranscript;
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/** Provider speaker IDs belong to a clip. Clients reconcile the overlapping
|
|
2
|
+
* word timestamps before assigning a voice; these are not people's identities. */
|
|
3
|
+
import { reliableText } from "./speech.js";
|
|
4
|
+
import { voiceProfile } from "./voice-profile.js";
|
|
5
|
+
const ISO = Object.fromEntries("eng:en spa:es deu:de ger:de fra:fr fre:fr por:pt ita:it nld:nl dut:nl swe:sv dan:da fin:fi rus:ru ukr:uk ces:cs cze:cs hun:hu cmn:zh zho:zh ara:ar hin:hi vie:vi ind:id jpn:ja kor:ko pol:pl tur:tr ron:ro bul:bg ell:el nor:no nob:no nno:no".split(" ").map(pair => pair.split(":")));
|
|
6
|
+
export function speakerTurns(body, wav) {
|
|
7
|
+
const seconds = wav.samples.length / wav.rate;
|
|
8
|
+
const language = ISO[body.language_code ?? ""] ?? body.language_code ?? "";
|
|
9
|
+
const turns = [];
|
|
10
|
+
for (const word of (body.words ?? []).slice(0, 1500)) {
|
|
11
|
+
if (word.type !== "word" || typeof word.text !== "string" || !Number.isFinite(word.start) || !Number.isFinite(word.end) || word.start < 0 || word.end < word.start || word.end > seconds + 0.5)
|
|
12
|
+
continue;
|
|
13
|
+
const speaker = typeof word.speaker_id === "string" && /^[\w-]{1,80}$/.test(word.speaker_id) ? word.speaker_id : "unknown";
|
|
14
|
+
const last = turns.at(-1);
|
|
15
|
+
if (last && last.speaker === speaker && word.start - last.end < 0.8 && last.text.length + word.text.length < 400) {
|
|
16
|
+
last.text += ` ${word.text}`;
|
|
17
|
+
last.end = Math.max(last.end, word.end);
|
|
18
|
+
last.words.push({ text: word.text, start: word.start, end: word.end });
|
|
19
|
+
}
|
|
20
|
+
else
|
|
21
|
+
turns.push({ start: word.start, end: word.end, text: word.text, speaker, profile: "unknown", words: [{ text: word.text, start: word.start, end: word.end }] });
|
|
22
|
+
}
|
|
23
|
+
const pcm = Buffer.alloc(wav.samples.length * 2);
|
|
24
|
+
wav.samples.forEach((sample, i) => pcm.writeInt16LE(Math.round(Math.max(-1, Math.min(1, sample)) * 32767), i * 2));
|
|
25
|
+
for (const turn of turns) {
|
|
26
|
+
turn.text = reliableText(turn.text, Math.max(1, turn.end - turn.start));
|
|
27
|
+
turn.profile = voiceProfile(pcm.subarray(Math.floor(turn.start * 16000) * 2, Math.ceil(turn.end * 16000) * 2));
|
|
28
|
+
}
|
|
29
|
+
return { language, seconds, turns: turns.filter(turn => turn.text !== "") };
|
|
30
|
+
}
|