nixamp 0.25.0 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +108 -7
  2. package/dist/captions.d.ts +7 -0
  3. package/dist/captions.js +92 -28
  4. package/dist/live-voice.d.ts +70 -0
  5. package/dist/live-voice.js +299 -0
  6. package/dist/server.d.ts +2 -0
  7. package/dist/server.js +154 -5
  8. package/dist/speaker-turns.d.ts +32 -0
  9. package/dist/speaker-turns.js +30 -0
  10. package/dist/speech.d.ts +47 -0
  11. package/dist/speech.js +49 -4
  12. package/dist/transcript-client.d.ts +2 -2
  13. package/dist/transcript-client.js +5 -2
  14. package/dist/transcripts.d.ts +5 -0
  15. package/dist/transcripts.js +8 -3
  16. package/dist/translate-jobs.js +13 -7
  17. package/dist/translate.d.ts +2 -0
  18. package/dist/translate.js +9 -0
  19. package/dist/voice-profile.d.ts +7 -0
  20. package/dist/voice-profile.js +53 -0
  21. package/dist/warm.js +3 -3
  22. package/package.json +1 -1
  23. package/src/captions.ts +86 -25
  24. package/src/live-voice.ts +255 -0
  25. package/src/server.ts +104 -5
  26. package/src/speaker-turns.ts +34 -0
  27. package/src/speech.ts +46 -7
  28. package/src/transcript-client.ts +4 -0
  29. package/src/transcripts.ts +13 -3
  30. package/src/translate-jobs.ts +13 -7
  31. package/src/translate.ts +7 -1
  32. package/src/voice-profile.ts +46 -0
  33. package/src/warm.ts +3 -3
  34. package/web/dist/assets/{hls-3VKVEQE3-B4ltbKDh.js → hls-3VKVEQE3-vgax_tk1.js} +1 -1
  35. package/web/dist/assets/index-BzjrTOLf.js +1 -0
  36. package/web/dist/assets/index-D2Iy07pG.css +1 -0
  37. package/web/dist/assets/{mpegts-Byy3EkfT.js → mpegts-DmcUOiHq.js} +1 -1
  38. package/web/dist/assets/{mpegts-LO6RVLD6-C9qqolrW.js → mpegts-LO6RVLD6-CH3EQi6L.js} +1 -1
  39. package/web/dist/index.html +42 -22
  40. package/web/dist/sw.js +6 -6
  41. package/web/dist/assets/index-d7TvpeFZ.js +0 -1
  42. package/web/dist/assets/index-oyp61Kly.css +0 -1
@@ -0,0 +1,299 @@
1
+ /** ElevenLabs Flash for live captions. The key stays on the account server. */
2
+ import { createHash, randomBytes } from "node:crypto";
3
+ import { SpeechError, decodeWav, encodeWav, quietSamples } from "./speech.js";
4
+ import { speakerTurns } from "./speaker-turns.js";
5
+ import { Guard } from "./guard.js";
6
+ export const LIVE_VOICE_MODEL = "eleven_flash_v2_5";
7
+ export const LIVE_VOICE_RATE = 16_000;
8
+ export const LIVE_VOICE_LANGUAGES = new Set("en ja zh de hi fr ko pt it es id nl tr fil pl sv bg ro ar cs el fi hr ms sk da ta uk ru hu no vi".split(" "));
9
+ const HEADERS = { "content-type": "audio/pcm", "cache-control": "no-store", "x-audio-sample-rate": String(LIVE_VOICE_RATE) };
10
+ const budget = (value, fallback) => Number.isFinite(value) && value >= 0 ? Math.floor(value) : fallback;
11
+ export class LiveVoice {
12
+ key;
13
+ fetcher;
14
+ now;
15
+ catalog = null;
16
+ catalogUntil = 0;
17
+ cache = new Map();
18
+ pending = new Map();
19
+ usage = new Map();
20
+ charsPerMinute;
21
+ requests;
22
+ grants = new Map();
23
+ activeBy = new Map();
24
+ db;
25
+ schema = null;
26
+ dailyChars;
27
+ cleanupAt = 0;
28
+ hearing = new Set();
29
+ dailyAudioSeconds;
30
+ userDailyChars;
31
+ userDailyAudioSeconds;
32
+ constructor(options = {}) {
33
+ this.key = options.apiKey ?? process.env["ELEVENLABS_API_KEY"] ?? "";
34
+ this.fetcher = options.fetcher ?? fetch;
35
+ this.now = options.now ?? Date.now;
36
+ this.charsPerMinute = budget(options.charsPerMinute, 3000);
37
+ this.dailyChars = budget(options.dailyChars, 200_000);
38
+ this.dailyAudioSeconds = budget(options.dailyAudioSeconds, 86_400);
39
+ this.userDailyChars = budget(options.userDailyChars, 120_000);
40
+ this.userDailyAudioSeconds = budget(options.userDailyAudioSeconds, 43_200);
41
+ this.requests = new Guard(this.now);
42
+ this.db = options.db;
43
+ }
44
+ available() { return this.key !== ""; }
45
+ /** Optional diarization, billed only while a signed-in listener requests it.
46
+ * Rolling audio is bounded to 15 seconds, including overlap. Every second
47
+ * submitted (also repeated context) consumes the persistent provider budget. */
48
+ async hear(bytes, by, signal) {
49
+ if (!this.available())
50
+ throw new SpeechError("speaker voices are unavailable", 503);
51
+ const wav = decodeWav(bytes);
52
+ const seconds = wav.samples.length / wav.rate;
53
+ if (wav.rate !== 16000 || wav.channels !== 1 || seconds < 0.2 || seconds > 15.1)
54
+ throw new SpeechError("send up to 15 seconds of mono 16 kHz WAV", 400);
55
+ if (wav.samples.some(sample => !Number.isFinite(sample)))
56
+ throw new SpeechError("invalid audio samples", 400);
57
+ if (!this.requests.check(`hear:${by}`, { allowed: 20, windowMs: 60_000 }).ok)
58
+ throw new SpeechError("too many speaker transcription requests", 429);
59
+ if (quietSamples(wav.samples))
60
+ return { language: "", seconds, turns: [] };
61
+ if (this.hearing.has(by) || this.hearing.size >= 4)
62
+ throw new SpeechError("speaker transcription is busy", 429);
63
+ this.hearing.add(by);
64
+ try {
65
+ const billed = Math.ceil(seconds);
66
+ await this.reserve(`scribe:user:${by}`, billed, 300, 60_000);
67
+ await this.reserve("scribe:server", billed, 600, 60_000);
68
+ await this.reserve(`scribe:user:${by}`, billed, this.userDailyAudioSeconds, 86_400_000);
69
+ await this.reserve("scribe:server", billed, this.dailyAudioSeconds, 86_400_000);
70
+ signal?.throwIfAborted();
71
+ const form = new FormData();
72
+ // Canonical PCM prevents a crafted container from billing more audio
73
+ // than the duration we validated, and removes uploaded metadata.
74
+ form.set("file", new Blob([new Uint8Array(encodeWav(wav.samples))], { type: "audio/wav" }), "listening.wav");
75
+ form.set("model_id", "scribe_v2");
76
+ form.set("diarize", "true");
77
+ form.set("tag_audio_events", "false");
78
+ form.set("timestamps_granularity", "word");
79
+ // No language_code: preserve the source language, including Spanish.
80
+ const answer = await this.fetcher("https://api.elevenlabs.io/v1/speech-to-text", {
81
+ method: "POST", headers: { "xi-api-key": this.key }, body: form,
82
+ signal: signal ? AbortSignal.any([signal, AbortSignal.timeout(10_000)]) : AbortSignal.timeout(10_000),
83
+ });
84
+ if (!answer.ok)
85
+ throw new SpeechError("speaker transcription could not run; check provider quota and permissions", answer.status === 429 ? 429 : 502);
86
+ return speakerTurns(await answer.json(), wav);
87
+ }
88
+ finally {
89
+ this.hearing.delete(by);
90
+ }
91
+ }
92
+ async voices() {
93
+ if (!this.available())
94
+ throw new SpeechError("translated audio needs ELEVENLABS_API_KEY on the account server", 503);
95
+ if (!this.catalog || this.now() >= this.catalogUntil) {
96
+ this.catalogUntil = this.now() + 3600_000;
97
+ this.catalog = this.fetcher("https://api.elevenlabs.io/v2/voices?page_size=100&voice_type=default", {
98
+ headers: { "xi-api-key": this.key }, signal: AbortSignal.timeout(8000),
99
+ }).then(async (response) => {
100
+ if (!response.ok)
101
+ throw new SpeechError("ElevenLabs could not list voices; check the server key and its voice permissions", 503);
102
+ const body = await response.json();
103
+ return body.voices.filter(voice => /^[a-zA-Z0-9_-]+$/.test(voice.voice_id)).map(voice => ({
104
+ id: voice.voice_id, name: voice.name, gender: voice.labels?.["gender"] ?? "neutral", language: voice.labels?.["language"] ?? "",
105
+ }));
106
+ }).catch(error => { this.catalog = null; throw error; });
107
+ }
108
+ return this.catalog;
109
+ }
110
+ /** A 90-second capability for one channel, never the viewer's account credential. */
111
+ async grant(by, channel) {
112
+ if (!/^[\w-]{1,80}$/.test(channel))
113
+ throw new SpeechError("choose a playback session or live channel", 400);
114
+ if (!this.requests.check(`grant:${by}`, { allowed: 10, windowMs: 60_000 }).ok)
115
+ throw new SpeechError("too many audio authorization requests", 429);
116
+ for (const [token, grant] of this.grants)
117
+ if (grant.expires <= this.now())
118
+ this.grants.delete(token);
119
+ if (this.grants.size >= 5000)
120
+ throw new SpeechError("audio authorization is busy", 503);
121
+ const token = `nxd_${randomBytes(32).toString("base64url")}`;
122
+ const expires = this.now() + 90_000;
123
+ if (this.db) {
124
+ await this.ensure();
125
+ await this.db.query("INSERT INTO live_voice_grants (token_hash, by_account, channel, expires_at, remaining) VALUES ($1, $2, $3, $4, 2000)", [createHash("sha256").update(token).digest("hex"), by, channel, new Date(expires)]);
126
+ }
127
+ else
128
+ this.grants.set(token, { by, channel, expires, remaining: 2000 });
129
+ return { token, expires };
130
+ }
131
+ async authorize(token, channel, chars) {
132
+ if (!/^nxd_[A-Za-z0-9_-]{43}$/.test(token))
133
+ throw new SpeechError("sign in to enable translated audio", 401);
134
+ if (!Number.isFinite(chars) || chars < 1 || chars > 600)
135
+ throw new SpeechError("invalid caption length", 400);
136
+ if (this.db) {
137
+ await this.ensure();
138
+ const result = await this.db.query(`UPDATE live_voice_grants SET remaining = remaining - $3
139
+ WHERE token_hash = $1 AND channel = $2 AND expires_at > $4 AND remaining >= $3 RETURNING by_account`, [createHash("sha256").update(token).digest("hex"), channel, chars, new Date(this.now())]);
140
+ if (!result.rows.length)
141
+ throw new SpeechError("audio authorization expired or reached its limit; enable translated audio again", 401);
142
+ return String(result.rows[0]?.["by_account"]);
143
+ }
144
+ const grant = this.grants.get(token);
145
+ if (!grant || grant.expires <= this.now() || grant.channel !== channel)
146
+ throw new SpeechError("sign in to renew translated audio", 401);
147
+ if (grant.remaining < chars)
148
+ throw new SpeechError("this audio authorization reached its character limit", 429);
149
+ grant.remaining -= chars;
150
+ return grant.by;
151
+ }
152
+ async ensure() {
153
+ if (!this.db)
154
+ return;
155
+ this.schema ??= (async () => {
156
+ await this.db.query("CREATE TABLE IF NOT EXISTS live_voice_usage (bucket TEXT PRIMARY KEY, chars BIGINT NOT NULL, expires_at TIMESTAMPTZ NOT NULL)");
157
+ await this.db.query("CREATE TABLE IF NOT EXISTS live_voice_grants (token_hash TEXT PRIMARY KEY, by_account TEXT NOT NULL, channel TEXT NOT NULL, expires_at TIMESTAMPTZ NOT NULL, remaining INTEGER NOT NULL)");
158
+ })().catch(error => { this.schema = null; throw error; });
159
+ await this.schema;
160
+ if (this.now() >= this.cleanupAt) {
161
+ this.cleanupAt = this.now() + 60_000;
162
+ // Keep only live windows; accounting survives restarts and is shared by replicas.
163
+ await this.db.query("DELETE FROM live_voice_usage WHERE expires_at <= $1", [new Date(this.now())]);
164
+ await this.db.query("DELETE FROM live_voice_grants WHERE expires_at <= $1", [new Date(this.now())]);
165
+ }
166
+ }
167
+ async reserve(bucket, chars, limit, windowMs) {
168
+ if (chars > limit)
169
+ throw new SpeechError("translated audio reached its character budget; try again later", 429);
170
+ const window = Math.floor(this.now() / windowMs);
171
+ const key = `${bucket}:${windowMs}:${window}`;
172
+ if (this.db) {
173
+ await this.ensure();
174
+ const result = await this.db.query(`INSERT INTO live_voice_usage (bucket, chars, expires_at) VALUES ($1, $2, $4)
175
+ ON CONFLICT (bucket) DO UPDATE SET chars = live_voice_usage.chars + EXCLUDED.chars
176
+ WHERE live_voice_usage.chars + EXCLUDED.chars <= $3 RETURNING chars`, [key, chars, limit, new Date((window + 1) * windowMs)]);
177
+ if (!result.rows.length)
178
+ throw new SpeechError("translated audio reached its character budget; try again later", 429);
179
+ return;
180
+ }
181
+ // Local/single-process installations without a database. Account servers pass their pool.
182
+ const record = this.usage.get(key) ?? { minute: (window + 1) * windowMs, chars: 0 };
183
+ if (record.chars + chars > limit)
184
+ throw new SpeechError("translated audio reached its character budget; try again later", 429);
185
+ record.chars += chars;
186
+ this.usage.set(key, record);
187
+ }
188
+ async charge(by, channel, chars) {
189
+ for (const [key, record] of this.usage)
190
+ if (record.minute <= this.now())
191
+ this.usage.delete(key);
192
+ await this.reserve(`user:${by}`, chars, this.charsPerMinute, 60_000);
193
+ await this.reserve(`channel:${channel}`, chars, 6000, 60_000);
194
+ await this.reserve("server", chars, 12_000, 60_000);
195
+ await this.reserve(`user:${by}`, chars, this.userDailyChars, 86400_000);
196
+ await this.reserve("server", chars, this.dailyChars, 86400_000);
197
+ }
198
+ checkRequest(by) {
199
+ if (!this.requests.check(`audio:${by}`, { allowed: 60, windowMs: 60_000 }).ok)
200
+ throw new SpeechError("too many translated audio requests", 429);
201
+ }
202
+ /** Stream the first request immediately; concurrent listeners share its cached result. */
203
+ async stream(ask, by, signal) {
204
+ this.checkRequest(by);
205
+ const text = typeof ask.text === "string" ? ask.text.trim() : "";
206
+ if (!text || text.length > 600)
207
+ throw new SpeechError("translated audio needs a caption of 1–600 characters", 400);
208
+ if (!LIVE_VOICE_LANGUAGES.has(ask.language))
209
+ throw new SpeechError("this language is not supported by Flash voices", 400);
210
+ const voices = await this.voices();
211
+ signal?.throwIfAborted();
212
+ const gender = ask.profile === "lower" ? "male" : ask.profile === "higher" ? "female" : "neutral";
213
+ const voice = ask.voice && ask.voice !== "auto"
214
+ ? voices.find(voice => voice.id === ask.voice)
215
+ : voices.find(voice => voice.gender === gender) ?? voices[0];
216
+ if (!voice)
217
+ throw new SpeechError("choose an available voice", 400);
218
+ const id = createHash("sha256").update(JSON.stringify([LIVE_VOICE_MODEL, voice.id, ask.language, text])).digest("hex");
219
+ for (const [key, item] of this.cache)
220
+ if (item.until < this.now())
221
+ this.cache.delete(key);
222
+ const cached = this.cache.get(id);
223
+ if (cached)
224
+ return new Response(new Uint8Array(cached.bytes), { headers: HEADERS });
225
+ const pending = this.pending.get(id);
226
+ if (pending)
227
+ return new Response(new Uint8Array(await pending), { headers: HEADERS });
228
+ if (this.pending.size >= 4)
229
+ throw new SpeechError("translated audio is busy; waiting for the next caption", 429);
230
+ if ((this.activeBy.get(by) ?? 0) >= 2)
231
+ throw new SpeechError("two voice requests are already active for this account", 429);
232
+ // Register before the first provider await, so identical requests cannot both bill.
233
+ let done;
234
+ let fail;
235
+ const finished = new Promise((resolve, reject) => { done = resolve; fail = reject; });
236
+ this.pending.set(id, finished);
237
+ this.activeBy.set(by, (this.activeBy.get(by) ?? 0) + 1);
238
+ const release = () => { this.pending.delete(id); this.activeBy.set(by, Math.max(0, (this.activeBy.get(by) ?? 1) - 1)); if (!this.activeBy.get(by))
239
+ this.activeBy.delete(by); };
240
+ void finished.catch(() => undefined);
241
+ try {
242
+ await this.charge(by, ask.channel ?? "direct", text.length);
243
+ signal?.throwIfAborted();
244
+ const response = await this.fetcher(`https://api.elevenlabs.io/v1/text-to-speech/${voice.id}/stream?output_format=pcm_16000`, {
245
+ method: "POST",
246
+ headers: { "xi-api-key": this.key, "content-type": "application/json" },
247
+ body: JSON.stringify({ text, model_id: LIVE_VOICE_MODEL, language_code: ask.language }),
248
+ signal: signal ? AbortSignal.any([signal, AbortSignal.timeout(10_000)]) : AbortSignal.timeout(10_000),
249
+ });
250
+ if (!response.ok || !response.body)
251
+ throw new SpeechError(response.status === 429 ? "ElevenLabs audio quota is temporarily exhausted" : "ElevenLabs could not generate audio; check the server key and quota", response.status === 429 ? 429 : 502);
252
+ const [play, keep] = response.body.tee();
253
+ void (async () => {
254
+ const reader = keep.getReader();
255
+ const chunks = [];
256
+ let size = 0;
257
+ try {
258
+ while (true) {
259
+ const { done: ended, value } = await reader.read();
260
+ if (ended)
261
+ break;
262
+ size += value.length;
263
+ if (size > 2 * 1024 * 1024) {
264
+ await reader.cancel();
265
+ throw new SpeechError("voice response was too long", 502);
266
+ }
267
+ chunks.push(value);
268
+ }
269
+ if (!size || size % 2)
270
+ throw new SpeechError("voice response contained incomplete audio", 502);
271
+ const bytes = new Uint8Array(size);
272
+ let at = 0;
273
+ for (const chunk of chunks) {
274
+ bytes.set(chunk, at);
275
+ at += chunk.length;
276
+ }
277
+ this.cache.set(id, { until: this.now() + 60_000, bytes });
278
+ // At most a few MB for recent lines, never a recording archive.
279
+ while (this.cache.size > 32)
280
+ this.cache.delete(this.cache.keys().next().value);
281
+ done(bytes);
282
+ }
283
+ catch (error) {
284
+ fail(error);
285
+ }
286
+ finally {
287
+ reader.releaseLock();
288
+ release();
289
+ }
290
+ })();
291
+ return new Response(play, { headers: HEADERS });
292
+ }
293
+ catch (error) {
294
+ fail(error);
295
+ release();
296
+ throw error;
297
+ }
298
+ }
299
+ }
package/dist/server.d.ts CHANGED
@@ -1,4 +1,5 @@
1
1
  import { type IncomingMessage, type Server, type ServerResponse } from "node:http";
2
+ import { LiveVoice } from "./live-voice.ts";
2
3
  import { Connections } from "./connections.ts";
3
4
  import { Broadcaster, type Destination, type EncoderSettings } from "./broadcast.ts";
4
5
  import { Ingest } from "./ingest.ts";
@@ -703,6 +704,7 @@ export interface HandlerOptions {
703
704
  trollbox?: Trollbox;
704
705
  /** Speech to text: a line said out loud, heard here. Needs the optional model. */
705
706
  speech?: Speech;
707
+ liveVoice?: LiveVoice;
706
708
  /** Translation: texts in another language, by a model here. The same optional library. */
707
709
  translator?: Translator;
708
710
  /** The transcript store: what was heard, kept under the media's identity. Where the accounts are. */
package/dist/server.js CHANGED
@@ -16,6 +16,9 @@ import { randomBytes } from "node:crypto";
16
16
  import { hostname } from "node:os";
17
17
  import { spawn, spawnSync } from "node:child_process";
18
18
  import { readFileSync } from "node:fs";
19
+ import { Readable } from "node:stream";
20
+ import { pipeline as pipeStream } from "node:stream/promises";
21
+ import { LiveVoice, LIVE_VOICE_LANGUAGES, LIVE_VOICE_MODEL } from "./live-voice.js";
19
22
  import { Connections } from "./connections.js";
20
23
  import { Broadcaster, DEFAULT_ENCODER, PRESETS, redact, } from "./broadcast.js";
21
24
  import { Ingest, normaliseFormat } from "./ingest.js";
@@ -65,7 +68,7 @@ import { Tickets, needsTicket, ticketFrom, ticketsFromEnv } from "./tickets.js";
65
68
  import { Layouts } from "./layouts.js";
66
69
  import { Rooms } from "./rooms.js";
67
70
  import { Trollbox, TrollboxError, fallbackHandle, roomFor } from "./trollbox.js";
68
- import { MAX_BYTES as SPEECH_BYTES, Speech, SpeechError, isWav, languageOf } from "./speech.js";
71
+ import { MAX_BYTES as SPEECH_BYTES, NATIVE_REVISION, Speech, SpeechError, isWav, languageOf } from "./speech.js";
69
72
  import { Captions } from "./captions.js";
70
73
  import { LANGUAGES, Translator } from "./translate.js";
71
74
  import { StoredTranslations } from "./translate-jobs.js";
@@ -1106,7 +1109,7 @@ const CORS = {
1106
1109
  // paths and takes six commands; binding to 127.0.0.1 is what keeps it shut.
1107
1110
  "access-control-allow-origin": "*",
1108
1111
  "access-control-allow-methods": "GET, POST, PUT, DELETE, OPTIONS",
1109
- "access-control-allow-headers": "content-type",
1112
+ "access-control-allow-headers": "content-type, authorization",
1110
1113
  "access-control-max-age": "86400",
1111
1114
  };
1112
1115
  /**
@@ -1278,6 +1281,8 @@ export function joinDocument(shell, subject, site) {
1278
1281
  .replace("</head>", `${card}<meta name="nixamp-shell-title" content="${htmlText(shellTitle)}" />\n </head>`);
1279
1282
  }
1280
1283
  function json(response, code, body) {
1284
+ if (code === 429)
1285
+ response.setHeader("retry-after", "60");
1281
1286
  const text = JSON.stringify(body);
1282
1287
  response.writeHead(code, {
1283
1288
  ...CORS,
@@ -1287,6 +1292,18 @@ function json(response, code, body) {
1287
1292
  });
1288
1293
  response.end(text);
1289
1294
  }
1295
+ /** Stream PCM without waiting for a complete utterance, respecting browser backpressure. */
1296
+ async function voiceResponse(response, upstream) {
1297
+ response.writeHead(upstream.status, {
1298
+ ...CORS, "content-type": upstream.headers.get("content-type") ?? "application/json",
1299
+ "cache-control": "no-store", "x-accel-buffering": "no",
1300
+ });
1301
+ if (!upstream.body) {
1302
+ response.end();
1303
+ return;
1304
+ }
1305
+ await pipeStream(Readable.fromWeb(upstream.body), response);
1306
+ }
1290
1307
  async function readBody(request, limit = 64 * 1024) {
1291
1308
  const chunks = [];
1292
1309
  let size = 0;
@@ -2314,6 +2331,80 @@ export function createHandler(engine, options) {
2314
2331
  * room, the words go straight into that room's trollbox as a line by
2315
2332
  * whoever spoke them. The page, the CLI and the MCP tools all come here.
2316
2333
  */
2334
+ if (["/api/v1/speech/voices", "/api/v1/speech/grant", "/api/v1/speech/synthesize", "/api/v1/speech/speakers"].includes(path) && options.accounts) {
2335
+ const controller = new AbortController();
2336
+ response.once("close", () => controller.abort());
2337
+ try {
2338
+ const caller = callerOf(request.headers, request.socket.remoteAddress, options.behindProxy ?? false);
2339
+ if (!guard.check(`voice-ip:${caller}`, { allowed: 600, windowMs: 60_000 }).ok) {
2340
+ json(response, 429, { error: "too many voice requests" });
2341
+ return;
2342
+ }
2343
+ const service = options.liveVoice;
2344
+ if (!service?.available()) {
2345
+ json(response, 503, { error: "translated audio needs ELEVENLABS_API_KEY on the account server" });
2346
+ return;
2347
+ }
2348
+ if (path.endsWith("/synthesize") && request.method === "POST") {
2349
+ let body;
2350
+ try {
2351
+ body = JSON.parse(await readBody(request, 4096));
2352
+ }
2353
+ catch {
2354
+ json(response, 400, { error: "send a caption as JSON" });
2355
+ return;
2356
+ }
2357
+ if (!body || typeof body !== "object") {
2358
+ json(response, 400, { error: "send a caption as JSON" });
2359
+ return;
2360
+ }
2361
+ const by = await service.authorize(tokenFrom(request.headers), body.channel ?? "", typeof body.text === "string" ? body.text.length : 0);
2362
+ await voiceResponse(response, await service.stream(body, by, controller.signal));
2363
+ }
2364
+ else {
2365
+ const who = await options.accounts.whoIs(tokenFrom(request.headers));
2366
+ if (!who) {
2367
+ json(response, 401, { error: "sign in to use translated audio" });
2368
+ return;
2369
+ }
2370
+ if (path.endsWith("/voices") && request.method === "GET") {
2371
+ json(response, 200, { voices: await service.voices(), model: LIVE_VOICE_MODEL, languages: [...LIVE_VOICE_LANGUAGES] });
2372
+ }
2373
+ else if (path.endsWith("/speakers") && request.method === "POST") {
2374
+ if (!request.headers["content-type"]?.startsWith("audio/wav")) {
2375
+ json(response, 415, { error: "send a mono 16 kHz WAV" });
2376
+ return;
2377
+ }
2378
+ const bytes = await readBytes(request, 484_000);
2379
+ json(response, 200, await service.hear(bytes, who.id, controller.signal));
2380
+ }
2381
+ else if (path.endsWith("/grant") && request.method === "POST") {
2382
+ if (!request.headers["content-type"]?.startsWith("application/json")) {
2383
+ json(response, 415, { error: "send JSON" });
2384
+ return;
2385
+ }
2386
+ let body;
2387
+ try {
2388
+ body = JSON.parse(await readBody(request, 1024));
2389
+ }
2390
+ catch {
2391
+ json(response, 400, { error: "send a channel as JSON" });
2392
+ return;
2393
+ }
2394
+ json(response, 200, await service.grant(who.id, typeof body?.channel === "string" ? body.channel : ""));
2395
+ }
2396
+ else
2397
+ json(response, 405, { error: "GET voices or POST an audio request" });
2398
+ }
2399
+ }
2400
+ catch (error) {
2401
+ if (response.headersSent)
2402
+ response.destroy();
2403
+ else
2404
+ json(response, error instanceof SpeechError ? error.status : 502, { error: error instanceof SpeechError ? error.message : "translated audio is unavailable" });
2405
+ }
2406
+ return;
2407
+ }
2317
2408
  if (path === "/api/v1/speech/transcribe" && options.speech && options.accounts) {
2318
2409
  const speech = options.speech;
2319
2410
  try {
@@ -2343,6 +2434,7 @@ export function createHandler(engine, options) {
2343
2434
  // Pieces with their timing, for a whole file being written down.
2344
2435
  timestamps: url.searchParams.get("timestamps") === "1",
2345
2436
  by: who.id,
2437
+ ...(url.searchParams.get("live") === "1" ? { deadline: Date.now() + 10_000 } : {}),
2346
2438
  });
2347
2439
  // The language travels back: as told, or as the ear guessed it, so
2348
2440
  // a captioner can say it next time and a transcript can be kept as it.
@@ -2563,7 +2655,7 @@ export function createHandler(engine, options) {
2563
2655
  return;
2564
2656
  }
2565
2657
  try {
2566
- json(response, 200, await translator.translate(texts, from, to, { by: who.id }));
2658
+ json(response, 200, await translator.translate(texts, from, to, { by: who.id, ...(url.searchParams.get("live") === "1" ? { deadline: Date.now() + 10_000 } : {}) }));
2567
2659
  }
2568
2660
  catch (error) {
2569
2661
  if (error instanceof SpeechError)
@@ -2620,6 +2712,12 @@ export function createHandler(engine, options) {
2620
2712
  json(response, 400, { error: "format is json, srt, vtt or txt" });
2621
2713
  return;
2622
2714
  }
2715
+ // Joining a live must not start translating an entire old transcript.
2716
+ if (url.searchParams.get("cached") === "1") {
2717
+ const saved = await store.get(id, language);
2718
+ json(response, saved ? 200 : 404, saved ? wire(saved) : { error: "no cached transcript" });
2719
+ return;
2720
+ }
2623
2721
  const answer = await options.translations.get(id, language, who.id);
2624
2722
  if (answer.status !== 200 && answer.status !== 202) {
2625
2723
  json(response, answer.status, { error: answer.error });
@@ -3930,6 +4028,44 @@ export function createHandler(engine, options) {
3930
4028
  * started by the first person asking. `captions` is the live stream
3931
4029
  * of lines; `transcript` is the recent ones as JSON, for a poll.
3932
4030
  */
4031
+ if ((action === "voice" || action === "voice-options") && request.method === "GET") {
4032
+ const controller = new AbortController();
4033
+ response.once("close", () => controller.abort());
4034
+ const timeout = setTimeout(() => controller.abort(), 12_000);
4035
+ try {
4036
+ const caller = callerOf(request.headers, request.socket.remoteAddress, options.behindProxy ?? false);
4037
+ if (!guard.check(`channel-voice-ip:${caller}`, { allowed: 60, windowMs: 60_000 }).ok || !guard.check(`channel-voice:${id}`, { allowed: 240, windowMs: 60_000 }).ok) {
4038
+ json(response, 429, { error: "too many translated audio requests" });
4039
+ return;
4040
+ }
4041
+ if (!channels.has(id)) {
4042
+ json(response, 404, { error: "nothing is playing on that channel" });
4043
+ return;
4044
+ }
4045
+ if (!options.captions) {
4046
+ json(response, 503, { error: "this server cannot caption" });
4047
+ return;
4048
+ }
4049
+ const at = action === "voice-options" ? null : Number(url.searchParams.get("at"));
4050
+ const language = languageCode(url.searchParams.get("language"));
4051
+ if ((at !== null && (!url.searchParams.has("at") || !Number.isFinite(at))) || language === null) {
4052
+ json(response, 400, { error: "audio needs a caption time and language" });
4053
+ return;
4054
+ }
4055
+ const upstream = await options.captions.voiceRequest(id, at, language, url.searchParams.get("voice") ?? "auto", controller.signal, tokenFrom(request.headers));
4056
+ await voiceResponse(response, upstream);
4057
+ }
4058
+ catch (error) {
4059
+ if (response.headersSent)
4060
+ response.destroy();
4061
+ else
4062
+ json(response, error instanceof SpeechError ? error.status : 502, { error: error instanceof SpeechError ? error.message : "translated audio is unavailable" });
4063
+ }
4064
+ finally {
4065
+ clearTimeout(timeout);
4066
+ }
4067
+ return;
4068
+ }
3933
4069
  if ((action === "captions" || action === "transcript") && request.method === "GET") {
3934
4070
  const captions = options.captions;
3935
4071
  if (!captions) {
@@ -3940,6 +4076,11 @@ export function createHandler(engine, options) {
3940
4076
  json(response, 503, { error: "this server is not signed in to nixamp.com, so it cannot caption; run `nixamp login` on it" });
3941
4077
  return;
3942
4078
  }
4079
+ const caller = callerOf(request.headers, request.socket.remoteAddress, options.behindProxy ?? false);
4080
+ if (!guard.check(`captions:${caller}`, { allowed: 120, windowMs: 60_000 }).ok || !captions.capacity(id)) {
4081
+ json(response, 429, { error: "captioning is at capacity; try again in a moment" });
4082
+ return;
4083
+ }
3943
4084
  if (!channels.has(id)) {
3944
4085
  json(response, 404, { error: "nothing is playing on that channel" });
3945
4086
  return;
@@ -3958,7 +4099,7 @@ export function createHandler(engine, options) {
3958
4099
  // Asking keeps the captioner up: it stops a minute after the last ask.
3959
4100
  captions.subscribe(id, () => undefined, wanted)?.();
3960
4101
  json(response, 200, {
3961
- channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted),
4102
+ channel: id, recognitionRevision: NATIVE_REVISION, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted),
3962
4103
  });
3963
4104
  return;
3964
4105
  }
@@ -3979,7 +4120,7 @@ export function createHandler(engine, options) {
3979
4120
  }
3980
4121
  // How far behind the live edge a newcomer's playback starts, so the
3981
4122
  // page can hold each line until its own sound gets there.
3982
- write("hello", { channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted) });
4123
+ write("hello", { channel: id, recognitionRevision: NATIVE_REVISION, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted) });
3983
4124
  const beat = setInterval(() => response.write(": beat\n\n"), 20_000);
3984
4125
  beat.unref?.();
3985
4126
  const done = () => {
@@ -5276,6 +5417,13 @@ export async function serve(argv, version = "0.1.0") {
5276
5417
  // the first person to speak does not wait for the model to arrive; a box
5277
5418
  // without the optional model never says it is ready, and answers 503.
5278
5419
  const speech = accounts && process.env["NIXAMP_STT"] !== "off" ? new Speech() : undefined;
5420
+ const liveVoice = accounts && process.env["NIXAMP_DUBBING"] !== "off" ? new LiveVoice({
5421
+ ...(pool ? { db: pool } : {}),
5422
+ dailyChars: Number(process.env["NIXAMP_DUB_DAILY_CHARS"] ?? 200_000),
5423
+ dailyAudioSeconds: Number(process.env["NIXAMP_DUB_DAILY_AUDIO_SECONDS"] ?? 86_400),
5424
+ userDailyChars: Number(process.env["NIXAMP_DUB_USER_DAILY_CHARS"] ?? 120_000),
5425
+ userDailyAudioSeconds: Number(process.env["NIXAMP_DUB_USER_DAILY_AUDIO_SECONDS"] ?? 43_200),
5426
+ }) : undefined;
5279
5427
  if (speech) {
5280
5428
  void speech.warm().then((ready) => {
5281
5429
  console.error(ready ? `nixamp: hearing with ${speech.model}` : `nixamp: not hearing: ${speech.lastFailure}`);
@@ -5759,6 +5907,7 @@ export async function serve(argv, version = "0.1.0") {
5759
5907
  ...(rooms ? { rooms } : {}),
5760
5908
  ...(trollbox ? { trollbox } : {}),
5761
5909
  ...(speech ? { speech } : {}),
5910
+ ...(liveVoice ? { liveVoice } : {}),
5762
5911
  ...(translator ? { translator } : {}),
5763
5912
  ...(transcripts ? { transcripts } : {}),
5764
5913
  ...(translations ? { translations } : {}),
@@ -0,0 +1,32 @@
1
+ /** Provider speaker IDs belong to a clip. Clients reconcile the overlapping
2
+ * word timestamps before assigning a voice; these are not people's identities. */
3
+ import { type Wav } from "./speech.ts";
4
+ import { type VoiceProfile } from "./voice-profile.ts";
5
+ export interface SpeakerTurn {
6
+ start: number;
7
+ end: number;
8
+ text: string;
9
+ speaker: string;
10
+ profile: VoiceProfile;
11
+ words: {
12
+ text: string;
13
+ start: number;
14
+ end: number;
15
+ }[];
16
+ }
17
+ export interface SpeakerTranscript {
18
+ language: string;
19
+ seconds: number;
20
+ turns: SpeakerTurn[];
21
+ }
22
+ export interface ScribeResult {
23
+ language_code?: string;
24
+ words?: {
25
+ text: string;
26
+ start: number;
27
+ end: number;
28
+ type: string;
29
+ speaker_id?: string | null;
30
+ }[];
31
+ }
32
+ export declare function speakerTurns(body: ScribeResult, wav: Wav): SpeakerTranscript;
@@ -0,0 +1,30 @@
1
+ /** Provider speaker IDs belong to a clip. Clients reconcile the overlapping
2
+ * word timestamps before assigning a voice; these are not people's identities. */
3
+ import { reliableText } from "./speech.js";
4
+ import { voiceProfile } from "./voice-profile.js";
5
+ const ISO = Object.fromEntries("eng:en spa:es deu:de ger:de fra:fr fre:fr por:pt ita:it nld:nl dut:nl swe:sv dan:da fin:fi rus:ru ukr:uk ces:cs cze:cs hun:hu cmn:zh zho:zh ara:ar hin:hi vie:vi ind:id jpn:ja kor:ko pol:pl tur:tr ron:ro bul:bg ell:el nor:no nob:no nno:no".split(" ").map(pair => pair.split(":")));
6
+ export function speakerTurns(body, wav) {
7
+ const seconds = wav.samples.length / wav.rate;
8
+ const language = ISO[body.language_code ?? ""] ?? body.language_code ?? "";
9
+ const turns = [];
10
+ for (const word of (body.words ?? []).slice(0, 1500)) {
11
+ if (word.type !== "word" || typeof word.text !== "string" || !Number.isFinite(word.start) || !Number.isFinite(word.end) || word.start < 0 || word.end < word.start || word.end > seconds + 0.5)
12
+ continue;
13
+ const speaker = typeof word.speaker_id === "string" && /^[\w-]{1,80}$/.test(word.speaker_id) ? word.speaker_id : "unknown";
14
+ const last = turns.at(-1);
15
+ if (last && last.speaker === speaker && word.start - last.end < 0.8 && last.text.length + word.text.length < 400) {
16
+ last.text += ` ${word.text}`;
17
+ last.end = Math.max(last.end, word.end);
18
+ last.words.push({ text: word.text, start: word.start, end: word.end });
19
+ }
20
+ else
21
+ turns.push({ start: word.start, end: word.end, text: word.text, speaker, profile: "unknown", words: [{ text: word.text, start: word.start, end: word.end }] });
22
+ }
23
+ const pcm = Buffer.alloc(wav.samples.length * 2);
24
+ wav.samples.forEach((sample, i) => pcm.writeInt16LE(Math.round(Math.max(-1, Math.min(1, sample)) * 32767), i * 2));
25
+ for (const turn of turns) {
26
+ turn.text = reliableText(turn.text, Math.max(1, turn.end - turn.start));
27
+ turn.profile = voiceProfile(pcm.subarray(Math.floor(turn.start * 16000) * 2, Math.ceil(turn.end * 16000) * 2));
28
+ }
29
+ return { language, seconds, turns: turns.filter(turn => turn.text !== "") };
30
+ }