nixamp 0.25.0 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +108 -7
  2. package/dist/captions.d.ts +7 -0
  3. package/dist/captions.js +92 -28
  4. package/dist/live-voice.d.ts +70 -0
  5. package/dist/live-voice.js +299 -0
  6. package/dist/server.d.ts +2 -0
  7. package/dist/server.js +154 -5
  8. package/dist/speaker-turns.d.ts +32 -0
  9. package/dist/speaker-turns.js +30 -0
  10. package/dist/speech.d.ts +47 -0
  11. package/dist/speech.js +49 -4
  12. package/dist/transcript-client.d.ts +2 -2
  13. package/dist/transcript-client.js +5 -2
  14. package/dist/transcripts.d.ts +5 -0
  15. package/dist/transcripts.js +8 -3
  16. package/dist/translate-jobs.js +13 -7
  17. package/dist/translate.d.ts +2 -0
  18. package/dist/translate.js +9 -0
  19. package/dist/voice-profile.d.ts +7 -0
  20. package/dist/voice-profile.js +53 -0
  21. package/dist/warm.js +3 -3
  22. package/package.json +1 -1
  23. package/src/captions.ts +86 -25
  24. package/src/live-voice.ts +255 -0
  25. package/src/server.ts +104 -5
  26. package/src/speaker-turns.ts +34 -0
  27. package/src/speech.ts +46 -7
  28. package/src/transcript-client.ts +4 -0
  29. package/src/transcripts.ts +13 -3
  30. package/src/translate-jobs.ts +13 -7
  31. package/src/translate.ts +7 -1
  32. package/src/voice-profile.ts +46 -0
  33. package/src/warm.ts +3 -3
  34. package/web/dist/assets/{hls-3VKVEQE3-B4ltbKDh.js → hls-3VKVEQE3-vgax_tk1.js} +1 -1
  35. package/web/dist/assets/index-BzjrTOLf.js +1 -0
  36. package/web/dist/assets/index-D2Iy07pG.css +1 -0
  37. package/web/dist/assets/{mpegts-Byy3EkfT.js → mpegts-DmcUOiHq.js} +1 -1
  38. package/web/dist/assets/{mpegts-LO6RVLD6-C9qqolrW.js → mpegts-LO6RVLD6-CH3EQi6L.js} +1 -1
  39. package/web/dist/index.html +42 -22
  40. package/web/dist/sw.js +6 -6
  41. package/web/dist/assets/index-d7TvpeFZ.js +0 -1
  42. package/web/dist/assets/index-oyp61Kly.css +0 -1
package/src/captions.ts CHANGED
@@ -32,7 +32,8 @@
32
32
  */
33
33
  import { spawn } from "node:child_process";
34
34
  import type { Listener } from "./channels.ts";
35
- import { RATE } from "./speech.ts";
35
+ import { NATIVE_REVISION, RATE, SpeechError, reliableText } from "./speech.ts";
36
+ import { voiceProfile } from "./voice-profile.ts";
36
37
  import { fetchTranscript, keepLines, keepMedia, translateTexts, type MediaToKeep } from "./transcript-client.ts";
37
38
  import { covered, lineAt, transcriptIdOf, type TranscriptLine } from "./transcripts.ts";
38
39
 
@@ -47,6 +48,8 @@ export interface CaptionLine {
47
48
  language?: string;
48
49
  /** What was heard, when this line is a translation of it. */
49
50
  original?: string;
51
+ sourceLanguage?: string;
52
+ voiceProfile?: "lower" | "higher" | "unknown";
50
53
  }
51
54
 
52
55
  /** What turns a channel's bytes into 16 kHz mono 16-bit PCM. ffmpeg, or a test's stand-in. */
@@ -104,6 +107,8 @@ export const IDLE_MS = 60_000;
104
107
  export const QUIET = 0.004;
105
108
  /** Windows waiting on the ear at once. Past this the sound is dropped, not queued: late words are worse than none. */
106
109
  export const IN_FLIGHT = 2;
110
+ export const LIVE_DEADLINE_MS = 12_000;
111
+ export const MAX_CAPTIONERS = 4;
107
112
  /** Heard lines wait this long, at most, before they are kept. */
108
113
  export const FLUSH_MS = 20_000;
109
114
  /** Or this many. */
@@ -252,6 +257,9 @@ class Captioner {
252
257
  private stopped = false;
253
258
  private complainedAt = 0;
254
259
  private windows = 0;
260
+ private audioUntil = 0;
261
+ private lastEmittedAt = -Infinity;
262
+ private readonly requests = new Set<AbortController>();
255
263
  /** What the channel is playing, when the store is to be told. */
256
264
  private readonly media: ChannelMedia | null;
257
265
  private readonly transcriptId: string | null;
@@ -264,6 +272,8 @@ class Captioner {
264
272
  private flush: ReturnType<typeof setTimeout> | null = null;
265
273
  /** Translations in order, per language: a slow one must not overtake the next. */
266
274
  private readonly chains = new Map<string, Promise<void>>();
275
+ private readonly translating = new Set<string>();
276
+ private readonly nextTranslation = new Map<string, { line: CaptionLine; mediaStart: number | null }>();
267
277
 
268
278
  constructor(
269
279
  readonly id: string,
@@ -326,16 +336,18 @@ class Captioner {
326
336
  this.asked.add(language);
327
337
  const session = this.options.session();
328
338
  if (session === null) return;
329
- const got = await fetchTranscript(session, this.transcriptId, language, this.options.fetcher ?? fetch);
339
+ const got = await fetchTranscript(session, this.transcriptId, language, this.options.fetcher ?? fetch, true);
330
340
  if (this.stopped) return;
331
341
  if (!got.ok) {
332
342
  if (got.status !== 404) this.complain(`the store did not answer: ${got.error}`);
333
343
  return;
334
344
  }
335
- this.known.set(language, got.body.lines);
336
- if (language === "" && this.language === "" && got.body.language) this.language = got.body.language;
337
- if (got.body.lines.length > 0) {
338
- this.options.onEvent?.(`captions for "${this.id}": the store knows ${got.body.lines.length} lines of this${language ? ` in ${language}` : ""}`);
345
+ // A row's model/language could have been updated while its old bad lines remained.
346
+ // Trust individual lines made with the corrected native pipeline only.
347
+ const usable = got.body.lines.filter((line) => line.revision === NATIVE_REVISION && reliableText(line.text, line.end - line.start));
348
+ this.known.set(language, usable);
349
+ if (usable.length > 0) {
350
+ this.options.onEvent?.(`captions for "${this.id}": the store knows ${usable.length} lines of this${language ? ` in ${language}` : ""}`);
339
351
  }
340
352
  }
341
353
 
@@ -350,7 +362,9 @@ class Captioner {
350
362
  const rest = all.subarray(size);
351
363
  this.pending = rest.length > 0 ? [Buffer.from(rest)] : [];
352
364
  this.pendingBytes = rest.length;
353
- const until = this.now();
365
+ const duration = this.options.windowMs ?? WINDOW_MS;
366
+ const until = Math.max(this.audioUntil + duration, this.now() - rest.length / (RATE * 2) * 1000);
367
+ this.audioUntil = until;
354
368
  const index = this.windows;
355
369
  this.windows += 1;
356
370
  void this.hear(Buffer.from(window), until - (this.options.windowMs ?? WINDOW_MS), until, index);
@@ -368,7 +382,8 @@ class Captioner {
368
382
  if (this.readOut.has(line.start)) continue;
369
383
  this.readOut.add(line.start);
370
384
  const lineAt = at + (line.start - span.start) * 1000;
371
- this.emit({ channel: this.id, at: lineAt, until: lineAt + (line.end - line.start) * 1000, text: line.text, ...(this.language ? { language: this.language } : {}) }, line.start);
385
+ this.emit({ channel: this.id, at: lineAt, until: lineAt + (line.end - line.start) * 1000, text: line.text,
386
+ ...(line.language ? { language: line.language } : {}), ...(line.voiceProfile ? { voiceProfile: line.voiceProfile } : {}) }, line.start);
372
387
  }
373
388
  return;
374
389
  }
@@ -380,37 +395,47 @@ class Captioner {
380
395
  return;
381
396
  }
382
397
  this.inFlight += 1;
398
+ const controller = new AbortController();
399
+ this.requests.add(controller);
400
+ const timeout = setTimeout(() => controller.abort(), LIVE_DEADLINE_MS);
383
401
  try {
384
402
  const wav = wavAround(pcm);
385
403
  const url = new URL(`${session.site.replace(/\/+$/, "")}/api/v1/speech/transcribe`);
386
- if (this.language) url.searchParams.set("language", this.language);
404
+ // Let each audio window detect its own language. Cached text and a viewer's
405
+ // translation selection must never constrain the recognizer.
406
+ url.searchParams.set("live", "1");
387
407
  const response = await (this.options.fetcher ?? fetch)(url.toString(), {
388
408
  method: "POST",
389
409
  headers: { authorization: `Bearer ${session.token}`, "content-type": "audio/wav" },
390
410
  body: new Blob([wav.buffer.slice(wav.byteOffset, wav.byteOffset + wav.byteLength) as ArrayBuffer]),
411
+ signal: controller.signal,
391
412
  });
392
413
  const body = (await response.json().catch(() => ({}))) as { text?: string; language?: string; model?: string; error?: string };
393
414
  if (!response.ok) {
394
415
  this.complain(body.error ?? `nixamp.com answered ${response.status}`);
395
416
  return;
396
417
  }
397
- const text = (body.text ?? "").trim();
398
- if (text === "" || this.stopped) return;
418
+ const text = reliableText(body.text ?? "", (until - at) / 1000);
419
+ if (text === "" || this.stopped || controller.signal.aborted || this.now() - until > LIVE_DEADLINE_MS) return;
399
420
  this.error = "";
400
- if (body.language && this.language === "") this.language = body.language;
401
421
  if (body.model) this.model = body.model;
402
- const line: CaptionLine = { channel: this.id, at, until, text, ...(this.language ? { language: this.language } : {}) };
422
+ const line: CaptionLine = { channel: this.id, at, until, text, ...(body.language ? { language: body.language } : {}), voiceProfile: voiceProfile(pcm) };
403
423
  this.emit(line, span?.start ?? null);
404
- if (span) this.keep("", { start: span.start, end: span.end, text });
424
+ if (span) this.keep("", { start: span.start, end: span.end, text, language: line.language, voiceProfile: line.voiceProfile, revision: NATIVE_REVISION });
405
425
  } catch (error) {
406
426
  this.complain(`could not reach the ear: ${(error as Error).message}`);
407
427
  } finally {
428
+ clearTimeout(timeout);
429
+ this.requests.delete(controller);
408
430
  this.inFlight -= 1;
409
431
  }
410
432
  }
411
433
 
412
434
  /** A line as heard, to whoever wants the original, and translated to whoever wants another language. */
413
435
  private emit(line: CaptionLine, mediaStart: number | null): void {
436
+ if (line.at <= this.lastEmittedAt) return;
437
+ this.lastEmittedAt = line.at;
438
+ this.language = line.language ?? "";
414
439
  this.lines.push(line);
415
440
  while (this.lines.length > KEEP) this.lines.shift();
416
441
  for (const [subscriber, language] of this.subscribers) {
@@ -436,33 +461,46 @@ class Captioner {
436
461
 
437
462
  /** The line in another language: from the store when it has been through this moment, from nixamp.com otherwise. */
438
463
  private translated(language: string, line: CaptionLine, mediaStart: number | null): void {
439
- const chain = (this.chains.get(language) ?? Promise.resolve()).then(async () => {
440
- if (this.stopped) return;
464
+ // One active request and the latest pending line per target, never a promise
465
+ // chain containing minutes of stale commentary.
466
+ if (this.translating.has(language)) {
467
+ this.nextTranslation.set(language, { line, mediaStart });
468
+ return;
469
+ }
470
+ this.translating.add(language);
471
+ const chain = Promise.resolve().then(async () => {
472
+ if (this.stopped || !this.wanted().has(language) || this.now() - line.until > LIVE_DEADLINE_MS) return;
441
473
  let text = "";
442
474
  const stored = mediaStart === null ? null : lineAt(this.known.get(language) ?? [], mediaStart);
443
- if (stored) {
475
+ if (stored && stored.original === line.text) {
444
476
  text = stored.text;
445
477
  } else {
446
478
  const session = this.options.session();
447
479
  if (session === null) return;
448
- const got = await translateTexts(session, [line.text], this.language, language, this.options.fetcher ?? fetch);
480
+ if (!line.language) return;
481
+ const got = await translateTexts(session, [line.text], line.language, language, this.options.fetcher ?? fetch, AbortSignal.timeout(LIVE_DEADLINE_MS));
449
482
  if (!got.ok) {
450
483
  this.complain(`could not translate to ${language}: ${got.error}`);
451
484
  return;
452
485
  }
453
- text = (got.body.texts[0] ?? "").trim();
486
+ text = reliableText(got.body.texts[0] ?? "", (line.until - line.at) / 1000);
454
487
  if (text === "") return;
455
- if (mediaStart !== null) this.keep(language, { start: mediaStart, end: round(mediaStart + (line.until - line.at) / 1000), text });
488
+ if (mediaStart !== null) this.keep(language, { start: mediaStart, end: round(mediaStart + (line.until - line.at) / 1000), text, original: line.text, revision: NATIVE_REVISION });
456
489
  }
457
- if (this.stopped) return;
458
- const said: CaptionLine = { ...line, text, language, original: line.text };
490
+ if (this.stopped || !this.wanted().has(language) || this.now() - line.until > LIVE_DEADLINE_MS) return;
491
+ const said: CaptionLine = { ...line, text, language, sourceLanguage: line.language, original: line.text };
459
492
  const lines = this.linesBy.get(language) ?? [];
460
493
  lines.push(said);
461
494
  while (lines.length > KEEP) lines.shift();
462
495
  this.linesBy.set(language, lines);
463
496
  for (const [subscriber, wanted] of this.subscribers) if (wanted === language) this.tell(subscriber, said);
464
497
  });
465
- this.chains.set(language, chain.catch(() => undefined));
498
+ this.chains.set(language, chain.catch(() => undefined).finally(() => {
499
+ this.translating.delete(language);
500
+ const next = this.nextTranslation.get(language);
501
+ this.nextTranslation.delete(language);
502
+ if (next && !this.stopped) this.translated(language, next.line, next.mediaStart);
503
+ }));
466
504
  }
467
505
 
468
506
  /** A line for the store, kept with the others of its language until the next flush. */
@@ -494,7 +532,7 @@ class Captioner {
494
532
  const got = await keepLines(session, this.transcriptId, {
495
533
  media: this.media.media,
496
534
  title: this.media.title,
497
- language: language === "" ? this.language : language,
535
+ language,
498
536
  ...(language === "" ? {} : { translatedFrom: this.language }),
499
537
  ...(this.model ? { model: this.model } : {}),
500
538
  lines,
@@ -529,7 +567,7 @@ class Captioner {
529
567
  }
530
568
 
531
569
  recent(after: number, language = ""): CaptionLine[] {
532
- const lines = language === "" || language === this.language ? this.lines : (this.linesBy.get(language) ?? []);
570
+ const lines = language === "" ? this.lines : [...this.lines.filter((line) => line.language === language), ...(this.linesBy.get(language) ?? [])].sort((a, b) => a.at - b.at);
533
571
  return after > 0 ? lines.filter((line) => line.at > after) : [...lines];
534
572
  }
535
573
 
@@ -548,6 +586,9 @@ class Captioner {
548
586
  stop(): void {
549
587
  if (this.stopped) return;
550
588
  this.stopped = true;
589
+ for (const request of this.requests) request.abort();
590
+ this.requests.clear();
591
+ this.nextTranslation.clear();
551
592
  if (this.idle) clearTimeout(this.idle);
552
593
  this.idle = null;
553
594
  this.detach?.();
@@ -572,6 +613,25 @@ export class Captions {
572
613
  return this.options.session() !== null;
573
614
  }
574
615
 
616
+ capacity(id: string): boolean { return this.running.has(id) || this.running.size < MAX_CAPTIONERS; }
617
+
618
+ /** Voice requests are constrained to captions this channel actually produced. */
619
+ async voiceRequest(id: string, at: number | null, language: string, voice: string, signal: AbortSignal, grant = ""): Promise<Response> {
620
+ const session = this.options.session();
621
+ if (!session) throw new SpeechError("this server must sign in to use translated audio", 503);
622
+ if (at === null) return (this.options.fetcher ?? fetch)(`${session.site.replace(/\/+$/, "")}/api/v1/speech/voices`, {
623
+ headers: { authorization: `Bearer ${session.token}` }, signal,
624
+ });
625
+ if (!language) throw new SpeechError("choose a translation language first", 400);
626
+ if (!/^nxd_[A-Za-z0-9_-]{43}$/.test(grant)) throw new SpeechError("sign in to enable translated audio", 401);
627
+ const line = this.recent(id, 0, language).find(line => line.at === at);
628
+ if (!line || (this.options.now ?? Date.now)() - line.until > 30_000) throw new SpeechError("that live caption is no longer available for audio", 404);
629
+ return (this.options.fetcher ?? fetch)(`${session.site.replace(/\/+$/, "")}/api/v1/speech/synthesize`, {
630
+ method: "POST", headers: { authorization: `Bearer ${grant}`, "content-type": "application/json" },
631
+ body: JSON.stringify({ text: line.text, language, voice, profile: line.voiceProfile, channel: id }), signal,
632
+ });
633
+ }
634
+
575
635
  /**
576
636
  * Lines for a channel as they are heard, starting the captioner if it is
577
637
  * not running. Null when there is no such channel. The returned function
@@ -580,6 +640,7 @@ export class Captions {
580
640
  * "" is the original.
581
641
  */
582
642
  subscribe(id: string, subscriber: Subscriber, language = ""): (() => void) | null {
643
+ if (!this.capacity(id)) return null;
583
644
  let captioner = this.running.get(id);
584
645
  if (!captioner) {
585
646
  const made = new Captioner(id, this.options, () => {
@@ -0,0 +1,255 @@
1
+ /** ElevenLabs Flash for live captions. The key stays on the account server. */
2
+ import { createHash, randomBytes } from "node:crypto";
3
+ import { SpeechError, decodeWav, encodeWav, quietSamples } from "./speech.ts";
4
+ import { speakerTurns, type ScribeResult, type SpeakerTranscript } from "./speaker-turns.ts";
5
+ import type { VoiceProfile } from "./voice-profile.ts";
6
+ import { Guard } from "./guard.ts";
7
+ import type { Queryable } from "./follows.ts";
8
+
9
+ export const LIVE_VOICE_MODEL = "eleven_flash_v2_5";
10
+ export const LIVE_VOICE_RATE = 16_000;
11
+ export const LIVE_VOICE_LANGUAGES = new Set("en ja zh de hi fr ko pt it es id nl tr fil pl sv bg ro ar cs el fi hr ms sk da ta uk ru hu no vi".split(" "));
12
+ export interface LiveVoiceChoice { id: string; name: string; gender: string; language: string; }
13
+ export interface VoiceRequest { text: string; language: string; voice?: string; profile?: VoiceProfile; channel?: string; }
14
+ const HEADERS = { "content-type": "audio/pcm", "cache-control": "no-store", "x-audio-sample-rate": String(LIVE_VOICE_RATE) };
15
+ const budget = (value: number | undefined, fallback: number): number => Number.isFinite(value) && value! >= 0 ? Math.floor(value!) : fallback;
16
+
17
+ export class LiveVoice {
18
+ private readonly key: string;
19
+ private readonly fetcher: typeof fetch;
20
+ private readonly now: () => number;
21
+ private catalog: Promise<LiveVoiceChoice[]> | null = null;
22
+ private catalogUntil = 0;
23
+ private readonly cache = new Map<string, { until: number; bytes: Uint8Array }>();
24
+ private readonly pending = new Map<string, Promise<Uint8Array>>();
25
+ private readonly usage = new Map<string, { minute: number; chars: number }>();
26
+ private readonly charsPerMinute: number;
27
+ private readonly requests: Guard;
28
+ private readonly grants = new Map<string, { by: string; channel: string; expires: number; remaining: number }>();
29
+ private readonly activeBy = new Map<string, number>();
30
+ private readonly db?: Queryable;
31
+ private schema: Promise<unknown> | null = null;
32
+ private readonly dailyChars: number;
33
+ private cleanupAt = 0;
34
+ private readonly hearing = new Set<string>();
35
+ private readonly dailyAudioSeconds: number;
36
+ private readonly userDailyChars: number;
37
+ private readonly userDailyAudioSeconds: number;
38
+
39
+ constructor(options: { apiKey?: string; fetcher?: typeof fetch; now?: () => number; charsPerMinute?: number; dailyChars?: number; dailyAudioSeconds?: number; userDailyChars?: number; userDailyAudioSeconds?: number; db?: Queryable } = {}) {
40
+ this.key = options.apiKey ?? process.env["ELEVENLABS_API_KEY"] ?? "";
41
+ this.fetcher = options.fetcher ?? fetch;
42
+ this.now = options.now ?? Date.now;
43
+ this.charsPerMinute = budget(options.charsPerMinute, 3000);
44
+ this.dailyChars = budget(options.dailyChars, 200_000);
45
+ this.dailyAudioSeconds = budget(options.dailyAudioSeconds, 86_400);
46
+ this.userDailyChars = budget(options.userDailyChars, 120_000);
47
+ this.userDailyAudioSeconds = budget(options.userDailyAudioSeconds, 43_200);
48
+ this.requests = new Guard(this.now);
49
+ this.db = options.db;
50
+ }
51
+
52
+ available(): boolean { return this.key !== ""; }
53
+
54
+ /** Optional diarization, billed only while a signed-in listener requests it.
55
+ * Rolling audio is bounded to 15 seconds, including overlap. Every second
56
+ * submitted (also repeated context) consumes the persistent provider budget. */
57
+ async hear(bytes: Uint8Array, by: string, signal?: AbortSignal): Promise<SpeakerTranscript> {
58
+ if (!this.available()) throw new SpeechError("speaker voices are unavailable", 503);
59
+ const wav = decodeWav(bytes);
60
+ const seconds = wav.samples.length / wav.rate;
61
+ if (wav.rate !== 16000 || wav.channels !== 1 || seconds < 0.2 || seconds > 15.1) throw new SpeechError("send up to 15 seconds of mono 16 kHz WAV", 400);
62
+ if (wav.samples.some(sample => !Number.isFinite(sample))) throw new SpeechError("invalid audio samples", 400);
63
+ if (!this.requests.check(`hear:${by}`, { allowed: 20, windowMs: 60_000 }).ok) throw new SpeechError("too many speaker transcription requests", 429);
64
+ if (quietSamples(wav.samples)) return { language: "", seconds, turns: [] };
65
+ if (this.hearing.has(by) || this.hearing.size >= 4) throw new SpeechError("speaker transcription is busy", 429);
66
+ this.hearing.add(by);
67
+ try {
68
+ const billed = Math.ceil(seconds);
69
+ await this.reserve(`scribe:user:${by}`, billed, 300, 60_000);
70
+ await this.reserve("scribe:server", billed, 600, 60_000);
71
+ await this.reserve(`scribe:user:${by}`, billed, this.userDailyAudioSeconds, 86_400_000);
72
+ await this.reserve("scribe:server", billed, this.dailyAudioSeconds, 86_400_000);
73
+ signal?.throwIfAborted();
74
+ const form = new FormData();
75
+ // Canonical PCM prevents a crafted container from billing more audio
76
+ // than the duration we validated, and removes uploaded metadata.
77
+ form.set("file", new Blob([new Uint8Array(encodeWav(wav.samples))], { type: "audio/wav" }), "listening.wav");
78
+ form.set("model_id", "scribe_v2");
79
+ form.set("diarize", "true");
80
+ form.set("tag_audio_events", "false");
81
+ form.set("timestamps_granularity", "word");
82
+ // No language_code: preserve the source language, including Spanish.
83
+ const answer = await this.fetcher("https://api.elevenlabs.io/v1/speech-to-text", {
84
+ method: "POST", headers: { "xi-api-key": this.key }, body: form,
85
+ signal: signal ? AbortSignal.any([signal, AbortSignal.timeout(10_000)]) : AbortSignal.timeout(10_000),
86
+ });
87
+ if (!answer.ok) throw new SpeechError("speaker transcription could not run; check provider quota and permissions", answer.status === 429 ? 429 : 502);
88
+ return speakerTurns(await answer.json() as ScribeResult, wav);
89
+ } finally { this.hearing.delete(by); }
90
+ }
91
+
92
+ async voices(): Promise<LiveVoiceChoice[]> {
93
+ if (!this.available()) throw new SpeechError("translated audio needs ELEVENLABS_API_KEY on the account server", 503);
94
+ if (!this.catalog || this.now() >= this.catalogUntil) {
95
+ this.catalogUntil = this.now() + 3600_000;
96
+ this.catalog = this.fetcher("https://api.elevenlabs.io/v2/voices?page_size=100&voice_type=default", {
97
+ headers: { "xi-api-key": this.key }, signal: AbortSignal.timeout(8000),
98
+ }).then(async response => {
99
+ if (!response.ok) throw new SpeechError("ElevenLabs could not list voices; check the server key and its voice permissions", 503);
100
+ const body = await response.json() as { voices: { voice_id: string; name: string; labels?: Record<string, string> }[] };
101
+ return body.voices.filter(voice => /^[a-zA-Z0-9_-]+$/.test(voice.voice_id)).map(voice => ({
102
+ id: voice.voice_id, name: voice.name, gender: voice.labels?.["gender"] ?? "neutral", language: voice.labels?.["language"] ?? "",
103
+ }));
104
+ }).catch(error => { this.catalog = null; throw error; });
105
+ }
106
+ return this.catalog;
107
+ }
108
+
109
+ /** A 90-second capability for one channel, never the viewer's account credential. */
110
+ async grant(by: string, channel: string): Promise<{ token: string; expires: number }> {
111
+ if (!/^[\w-]{1,80}$/.test(channel)) throw new SpeechError("choose a playback session or live channel", 400);
112
+ if (!this.requests.check(`grant:${by}`, { allowed: 10, windowMs: 60_000 }).ok) throw new SpeechError("too many audio authorization requests", 429);
113
+ for (const [token, grant] of this.grants) if (grant.expires <= this.now()) this.grants.delete(token);
114
+ if (this.grants.size >= 5000) throw new SpeechError("audio authorization is busy", 503);
115
+ const token = `nxd_${randomBytes(32).toString("base64url")}`;
116
+ const expires = this.now() + 90_000;
117
+ if (this.db) {
118
+ await this.ensure();
119
+ await this.db.query("INSERT INTO live_voice_grants (token_hash, by_account, channel, expires_at, remaining) VALUES ($1, $2, $3, $4, 2000)", [createHash("sha256").update(token).digest("hex"), by, channel, new Date(expires)]);
120
+ } else this.grants.set(token, { by, channel, expires, remaining: 2000 });
121
+ return { token, expires };
122
+ }
123
+
124
+ async authorize(token: string, channel: string, chars: number): Promise<string> {
125
+ if (!/^nxd_[A-Za-z0-9_-]{43}$/.test(token)) throw new SpeechError("sign in to enable translated audio", 401);
126
+ if (!Number.isFinite(chars) || chars < 1 || chars > 600) throw new SpeechError("invalid caption length", 400);
127
+ if (this.db) {
128
+ await this.ensure();
129
+ const result = await this.db.query(`UPDATE live_voice_grants SET remaining = remaining - $3
130
+ WHERE token_hash = $1 AND channel = $2 AND expires_at > $4 AND remaining >= $3 RETURNING by_account`, [createHash("sha256").update(token).digest("hex"), channel, chars, new Date(this.now())]);
131
+ if (!result.rows.length) throw new SpeechError("audio authorization expired or reached its limit; enable translated audio again", 401);
132
+ return String(result.rows[0]?.["by_account"]);
133
+ }
134
+ const grant = this.grants.get(token);
135
+ if (!grant || grant.expires <= this.now() || grant.channel !== channel) throw new SpeechError("sign in to renew translated audio", 401);
136
+ if (grant.remaining < chars) throw new SpeechError("this audio authorization reached its character limit", 429);
137
+ grant.remaining -= chars;
138
+ return grant.by;
139
+ }
140
+
141
+ private async ensure(): Promise<void> {
142
+ if (!this.db) return;
143
+ this.schema ??= (async () => {
144
+ await this.db!.query("CREATE TABLE IF NOT EXISTS live_voice_usage (bucket TEXT PRIMARY KEY, chars BIGINT NOT NULL, expires_at TIMESTAMPTZ NOT NULL)");
145
+ await this.db!.query("CREATE TABLE IF NOT EXISTS live_voice_grants (token_hash TEXT PRIMARY KEY, by_account TEXT NOT NULL, channel TEXT NOT NULL, expires_at TIMESTAMPTZ NOT NULL, remaining INTEGER NOT NULL)");
146
+ })().catch(error => { this.schema = null; throw error; });
147
+ await this.schema;
148
+ if (this.now() >= this.cleanupAt) {
149
+ this.cleanupAt = this.now() + 60_000;
150
+ // Keep only live windows; accounting survives restarts and is shared by replicas.
151
+ await this.db.query("DELETE FROM live_voice_usage WHERE expires_at <= $1", [new Date(this.now())]);
152
+ await this.db.query("DELETE FROM live_voice_grants WHERE expires_at <= $1", [new Date(this.now())]);
153
+ }
154
+ }
155
+
156
+ private async reserve(bucket: string, chars: number, limit: number, windowMs: number): Promise<void> {
157
+ if (chars > limit) throw new SpeechError("translated audio reached its character budget; try again later", 429);
158
+ const window = Math.floor(this.now() / windowMs);
159
+ const key = `${bucket}:${windowMs}:${window}`;
160
+ if (this.db) {
161
+ await this.ensure();
162
+ const result = await this.db.query(`INSERT INTO live_voice_usage (bucket, chars, expires_at) VALUES ($1, $2, $4)
163
+ ON CONFLICT (bucket) DO UPDATE SET chars = live_voice_usage.chars + EXCLUDED.chars
164
+ WHERE live_voice_usage.chars + EXCLUDED.chars <= $3 RETURNING chars`, [key, chars, limit, new Date((window + 1) * windowMs)]);
165
+ if (!result.rows.length) throw new SpeechError("translated audio reached its character budget; try again later", 429);
166
+ return;
167
+ }
168
+ // Local/single-process installations without a database. Account servers pass their pool.
169
+ const record = this.usage.get(key) ?? { minute: (window + 1) * windowMs, chars: 0 };
170
+ if (record.chars + chars > limit) throw new SpeechError("translated audio reached its character budget; try again later", 429);
171
+ record.chars += chars; this.usage.set(key, record);
172
+ }
173
+
174
+ private async charge(by: string, channel: string, chars: number): Promise<void> {
175
+ for (const [key, record] of this.usage) if (record.minute <= this.now()) this.usage.delete(key);
176
+ await this.reserve(`user:${by}`, chars, this.charsPerMinute, 60_000);
177
+ await this.reserve(`channel:${channel}`, chars, 6000, 60_000);
178
+ await this.reserve("server", chars, 12_000, 60_000);
179
+ await this.reserve(`user:${by}`, chars, this.userDailyChars, 86400_000);
180
+ await this.reserve("server", chars, this.dailyChars, 86400_000);
181
+ }
182
+
183
+ private checkRequest(by: string): void {
184
+ if (!this.requests.check(`audio:${by}`, { allowed: 60, windowMs: 60_000 }).ok) throw new SpeechError("too many translated audio requests", 429);
185
+ }
186
+
187
+ /** Stream the first request immediately; concurrent listeners share its cached result. */
188
+ async stream(ask: VoiceRequest, by: string, signal?: AbortSignal): Promise<Response> {
189
+ this.checkRequest(by);
190
+ const text = typeof ask.text === "string" ? ask.text.trim() : "";
191
+ if (!text || text.length > 600) throw new SpeechError("translated audio needs a caption of 1–600 characters", 400);
192
+ if (!LIVE_VOICE_LANGUAGES.has(ask.language)) throw new SpeechError("this language is not supported by Flash voices", 400);
193
+ const voices = await this.voices();
194
+ signal?.throwIfAborted();
195
+ const gender = ask.profile === "lower" ? "male" : ask.profile === "higher" ? "female" : "neutral";
196
+ const voice = ask.voice && ask.voice !== "auto"
197
+ ? voices.find(voice => voice.id === ask.voice)
198
+ : voices.find(voice => voice.gender === gender) ?? voices[0];
199
+ if (!voice) throw new SpeechError("choose an available voice", 400);
200
+ const id = createHash("sha256").update(JSON.stringify([LIVE_VOICE_MODEL, voice.id, ask.language, text])).digest("hex");
201
+ for (const [key, item] of this.cache) if (item.until < this.now()) this.cache.delete(key);
202
+ const cached = this.cache.get(id);
203
+ if (cached) return new Response(new Uint8Array(cached.bytes), { headers: HEADERS });
204
+ const pending = this.pending.get(id);
205
+ if (pending) return new Response(new Uint8Array(await pending), { headers: HEADERS });
206
+ if (this.pending.size >= 4) throw new SpeechError("translated audio is busy; waiting for the next caption", 429);
207
+ if ((this.activeBy.get(by) ?? 0) >= 2) throw new SpeechError("two voice requests are already active for this account", 429);
208
+ // Register before the first provider await, so identical requests cannot both bill.
209
+ let done!: (bytes: Uint8Array) => void;
210
+ let fail!: (error: unknown) => void;
211
+ const finished = new Promise<Uint8Array>((resolve, reject) => { done = resolve; fail = reject; });
212
+ this.pending.set(id, finished);
213
+ this.activeBy.set(by, (this.activeBy.get(by) ?? 0) + 1);
214
+ const release = (): void => { this.pending.delete(id); this.activeBy.set(by, Math.max(0, (this.activeBy.get(by) ?? 1) - 1)); if (!this.activeBy.get(by)) this.activeBy.delete(by); };
215
+ void finished.catch(() => undefined);
216
+ try {
217
+ await this.charge(by, ask.channel ?? "direct", text.length);
218
+ signal?.throwIfAborted();
219
+ const response = await this.fetcher(`https://api.elevenlabs.io/v1/text-to-speech/${voice.id}/stream?output_format=pcm_16000`, {
220
+ method: "POST",
221
+ headers: { "xi-api-key": this.key, "content-type": "application/json" },
222
+ body: JSON.stringify({ text, model_id: LIVE_VOICE_MODEL, language_code: ask.language }),
223
+ signal: signal ? AbortSignal.any([signal, AbortSignal.timeout(10_000)]) : AbortSignal.timeout(10_000),
224
+ });
225
+ if (!response.ok || !response.body) throw new SpeechError(response.status === 429 ? "ElevenLabs audio quota is temporarily exhausted" : "ElevenLabs could not generate audio; check the server key and quota", response.status === 429 ? 429 : 502);
226
+ const [play, keep] = response.body.tee();
227
+ void (async () => {
228
+ const reader = keep.getReader();
229
+ const chunks: Uint8Array[] = [];
230
+ let size = 0;
231
+ try {
232
+ while (true) {
233
+ const { done: ended, value } = await reader.read();
234
+ if (ended) break;
235
+ size += value.length;
236
+ if (size > 2 * 1024 * 1024) { await reader.cancel(); throw new SpeechError("voice response was too long", 502); }
237
+ chunks.push(value);
238
+ }
239
+ if (!size || size % 2) throw new SpeechError("voice response contained incomplete audio", 502);
240
+ const bytes = new Uint8Array(size);
241
+ let at = 0;
242
+ for (const chunk of chunks) { bytes.set(chunk, at); at += chunk.length; }
243
+ this.cache.set(id, { until: this.now() + 60_000, bytes });
244
+ // At most a few MB for recent lines, never a recording archive.
245
+ while (this.cache.size > 32) this.cache.delete(this.cache.keys().next().value as string);
246
+ done(bytes);
247
+ } catch (error) { fail(error); }
248
+ finally { reader.releaseLock(); release(); }
249
+ })();
250
+ return new Response(play, { headers: HEADERS });
251
+ } catch (error) {
252
+ fail(error); release(); throw error;
253
+ }
254
+ }
255
+ }