nixamp 0.25.0 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +108 -7
  2. package/dist/captions.d.ts +7 -0
  3. package/dist/captions.js +92 -28
  4. package/dist/live-voice.d.ts +70 -0
  5. package/dist/live-voice.js +299 -0
  6. package/dist/server.d.ts +2 -0
  7. package/dist/server.js +154 -5
  8. package/dist/speaker-turns.d.ts +32 -0
  9. package/dist/speaker-turns.js +30 -0
  10. package/dist/speech.d.ts +47 -0
  11. package/dist/speech.js +49 -4
  12. package/dist/transcript-client.d.ts +2 -2
  13. package/dist/transcript-client.js +5 -2
  14. package/dist/transcripts.d.ts +5 -0
  15. package/dist/transcripts.js +8 -3
  16. package/dist/translate-jobs.js +13 -7
  17. package/dist/translate.d.ts +2 -0
  18. package/dist/translate.js +9 -0
  19. package/dist/voice-profile.d.ts +7 -0
  20. package/dist/voice-profile.js +53 -0
  21. package/dist/warm.js +3 -3
  22. package/package.json +1 -1
  23. package/src/captions.ts +86 -25
  24. package/src/live-voice.ts +255 -0
  25. package/src/server.ts +104 -5
  26. package/src/speaker-turns.ts +34 -0
  27. package/src/speech.ts +46 -7
  28. package/src/transcript-client.ts +4 -0
  29. package/src/transcripts.ts +13 -3
  30. package/src/translate-jobs.ts +13 -7
  31. package/src/translate.ts +7 -1
  32. package/src/voice-profile.ts +46 -0
  33. package/src/warm.ts +3 -3
  34. package/web/dist/assets/{hls-3VKVEQE3-B4ltbKDh.js → hls-3VKVEQE3-vgax_tk1.js} +1 -1
  35. package/web/dist/assets/index-BzjrTOLf.js +1 -0
  36. package/web/dist/assets/index-D2Iy07pG.css +1 -0
  37. package/web/dist/assets/{mpegts-Byy3EkfT.js → mpegts-DmcUOiHq.js} +1 -1
  38. package/web/dist/assets/{mpegts-LO6RVLD6-C9qqolrW.js → mpegts-LO6RVLD6-CH3EQi6L.js} +1 -1
  39. package/web/dist/index.html +42 -22
  40. package/web/dist/sw.js +6 -6
  41. package/web/dist/assets/index-d7TvpeFZ.js +0 -1
  42. package/web/dist/assets/index-oyp61Kly.css +0 -1
package/README.md CHANGED
@@ -421,11 +421,18 @@ nixamp.com instead. `NIXAMP_STT_MODEL` picks another Whisper
421
421
  takes twice as long), `NIXAMP_STT_CACHE` says where its files are kept, and
422
422
  `NIXAMP_STT=off` leaves the ear out of a deployment altogether.
423
423
 
424
- The ear tells which language it heard: one pass over the first thirty
425
- seconds, the way whisper.cpp does it, before the words are read. Without
426
- that a Swedish channel came back as three English words repeated to the end
427
- of the window. A captioner learns the language from its first line and says
428
- it on every ask after that.
424
+ Captions default to **Original (auto-detect)**. Each audio window detects its
425
+ own language and explicitly transcribes it. A language selected in the menu
426
+ only affects translation; neither it nor a cached transcript can force the
427
+ recognizer into English. Short windows have a decoding limit, and silent
428
+ audio and repetitive hallucinations are discarded. Language is stored on
429
+ each line, so an interview can switch languages. Legacy live-caption cache
430
+ entries are heard again instead of replaying their incorrect words.
431
+
432
+ At most four channels are captioned per server, with two recognition requests
433
+ per channel in flight. Live work expires after twelve seconds. Translation
434
+ keeps one active request and the latest pending line per target; joining a
435
+ live reads cached translations without starting a whole-transcript job.
429
436
 
430
437
  ### Kept: written down once, for everybody
431
438
 
@@ -523,8 +530,89 @@ The page has the same choice beside the Captions switch, remembered per
523
530
  device; a translated line is marked with its language and shows what was
524
531
  heard under the pointer. An agent has `translate_text`. `NIXAMP_MT_WARM`
525
532
  names pairs to load at boot (`en-de,en-sv`), `NIXAMP_MT=off` leaves
526
- translation out, and the Docker image bakes the ear and the German and
527
- Swedish pairs in so a deploy never downloads them again.
533
+ translation out. The Docker image includes the ear, German/Swedish pairs
534
+ with English, and Spanish pairs with English and German. Spanish-to-German
535
+ uses its direct model; it does not first translate the audio into English.
536
+
537
+
538
+ ### Hear it in your language
539
+
540
+ The browser's Transcript panel works with movies, shows, sports, courses,
541
+ podcasts, live channels, and rooms. Choose a language and enable **Play
542
+ translated audio**. This is a listener preference: the video and the room's
543
+ shared playback clock keep running. Turn it off to restore the original sound.
544
+ **Start player captions** transcribes other playback in its detected source
545
+ language. Files are played locally; these explicit controls opt into sending
546
+ short audio clips for processing. Ordinary playback uploads no audio.
547
+
548
+ **Translate another tab** opens the browser's audio-sharing chooser, so a
549
+ watch party or video hosted on another site can be interpreted too. Select a
550
+ browser tab with **Share audio** enabled; **Stop listening** releases sharing.
551
+ Only audio is sent, even though the browser requires a video track to select
552
+ its source. Sharing support depends on the browser and the source's permissions;
553
+ protected media and sources whose audio cannot be captured remain unavailable.
554
+ The control requests suppression of the source tab's local sound. If a browser
555
+ ignores that option, mute the source tab to avoid hearing both languages.
556
+
557
+ Native captions use local Whisper Base through Transformers.js. Optional
558
+ speaker-aware audio uses ElevenLabs **Scribe v2** for native transcription with
559
+ speaker turns, local **OPUS-MT** for the selected translation, and ElevenLabs
560
+ **Flash v2.5** HTTP streaming for natural stock voices. The application sends
561
+ short audio clips and translated text to ElevenLabs only for this optional
562
+ feature. This uses the direct API; it does not need an MCP server or clone voices.
563
+ Supported translation pairs come from `/api/v1/translate`; voice languages are
564
+ also checked before enabling the audio toggle.
565
+
566
+ A rolling 15-second audio window advances every 5 seconds. Speaker labels are
567
+ reconciled using overlapping timestamps, with different voices assigned to
568
+ separate speakers. Lower/higher pitch suggests a male/female stock voice;
569
+ ambiguous audio uses a default. Each speaker's voice can be changed in the
570
+ panel. Pitch is not gender identity, and a speaker returning after leaving the
571
+ rolling context may receive a new label. Simultaneous speech and noisy crowds
572
+ can still confuse recognition. Native captions never translate to English as
573
+ an intermediate recognition step.
574
+
575
+ Processing has one active request and only the latest pending window per
576
+ listener; speech queues and response sizes are bounded. Old transcript history
577
+ is never spoken. Pause, seek, source changes, and disabling the feature cancel
578
+ queued speech; errors restore the original audio. This is a delayed live
579
+ interpreter, not a promise of exact lip sync or background-music separation.
580
+
581
+ The account server needs `ELEVENLABS_API_KEY`; `NIXAMP_DUBBING=off` disables
582
+ this feature. The key stays on the server. Sign-in is required for speaker
583
+ analysis, voice selection, and short-lived playback grants. Grants expire after
584
+ 90 seconds, authorize at most 2,000 characters, and are scoped to one playback
585
+ session or channel. Provider requests also have account/IP throttles, concurrency
586
+ limits, and cached duplicate voice generation. Native speech and local
587
+ translation retain their existing account and queue limits.
588
+
589
+ Postgres stores atomic usage reservations and hashed grants, so the feature's
590
+ budgets survive restarts and are shared between replicas. Provider failures
591
+ still consume reservations conservatively. The configurable daily limits are:
592
+
593
+ | Setting | Default | Counts |
594
+ | --- | ---: | --- |
595
+ | `NIXAMP_DUB_DAILY_CHARS` | 200,000 | New voice characters across this server |
596
+ | `NIXAMP_DUB_USER_DAILY_CHARS` | 120,000 | New voice characters per account |
597
+ | `NIXAMP_DUB_DAILY_AUDIO_SECONDS` | 86,400 | Scribe audio seconds across this server |
598
+ | `NIXAMP_DUB_USER_DAILY_AUDIO_SECONDS` | 43,200 | Scribe audio seconds per account |
599
+
600
+ Audio limits count overlapping context too: a continuous hour of speaker-aware
601
+ listening submits about three hours of Scribe audio. At the [published API
602
+ rates](https://elevenlabs.io/pricing/api) of $0.05 per 1,000 Flash characters and
603
+ $0.22 per Scribe audio hour, a listener producing 1,000 translated characters
604
+ per minute costs about $3.66/hour, before plan minimums or discounts. These default
605
+ server quotas limit this feature to about $15.28/day at those rates; they do not
606
+ cover other applications using the same provider key. Set a daily limit to zero
607
+ to block new use of that resource. Limits return 429 and never trigger an
608
+ unlimited fallback provider.
609
+
610
+ ```
611
+ GET /api/v1/speech/voices authenticated stock voices and supported audio languages
612
+ POST /api/v1/speech/speakers authenticated, bounded mono 16 kHz WAV -> native speaker turns
613
+ POST /api/v1/speech/grant authenticated {channel: playbackScope} -> short-lived grant
614
+ POST /api/v1/speech/synthesize scoped grant + {channel, text, language, voice, profile} -> streaming PCM
615
+ ```
528
616
 
529
617
  ## Several streams at once
530
618
 
@@ -908,3 +996,16 @@ make the name honest.
908
996
  ## Licence
909
997
 
910
998
  MIT
999
+
1000
+ ### Accessibility and interaction
1001
+
1002
+ Keyboard access, named controls, headings, skip links, visible focus, descriptive
1003
+ slider values, reduced motion, and concise screen-reader status announcements
1004
+ are built into the web player. Incoming transcripts, chat, and playback updates
1005
+ preserve scrolling, focus, caret, and the browsing page. The
1006
+ [UX and accessibility baseline](docs/ux.md) applies to all interface changes.
1007
+
1008
+ When an older broadcaster still provides unversioned English-first captions,
1009
+ a signed-in listener uses native-language recognition of the playing audio
1010
+ instead of that stale transcript cache. Update the broadcaster for shared native
1011
+ captions; translated audio uses the listener's current audio and selected language.
@@ -12,6 +12,8 @@ export interface CaptionLine {
12
12
  language?: string;
13
13
  /** What was heard, when this line is a translation of it. */
14
14
  original?: string;
15
+ sourceLanguage?: string;
16
+ voiceProfile?: "lower" | "higher" | "unknown";
15
17
  }
16
18
  /** What turns a channel's bytes into 16 kHz mono 16-bit PCM. ffmpeg, or a test's stand-in. */
17
19
  export interface Decoder {
@@ -71,6 +73,8 @@ export declare const IDLE_MS = 60000;
71
73
  export declare const QUIET = 0.004;
72
74
  /** Windows waiting on the ear at once. Past this the sound is dropped, not queued: late words are worse than none. */
73
75
  export declare const IN_FLIGHT = 2;
76
+ export declare const LIVE_DEADLINE_MS = 12000;
77
+ export declare const MAX_CAPTIONERS = 4;
74
78
  /** Heard lines wait this long, at most, before they are kept. */
75
79
  export declare const FLUSH_MS = 20000;
76
80
  /** Or this many. */
@@ -118,6 +122,9 @@ export declare class Captions {
118
122
  constructor(options: CaptionsOptions);
119
123
  /** Whether this server can caption at all: it has to be signed in for the ear to answer it. */
120
124
  available(): boolean;
125
+ capacity(id: string): boolean;
126
+ /** Voice requests are constrained to captions this channel actually produced. */
127
+ voiceRequest(id: string, at: number | null, language: string, voice: string, signal: AbortSignal, grant?: string): Promise<Response>;
121
128
  /**
122
129
  * Lines for a channel as they are heard, starting the captioner if it is
123
130
  * not running. Null when there is no such channel. The returned function
package/dist/captions.js CHANGED
@@ -31,7 +31,8 @@
31
31
  * handed to whoever wanted that language, and kept beside the original.
32
32
  */
33
33
  import { spawn } from "node:child_process";
34
- import { RATE } from "./speech.js";
34
+ import { NATIVE_REVISION, RATE, SpeechError, reliableText } from "./speech.js";
35
+ import { voiceProfile } from "./voice-profile.js";
35
36
  import { fetchTranscript, keepLines, keepMedia, translateTexts } from "./transcript-client.js";
36
37
  import { covered, lineAt, transcriptIdOf } from "./transcripts.js";
37
38
  export const WINDOW_MS = 5000;
@@ -42,6 +43,8 @@ export const IDLE_MS = 60_000;
42
43
  export const QUIET = 0.004;
43
44
  /** Windows waiting on the ear at once. Past this the sound is dropped, not queued: late words are worse than none. */
44
45
  export const IN_FLIGHT = 2;
46
+ export const LIVE_DEADLINE_MS = 12_000;
47
+ export const MAX_CAPTIONERS = 4;
45
48
  /** Heard lines wait this long, at most, before they are kept. */
46
49
  export const FLUSH_MS = 20_000;
47
50
  /** Or this many. */
@@ -174,6 +177,9 @@ class Captioner {
174
177
  stopped = false;
175
178
  complainedAt = 0;
176
179
  windows = 0;
180
+ audioUntil = 0;
181
+ lastEmittedAt = -Infinity;
182
+ requests = new Set();
177
183
  /** What the channel is playing, when the store is to be told. */
178
184
  media;
179
185
  transcriptId;
@@ -186,6 +192,8 @@ class Captioner {
186
192
  flush = null;
187
193
  /** Translations in order, per language: a slow one must not overtake the next. */
188
194
  chains = new Map();
195
+ translating = new Set();
196
+ nextTranslation = new Map();
189
197
  constructor(id, options, onStop) {
190
198
  this.id = id;
191
199
  this.options = options;
@@ -248,7 +256,7 @@ class Captioner {
248
256
  const session = this.options.session();
249
257
  if (session === null)
250
258
  return;
251
- const got = await fetchTranscript(session, this.transcriptId, language, this.options.fetcher ?? fetch);
259
+ const got = await fetchTranscript(session, this.transcriptId, language, this.options.fetcher ?? fetch, true);
252
260
  if (this.stopped)
253
261
  return;
254
262
  if (!got.ok) {
@@ -256,11 +264,12 @@ class Captioner {
256
264
  this.complain(`the store did not answer: ${got.error}`);
257
265
  return;
258
266
  }
259
- this.known.set(language, got.body.lines);
260
- if (language === "" && this.language === "" && got.body.language)
261
- this.language = got.body.language;
262
- if (got.body.lines.length > 0) {
263
- this.options.onEvent?.(`captions for "${this.id}": the store knows ${got.body.lines.length} lines of this${language ? ` in ${language}` : ""}`);
267
+ // A row's model/language could have been updated while its old bad lines remained.
268
+ // Trust individual lines made with the corrected native pipeline only.
269
+ const usable = got.body.lines.filter((line) => line.revision === NATIVE_REVISION && reliableText(line.text, line.end - line.start));
270
+ this.known.set(language, usable);
271
+ if (usable.length > 0) {
272
+ this.options.onEvent?.(`captions for "${this.id}": the store knows ${usable.length} lines of this${language ? ` in ${language}` : ""}`);
264
273
  }
265
274
  }
266
275
  onPcm(pcm) {
@@ -275,7 +284,9 @@ class Captioner {
275
284
  const rest = all.subarray(size);
276
285
  this.pending = rest.length > 0 ? [Buffer.from(rest)] : [];
277
286
  this.pendingBytes = rest.length;
278
- const until = this.now();
287
+ const duration = this.options.windowMs ?? WINDOW_MS;
288
+ const until = Math.max(this.audioUntil + duration, this.now() - rest.length / (RATE * 2) * 1000);
289
+ this.audioUntil = until;
279
290
  const index = this.windows;
280
291
  this.windows += 1;
281
292
  void this.hear(Buffer.from(window), until - (this.options.windowMs ?? WINDOW_MS), until, index);
@@ -294,7 +305,8 @@ class Captioner {
294
305
  continue;
295
306
  this.readOut.add(line.start);
296
307
  const lineAt = at + (line.start - span.start) * 1000;
297
- this.emit({ channel: this.id, at: lineAt, until: lineAt + (line.end - line.start) * 1000, text: line.text, ...(this.language ? { language: this.language } : {}) }, line.start);
308
+ this.emit({ channel: this.id, at: lineAt, until: lineAt + (line.end - line.start) * 1000, text: line.text,
309
+ ...(line.language ? { language: line.language } : {}), ...(line.voiceProfile ? { voiceProfile: line.voiceProfile } : {}) }, line.start);
298
310
  }
299
311
  return;
300
312
  }
@@ -307,43 +319,52 @@ class Captioner {
307
319
  return;
308
320
  }
309
321
  this.inFlight += 1;
322
+ const controller = new AbortController();
323
+ this.requests.add(controller);
324
+ const timeout = setTimeout(() => controller.abort(), LIVE_DEADLINE_MS);
310
325
  try {
311
326
  const wav = wavAround(pcm);
312
327
  const url = new URL(`${session.site.replace(/\/+$/, "")}/api/v1/speech/transcribe`);
313
- if (this.language)
314
- url.searchParams.set("language", this.language);
328
+ // Let each audio window detect its own language. Cached text and a viewer's
329
+ // translation selection must never constrain the recognizer.
330
+ url.searchParams.set("live", "1");
315
331
  const response = await (this.options.fetcher ?? fetch)(url.toString(), {
316
332
  method: "POST",
317
333
  headers: { authorization: `Bearer ${session.token}`, "content-type": "audio/wav" },
318
334
  body: new Blob([wav.buffer.slice(wav.byteOffset, wav.byteOffset + wav.byteLength)]),
335
+ signal: controller.signal,
319
336
  });
320
337
  const body = (await response.json().catch(() => ({})));
321
338
  if (!response.ok) {
322
339
  this.complain(body.error ?? `nixamp.com answered ${response.status}`);
323
340
  return;
324
341
  }
325
- const text = (body.text ?? "").trim();
326
- if (text === "" || this.stopped)
342
+ const text = reliableText(body.text ?? "", (until - at) / 1000);
343
+ if (text === "" || this.stopped || controller.signal.aborted || this.now() - until > LIVE_DEADLINE_MS)
327
344
  return;
328
345
  this.error = "";
329
- if (body.language && this.language === "")
330
- this.language = body.language;
331
346
  if (body.model)
332
347
  this.model = body.model;
333
- const line = { channel: this.id, at, until, text, ...(this.language ? { language: this.language } : {}) };
348
+ const line = { channel: this.id, at, until, text, ...(body.language ? { language: body.language } : {}), voiceProfile: voiceProfile(pcm) };
334
349
  this.emit(line, span?.start ?? null);
335
350
  if (span)
336
- this.keep("", { start: span.start, end: span.end, text });
351
+ this.keep("", { start: span.start, end: span.end, text, language: line.language, voiceProfile: line.voiceProfile, revision: NATIVE_REVISION });
337
352
  }
338
353
  catch (error) {
339
354
  this.complain(`could not reach the ear: ${error.message}`);
340
355
  }
341
356
  finally {
357
+ clearTimeout(timeout);
358
+ this.requests.delete(controller);
342
359
  this.inFlight -= 1;
343
360
  }
344
361
  }
345
362
  /** A line as heard, to whoever wants the original, and translated to whoever wants another language. */
346
363
  emit(line, mediaStart) {
364
+ if (line.at <= this.lastEmittedAt)
365
+ return;
366
+ this.lastEmittedAt = line.at;
367
+ this.language = line.language ?? "";
347
368
  this.lines.push(line);
348
369
  while (this.lines.length > KEEP)
349
370
  this.lines.shift();
@@ -372,32 +393,41 @@ class Captioner {
372
393
  }
373
394
  /** The line in another language: from the store when it has been through this moment, from nixamp.com otherwise. */
374
395
  translated(language, line, mediaStart) {
375
- const chain = (this.chains.get(language) ?? Promise.resolve()).then(async () => {
376
- if (this.stopped)
396
+ // One active request and the latest pending line per target, never a promise
397
+ // chain containing minutes of stale commentary.
398
+ if (this.translating.has(language)) {
399
+ this.nextTranslation.set(language, { line, mediaStart });
400
+ return;
401
+ }
402
+ this.translating.add(language);
403
+ const chain = Promise.resolve().then(async () => {
404
+ if (this.stopped || !this.wanted().has(language) || this.now() - line.until > LIVE_DEADLINE_MS)
377
405
  return;
378
406
  let text = "";
379
407
  const stored = mediaStart === null ? null : lineAt(this.known.get(language) ?? [], mediaStart);
380
- if (stored) {
408
+ if (stored && stored.original === line.text) {
381
409
  text = stored.text;
382
410
  }
383
411
  else {
384
412
  const session = this.options.session();
385
413
  if (session === null)
386
414
  return;
387
- const got = await translateTexts(session, [line.text], this.language, language, this.options.fetcher ?? fetch);
415
+ if (!line.language)
416
+ return;
417
+ const got = await translateTexts(session, [line.text], line.language, language, this.options.fetcher ?? fetch, AbortSignal.timeout(LIVE_DEADLINE_MS));
388
418
  if (!got.ok) {
389
419
  this.complain(`could not translate to ${language}: ${got.error}`);
390
420
  return;
391
421
  }
392
- text = (got.body.texts[0] ?? "").trim();
422
+ text = reliableText(got.body.texts[0] ?? "", (line.until - line.at) / 1000);
393
423
  if (text === "")
394
424
  return;
395
425
  if (mediaStart !== null)
396
- this.keep(language, { start: mediaStart, end: round(mediaStart + (line.until - line.at) / 1000), text });
426
+ this.keep(language, { start: mediaStart, end: round(mediaStart + (line.until - line.at) / 1000), text, original: line.text, revision: NATIVE_REVISION });
397
427
  }
398
- if (this.stopped)
428
+ if (this.stopped || !this.wanted().has(language) || this.now() - line.until > LIVE_DEADLINE_MS)
399
429
  return;
400
- const said = { ...line, text, language, original: line.text };
430
+ const said = { ...line, text, language, sourceLanguage: line.language, original: line.text };
401
431
  const lines = this.linesBy.get(language) ?? [];
402
432
  lines.push(said);
403
433
  while (lines.length > KEEP)
@@ -407,7 +437,13 @@ class Captioner {
407
437
  if (wanted === language)
408
438
  this.tell(subscriber, said);
409
439
  });
410
- this.chains.set(language, chain.catch(() => undefined));
440
+ this.chains.set(language, chain.catch(() => undefined).finally(() => {
441
+ this.translating.delete(language);
442
+ const next = this.nextTranslation.get(language);
443
+ this.nextTranslation.delete(language);
444
+ if (next && !this.stopped)
445
+ this.translated(language, next.line, next.mediaStart);
446
+ }));
411
447
  }
412
448
  /** A line for the store, kept with the others of its language until the next flush. */
413
449
  keep(language, line) {
@@ -441,7 +477,7 @@ class Captioner {
441
477
  const got = await keepLines(session, this.transcriptId, {
442
478
  media: this.media.media,
443
479
  title: this.media.title,
444
- language: language === "" ? this.language : language,
480
+ language,
445
481
  ...(language === "" ? {} : { translatedFrom: this.language }),
446
482
  ...(this.model ? { model: this.model } : {}),
447
483
  lines,
@@ -478,7 +514,7 @@ class Captioner {
478
514
  };
479
515
  }
480
516
  recent(after, language = "") {
481
- const lines = language === "" || language === this.language ? this.lines : (this.linesBy.get(language) ?? []);
517
+ const lines = language === "" ? this.lines : [...this.lines.filter((line) => line.language === language), ...(this.linesBy.get(language) ?? [])].sort((a, b) => a.at - b.at);
482
518
  return after > 0 ? lines.filter((line) => line.at > after) : [...lines];
483
519
  }
484
520
  status() {
@@ -496,6 +532,10 @@ class Captioner {
496
532
  if (this.stopped)
497
533
  return;
498
534
  this.stopped = true;
535
+ for (const request of this.requests)
536
+ request.abort();
537
+ this.requests.clear();
538
+ this.nextTranslation.clear();
499
539
  if (this.idle)
500
540
  clearTimeout(this.idle);
501
541
  this.idle = null;
@@ -520,6 +560,28 @@ export class Captions {
520
560
  available() {
521
561
  return this.options.session() !== null;
522
562
  }
563
+ capacity(id) { return this.running.has(id) || this.running.size < MAX_CAPTIONERS; }
564
+ /** Voice requests are constrained to captions this channel actually produced. */
565
+ async voiceRequest(id, at, language, voice, signal, grant = "") {
566
+ const session = this.options.session();
567
+ if (!session)
568
+ throw new SpeechError("this server must sign in to use translated audio", 503);
569
+ if (at === null)
570
+ return (this.options.fetcher ?? fetch)(`${session.site.replace(/\/+$/, "")}/api/v1/speech/voices`, {
571
+ headers: { authorization: `Bearer ${session.token}` }, signal,
572
+ });
573
+ if (!language)
574
+ throw new SpeechError("choose a translation language first", 400);
575
+ if (!/^nxd_[A-Za-z0-9_-]{43}$/.test(grant))
576
+ throw new SpeechError("sign in to enable translated audio", 401);
577
+ const line = this.recent(id, 0, language).find(line => line.at === at);
578
+ if (!line || (this.options.now ?? Date.now)() - line.until > 30_000)
579
+ throw new SpeechError("that live caption is no longer available for audio", 404);
580
+ return (this.options.fetcher ?? fetch)(`${session.site.replace(/\/+$/, "")}/api/v1/speech/synthesize`, {
581
+ method: "POST", headers: { authorization: `Bearer ${grant}`, "content-type": "application/json" },
582
+ body: JSON.stringify({ text: line.text, language, voice, profile: line.voiceProfile, channel: id }), signal,
583
+ });
584
+ }
523
585
  /**
524
586
  * Lines for a channel as they are heard, starting the captioner if it is
525
587
  * not running. Null when there is no such channel. The returned function
@@ -528,6 +590,8 @@ export class Captions {
528
590
  * "" is the original.
529
591
  */
530
592
  subscribe(id, subscriber, language = "") {
593
+ if (!this.capacity(id))
594
+ return null;
531
595
  let captioner = this.running.get(id);
532
596
  if (!captioner) {
533
597
  const made = new Captioner(id, this.options, () => {
@@ -0,0 +1,70 @@
1
+ import { type SpeakerTranscript } from "./speaker-turns.ts";
2
+ import type { VoiceProfile } from "./voice-profile.ts";
3
+ import type { Queryable } from "./follows.ts";
4
+ export declare const LIVE_VOICE_MODEL = "eleven_flash_v2_5";
5
+ export declare const LIVE_VOICE_RATE = 16000;
6
+ export declare const LIVE_VOICE_LANGUAGES: Set<string>;
7
+ export interface LiveVoiceChoice {
8
+ id: string;
9
+ name: string;
10
+ gender: string;
11
+ language: string;
12
+ }
13
+ export interface VoiceRequest {
14
+ text: string;
15
+ language: string;
16
+ voice?: string;
17
+ profile?: VoiceProfile;
18
+ channel?: string;
19
+ }
20
+ export declare class LiveVoice {
21
+ private readonly key;
22
+ private readonly fetcher;
23
+ private readonly now;
24
+ private catalog;
25
+ private catalogUntil;
26
+ private readonly cache;
27
+ private readonly pending;
28
+ private readonly usage;
29
+ private readonly charsPerMinute;
30
+ private readonly requests;
31
+ private readonly grants;
32
+ private readonly activeBy;
33
+ private readonly db?;
34
+ private schema;
35
+ private readonly dailyChars;
36
+ private cleanupAt;
37
+ private readonly hearing;
38
+ private readonly dailyAudioSeconds;
39
+ private readonly userDailyChars;
40
+ private readonly userDailyAudioSeconds;
41
+ constructor(options?: {
42
+ apiKey?: string;
43
+ fetcher?: typeof fetch;
44
+ now?: () => number;
45
+ charsPerMinute?: number;
46
+ dailyChars?: number;
47
+ dailyAudioSeconds?: number;
48
+ userDailyChars?: number;
49
+ userDailyAudioSeconds?: number;
50
+ db?: Queryable;
51
+ });
52
+ available(): boolean;
53
+ /** Optional diarization, billed only while a signed-in listener requests it.
54
+ * Rolling audio is bounded to 15 seconds, including overlap. Every second
55
+ * submitted (also repeated context) consumes the persistent provider budget. */
56
+ hear(bytes: Uint8Array, by: string, signal?: AbortSignal): Promise<SpeakerTranscript>;
57
+ voices(): Promise<LiveVoiceChoice[]>;
58
+ /** A 90-second capability for one channel, never the viewer's account credential. */
59
+ grant(by: string, channel: string): Promise<{
60
+ token: string;
61
+ expires: number;
62
+ }>;
63
+ authorize(token: string, channel: string, chars: number): Promise<string>;
64
+ private ensure;
65
+ private reserve;
66
+ private charge;
67
+ private checkRequest;
68
+ /** Stream the first request immediately; concurrent listeners share its cached result. */
69
+ stream(ask: VoiceRequest, by: string, signal?: AbortSignal): Promise<Response>;
70
+ }