nixamp 0.26.7 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -219,6 +219,18 @@ BackToSchool.help is a branded, mobile-first client for NixAmp live events. It
219
219
  uses the same NixAmp accounts, PostgreSQL data, rooms, invitations, layouts, and
220
220
  channel transport as the main app; it has no separate backend or user store.
221
221
 
222
+ The production Docker image builds both clients and serves the BackToSchool
223
+ client for `backtoschool.help` and `www.backtoschool.help`. Attach both domains
224
+ to the existing NixAmp service and point their DNS at the hosting provider's
225
+ targets. Accounts, event APIs, and live audio stay in that same process. FFmpeg
226
+ is installed in the image so hosts can broadcast from their browser.
227
+
228
+ `NIXAMP_WEB_SITES` maps public origins to built client directories, for example
229
+ `{"https://backtoschool.help":"/app/backtoschool/dist"}`. The configured origin
230
+ also supplies event metadata and invitation links. Other hosts use `--web`.
231
+ An invalid mapping or missing build stops startup rather than serving the wrong
232
+ client. `NIXAMP_SITE` continues to identify the shared NixAmp account service.
233
+
222
234
  Build the server and both web clients from the repository root:
223
235
 
224
236
  ```
@@ -545,6 +557,61 @@ uses its direct model; it does not first translate the audio into English.
545
557
 
546
558
  ### Hear it in your language
547
559
 
560
+ **Buy translated audio** (`$` in the player or Transcript title bar) offers
561
+ prepaid, account-bound passes: **$5 / 24 hours**, **$25 / 7 days**, or **$100 /
562
+ 30 days**. Each purchase provides that many dollars of usage credit, not
563
+ unlimited listening. There is no automatic renewal. Credit expires; buying
564
+ before expiry adds the credit and keeps the later expiry. At 1,000 translated
565
+ characters/minute with normal recognition overlap, the passes provide about
566
+ 16, 81, or 327 minutes respectively. Actual speech density changes the allowance.
567
+
568
+ Paid access is **5× base speech API cost (400% markup)**: $0.25 per 1,000 Flash
569
+ characters and $1.10 per submitted Scribe audio hour. Recognition includes
570
+ repeated context, normally three submitted hours per listening hour. The price
571
+ is the same for every listener, including reused audio; reuse reduces provider
572
+ spending. Captions and self-hosted text translation retain their existing free
573
+ access and throttles.
574
+
575
+ CoinPay hosts crypto checkout with the merchant's configured currencies. Network
576
+ fees are shown separately at checkout. Nixamp creates fixed-price orders on the
577
+ server and verifies the stored payment ID, confirmed status, USD currency, and
578
+ exact price before crediting the account. Returning from checkout or sending a
579
+ client-side `paid` flag never unlocks access. Pending purchases can be resumed
580
+ from the panel on another device signed into the same account.
581
+
582
+ PostgreSQL atomically reserves usage credit before paid calls, refunds rejected
583
+ provider requests, and credits a confirmed payment once across concurrent checks.
584
+ Accepted speech is charged even if playback is canceled. Money is stored as
585
+ integer micro-USD. The ledger uses base cost rounded up to a micro-dollar, then
586
+ multiplied by five. Credentials and balances never travel in checkout URLs.
587
+ New checkout creation is capped at five per account and fifty per account server
588
+ per UTC day, plus IP and request throttles; retries reuse the original invoice.
589
+
590
+ Account servers require a paid pass by default. Configure `COINPAY_X402_KEY`
591
+ with `payments:create` permission and at least one business wallet; the scoped
592
+ key supplies the merchant identity. Existing credit still works during a
593
+ checkout outage. A self-hosted operator explicitly sponsoring API usage may set
594
+ `NIXAMP_TRANSLATION_BILLING=off`.
595
+
596
+ Live Nixamp channels share **one recognition, translation, and voice pipeline
597
+ per source and target language** on the account server. Every listening account
598
+ pays the same access rate; joining adds no extra recognition or voice generation.
599
+ The pipeline persists while anyone remains and closes its source and pending
600
+ work when the last listener leaves. Disconnecting one viewer does not stop the
601
+ others. Two connections per account, four active source/language pipelines, and
602
+ 1,000 connections per pipeline bound resource use. A slow or unfunded listener
603
+ is disconnected independently. Background sound stays local and independently
604
+ switchable. Public source addresses are resolved and pinned before fetching;
605
+ redirects and ffmpeg network/file fetches are disabled.
606
+
607
+ Live pipeline sharing currently runs within one account-server process (as
608
+ nixamp.com's deployment does). Multiple replicas need stream affinity before
609
+ scaling this path; the payment ledger already works across replicas. Files and
610
+ individually timed browser media retain local capture, because viewers can be
611
+ at different playback positions. Shared live streams use the same speaker voices
612
+ for everyone; individual playback retains voice overrides.
613
+
614
+
548
615
  Use **Translate audio** beside the player's language menu to hear whatever
549
616
  Nixamp is playing in your language. One click starts translation; it selects
550
617
  your preferred supported language if the menu is still on Original. Turn it off
@@ -574,7 +641,7 @@ recognition. Native captions never translate to English as
574
641
  an intermediate recognition step.
575
642
 
576
643
  Recognition, text translation, and streaming voice playback run as separate
577
- stages. Each stage has at most one active request per listener. Overlapping
644
+ stages. Each stage has at most one active request per shared pipeline or individual playback session. Overlapping
578
645
  recognition windows recover unprocessed words; unfinished phrases briefly stay
579
646
  in context instead of translating every two-second fragment separately. The
580
647
  voice player preserves pending speaker turns and fetches the next phrase with
@@ -585,8 +652,9 @@ queued speech; errors restore the original audio. This is a delayed live
585
652
  interpreter, not a promise of exact lip sync or word-by-word streaming captions.
586
653
 
587
654
  OpenStream currently compresses server-to-server relays, not this browser
588
- translation path. The browser uploads bounded mono 16 kHz WAV clips and plays
589
- streaming PCM speech. Ordinary media playback already uses its audio/video
655
+ translation path. Individual playback uploads bounded mono 16 kHz WAV clips. Shared live channels
656
+ are decoded on the account server and distribute the generated PCM over one
657
+ authenticated event stream per viewer. Ordinary media playback already uses its audio/video
590
658
  codecs. The short-window overlap ratio and audio-second spending limits remain
591
659
  unchanged; smaller windows do not increase the steady-state audio submitted.
592
660
 
@@ -626,8 +694,8 @@ limits, and cached duplicate voice generation. Native speech and local
626
694
  translation retain their existing account and queue limits.
627
695
 
628
696
  Postgres stores atomic usage reservations and hashed grants, so the feature's
629
- budgets survive restarts and are shared between replicas. Provider failures
630
- still consume reservations conservatively. The configurable daily limits are:
697
+ budgets survive restarts and are shared between replicas. Provider failures still consume the abuse budgets conservatively;
698
+ the separate paid balance refunds requests rejected before provider acceptance. The configurable daily limits are:
631
699
 
632
700
  | Setting | Default | Counts |
633
701
  | --- | ---: | --- |
@@ -648,7 +716,11 @@ unlimited fallback provider.
648
716
 
649
717
  ```
650
718
  GET /api/v1/speech/voices authenticated stock voices and supported audio languages
651
- POST /api/v1/speech/speakers authenticated, bounded mono 16 kHz WAV -> native speaker turns
719
+ POST /api/v1/speech/shared paid {source: liveChannelUrl, language} -> shared captions and PCM events
720
+ GET /api/v1/translation-passes plans, balance and pending purchases
721
+ POST /api/v1/translation-passes/checkout authenticated {plan, coin, requestKey} -> hosted checkout
722
+ GET /api/v1/translation-passes/orders/:id authenticated owner payment verification
723
+ POST /api/v1/speech/speakers paid, bounded mono 16 kHz WAV -> native speaker turns
652
724
  POST /api/v1/speech/grant authenticated {channel: playbackScope} -> short-lived grant
653
725
  POST /api/v1/speech/synthesize scoped grant + {channel, text, language, voice, profile} -> streaming PCM
654
726
  ```
@@ -0,0 +1,70 @@
1
+ import type { SpeakerTurn } from "./speaker-turns.ts";
2
+ export interface Caption {
3
+ channel: string;
4
+ at: number;
5
+ until: number;
6
+ text: string;
7
+ language?: string;
8
+ original?: string;
9
+ sourceLanguage?: string;
10
+ speaker?: string;
11
+ voiceProfile?: "lower" | "higher" | "unknown";
12
+ }
13
+ export interface VoiceChoice {
14
+ id: string;
15
+ name: string;
16
+ gender: string;
17
+ language: string;
18
+ }
19
+ export interface AudioWindow {
20
+ samples: Float32Array;
21
+ at: number;
22
+ until: number;
23
+ freshAt: number;
24
+ }
25
+ export interface Speaker {
26
+ id: string;
27
+ profile: string;
28
+ voice: string;
29
+ }
30
+ /** Match provider labels across overlapping timestamps. A speaker absent from
31
+ * the rolling context gets a new label; pitch alone never identifies someone. */
32
+ export declare class SpeakerTracker {
33
+ private readonly random;
34
+ private previous;
35
+ private sequence;
36
+ readonly speakers: Map<string, Speaker>;
37
+ constructor(random?: () => number);
38
+ reset(): void;
39
+ reconcile(turns: SpeakerTurn[], at: number, voices: VoiceChoice[]): Map<string, Speaker>;
40
+ }
41
+ /** Listener-local interpretation for the playing media. One request in flight and only the latest pending audio window. */
42
+ export declare class Interpreter {
43
+ private readonly options;
44
+ readonly tracker: SpeakerTracker;
45
+ private generation;
46
+ private pending;
47
+ private committedUntil;
48
+ private running;
49
+ private controller;
50
+ private translating;
51
+ private translationController;
52
+ private translations;
53
+ constructor(options: {
54
+ language: () => string;
55
+ speakers: () => boolean;
56
+ voices: () => VoiceChoice[];
57
+ channel: () => string;
58
+ lines: (lines: Caption[]) => void;
59
+ status: (text: string) => void;
60
+ failed: () => void;
61
+ fetcher?: typeof fetch;
62
+ });
63
+ reset(): void;
64
+ push(window: AudioWindow): void;
65
+ private json;
66
+ private run;
67
+ private translate;
68
+ }
69
+ /** Preserve all words while respecting the speech API's character limit. */
70
+ export declare function splitCaption(line: Caption): Caption[];
@@ -0,0 +1,243 @@
1
+ import { encodeWav } from "./pcm-wav.js";
2
+ /** Match provider labels across overlapping timestamps. A speaker absent from
3
+ * the rolling context gets a new label; pitch alone never identifies someone. */
4
+ export class SpeakerTracker {
5
+ random;
6
+ previous = [];
7
+ sequence = 0;
8
+ speakers = new Map();
9
+ constructor(random = Math.random) {
10
+ this.random = random;
11
+ }
12
+ reset() { this.previous = []; this.sequence = 0; this.speakers.clear(); }
13
+ reconcile(turns, at, voices) {
14
+ const links = new Map();
15
+ const used = new Set();
16
+ const candidates = [];
17
+ const totals = new Map();
18
+ for (const turn of turns)
19
+ for (const old of this.previous) {
20
+ const overlap = Math.min(at + turn.end * 1000, old.end) - Math.max(at + turn.start * 1000, old.start);
21
+ if (overlap > 120) {
22
+ const key = `${turn.speaker}|${old.speaker}`;
23
+ totals.set(key, (totals.get(key) ?? 0) + overlap);
24
+ }
25
+ }
26
+ for (const [key, overlap] of totals) {
27
+ const [local, global] = key.split("|");
28
+ candidates.push({ local: local, global: global, overlap });
29
+ }
30
+ for (const one of candidates.sort((a, b) => b.overlap - a.overlap)) {
31
+ const speaker = this.speakers.get(one.global);
32
+ if (speaker && !links.has(one.local) && !used.has(one.global)) {
33
+ links.set(one.local, speaker);
34
+ used.add(one.global);
35
+ }
36
+ }
37
+ for (const turn of turns) {
38
+ let speaker = links.get(turn.speaker);
39
+ if (!speaker) {
40
+ // Assign contrasting stock voices, without guessing a person's
41
+ // gender from pitch or a noisy, short opening phrase.
42
+ const taken = new Set([...this.speakers.values()].map(one => one.voice));
43
+ const available = voices.filter(voice => !taken.has(voice.id));
44
+ const pool = available.length ? available : voices;
45
+ const voice = pool[Math.floor(this.random() * pool.length)];
46
+ speaker = { id: `speaker-${++this.sequence}`, profile: turn.profile, voice: voice?.id ?? "auto" };
47
+ this.speakers.set(speaker.id, speaker);
48
+ links.set(turn.speaker, speaker);
49
+ }
50
+ }
51
+ this.previous = turns.map(turn => ({ start: at + turn.start * 1000, end: at + turn.end * 1000, speaker: links.get(turn.speaker).id }));
52
+ const present = new Set(this.previous.map(turn => turn.speaker));
53
+ for (const key of this.speakers.keys())
54
+ if (this.speakers.size > 64 && !present.has(key))
55
+ this.speakers.delete(key);
56
+ return links;
57
+ }
58
+ }
59
+ /** Listener-local interpretation for the playing media. One request in flight and only the latest pending audio window. */
60
+ export class Interpreter {
61
+ options;
62
+ tracker = new SpeakerTracker();
63
+ generation = 0;
64
+ pending = null;
65
+ committedUntil = null;
66
+ running = null;
67
+ controller = null;
68
+ translating = null;
69
+ translationController = null;
70
+ translations = [];
71
+ constructor(options) {
72
+ this.options = options;
73
+ }
74
+ reset() {
75
+ this.generation++;
76
+ this.controller?.abort();
77
+ this.translationController?.abort();
78
+ this.pending = null;
79
+ this.running = null;
80
+ this.translating = null;
81
+ this.translations = [];
82
+ this.committedUntil = null;
83
+ this.tracker.reset();
84
+ }
85
+ push(window) {
86
+ this.committedUntil ??= window.freshAt;
87
+ this.pending = window;
88
+ if (this.running === null)
89
+ void this.run(this.generation);
90
+ }
91
+ async json(path, init, signal) {
92
+ const response = await (this.options.fetcher ?? fetch)(path, { ...init, signal });
93
+ const body = await response.json();
94
+ if (!response.ok)
95
+ throw new Error(body.error || "Audio translation is unavailable.");
96
+ return body;
97
+ }
98
+ async run(generation) {
99
+ this.running = generation;
100
+ try {
101
+ while (this.pending && generation === this.generation) {
102
+ const window = this.pending;
103
+ this.pending = null;
104
+ if (Date.now() - window.until > 12_000)
105
+ continue;
106
+ const controller = new AbortController();
107
+ this.controller = controller;
108
+ const signal = AbortSignal.any([controller.signal, AbortSignal.timeout(12_000)]);
109
+ const speakers = this.options.speakers(), target = this.options.language();
110
+ const heard = await this.json(speakers ? "/api/v1/speech/speakers" : "/api/v1/speech/transcribe?live=1", {
111
+ method: "POST", headers: { "content-type": "audio/wav" }, body: new Uint8Array(encodeWav(speakers ? window.samples : window.samples.slice(-80_000))),
112
+ }, signal);
113
+ if (generation !== this.generation)
114
+ return;
115
+ const language = heard.language;
116
+ let lines = [];
117
+ if (speakers) {
118
+ const links = this.tracker.reconcile(heard.turns ?? [], window.at, this.options.voices());
119
+ // The watermark follows emitted words, not the newest capture
120
+ // interval: an overlapping window can recover audio that arrived
121
+ // while recognition/translation was busy. Keep unfinished phrases
122
+ // in that overlap so short capture intervals do not split every
123
+ // sentence (especially damaging when German changes word order).
124
+ const cutoff = this.committedUntil ?? window.freshAt;
125
+ const turns = heard.turns ?? [];
126
+ for (const [index, turn] of turns.entries()) {
127
+ const speaker = links.get(turn.speaker);
128
+ const words = turn.words.filter(word => window.at + (word.start + word.end) * 500 >= cutoff &&
129
+ (window.at + word.end * 1000 <= window.until - 100 || /[.!?。!?]$/.test(word.text)));
130
+ let phrase = [];
131
+ const emit = () => {
132
+ if (!phrase.length)
133
+ return;
134
+ lines.push({ channel: this.options.channel(), at: window.at + phrase[0].start * 1000,
135
+ until: window.at + phrase.at(-1).end * 1000, text: phrase.map(word => word.text).join(" ").trim(),
136
+ language, speaker: speaker.id, voiceProfile: turn.profile });
137
+ phrase = [];
138
+ };
139
+ for (const word of words) {
140
+ if (phrase.length && word.start - phrase.at(-1).end >= 0.45)
141
+ emit();
142
+ phrase.push(word);
143
+ if (/[.!?。!?]$/.test(word.text) || phrase.map(word => word.text).join(" ").length >= 400)
144
+ emit();
145
+ }
146
+ if (phrase.length && (index < turns.length - 1 ||
147
+ window.until - (window.at + phrase.at(-1).end * 1000) >= 350 ||
148
+ window.until - (window.at + phrase[0].start * 1000) >= 3000))
149
+ emit();
150
+ }
151
+ }
152
+ else if (heard.text) {
153
+ lines = [{ channel: this.options.channel(), at: window.freshAt, until: window.until, text: heard.text, language }];
154
+ }
155
+ if (!lines.length)
156
+ continue;
157
+ this.committedUntil = Math.max(this.committedUntil ?? window.freshAt, ...lines.map(line => line.until));
158
+ if (this.translations.length >= 12)
159
+ throw new Error("Translation fell behind. Try again after the language model has warmed up.");
160
+ this.translations.push({ lines, language, target, until: window.until });
161
+ // Recognition of the next audio window can proceed while the text
162
+ // model translates this phrase. Voice synthesis is a third stage.
163
+ if (this.translating === null)
164
+ void this.translate(this.generation);
165
+ }
166
+ }
167
+ catch (error) {
168
+ if (generation === this.generation) {
169
+ this.reset();
170
+ this.options.failed();
171
+ this.options.status(error instanceof Error ? error.message : "Live translation stopped.");
172
+ }
173
+ }
174
+ finally {
175
+ if (this.running === generation)
176
+ this.running = null;
177
+ }
178
+ }
179
+ async translate(generation) {
180
+ this.translating = generation;
181
+ try {
182
+ while (this.translations.length && generation === this.generation) {
183
+ const item = this.translations.shift();
184
+ let { lines } = item;
185
+ const { language, target, until } = item;
186
+ if (Date.now() - until > 12_000)
187
+ throw new Error("Translation fell behind. Try again after the language model has warmed up.");
188
+ const controller = new AbortController();
189
+ this.translationController = controller;
190
+ if (target && language !== target) {
191
+ if (!language)
192
+ throw new Error("The audio language could not be detected. Waiting for clearer speech.");
193
+ const translated = await this.json("/api/v1/translate?live=1", {
194
+ method: "POST", headers: { "content-type": "application/json" },
195
+ body: JSON.stringify({ texts: lines.map(line => line.text), from: language, to: target }),
196
+ }, AbortSignal.any([controller.signal, AbortSignal.timeout(12_000)]));
197
+ if (translated.texts.length !== lines.length || translated.texts.some(text => !text.trim()))
198
+ throw new Error("Translation omitted a phrase. Enable translated audio again to retry.");
199
+ lines = lines.map((line, i) => ({ ...line, original: line.text, sourceLanguage: language, language: target, text: translated.texts[i] }));
200
+ }
201
+ if (generation !== this.generation)
202
+ return;
203
+ if (Date.now() - until > 12_000)
204
+ throw new Error("Translation fell behind. Try again after the language model has warmed up.");
205
+ this.options.lines(lines.flatMap(line => splitCaption(line)));
206
+ this.options.status(target ? `${language} → ${target} · live translation` : `Original audio language: ${language || "detecting…"}`);
207
+ }
208
+ }
209
+ catch (error) {
210
+ if (generation === this.generation) {
211
+ this.reset();
212
+ this.options.failed();
213
+ this.options.status(error instanceof Error ? error.message : "Live translation stopped.");
214
+ }
215
+ }
216
+ finally {
217
+ if (this.translating === generation)
218
+ this.translating = null;
219
+ }
220
+ }
221
+ }
222
+ /** Preserve all words while respecting the speech API's character limit. */
223
+ export function splitCaption(line) {
224
+ const text = line.text.trim();
225
+ if (!text)
226
+ return [];
227
+ const parts = [];
228
+ let rest = text;
229
+ while (rest.length > 600) {
230
+ const space = rest.lastIndexOf(" ", 600);
231
+ const end = space > 300 ? space : 600;
232
+ parts.push(rest.slice(0, end));
233
+ rest = rest.slice(end).trimStart();
234
+ }
235
+ if (rest)
236
+ parts.push(rest);
237
+ let offset = 0;
238
+ return parts.map(part => {
239
+ const at = line.at + (line.until - line.at) * offset / text.length;
240
+ offset += part.length + 1;
241
+ return { ...line, text: part, at, until: Math.min(line.until, line.at + (line.until - line.at) * offset / text.length) };
242
+ });
243
+ }
@@ -1,5 +1,6 @@
1
1
  import { type SpeakerTranscript } from "./speaker-turns.ts";
2
2
  import type { VoiceProfile } from "./voice-profile.ts";
3
+ import type { TranslationMeter } from "./translation-passes.ts";
3
4
  import type { Queryable } from "./follows.ts";
4
5
  export declare const LIVE_VOICE_MODEL = "eleven_flash_v2_5";
5
6
  export declare const LIVE_VOICE_RATE = 16000;
@@ -32,6 +33,7 @@ export declare class LiveVoice {
32
33
  private readonly grants;
33
34
  private readonly activeBy;
34
35
  private readonly db?;
36
+ private readonly billing?;
35
37
  private schema;
36
38
  private readonly dailyChars;
37
39
  private cleanupAt;
@@ -49,12 +51,13 @@ export declare class LiveVoice {
49
51
  userDailyChars?: number;
50
52
  userDailyAudioSeconds?: number;
51
53
  db?: Queryable;
54
+ billing?: TranslationMeter;
52
55
  });
53
56
  available(): boolean;
54
57
  /** Optional diarization, billed only while a signed-in listener requests it.
55
58
  * Rolling audio is bounded to 15 seconds, including overlap. Every second
56
59
  * submitted (also repeated context) consumes the persistent provider budget. */
57
- hear(bytes: Uint8Array, by: string, signal?: AbortSignal): Promise<SpeakerTranscript>;
60
+ hear(bytes: Uint8Array, by: string, signal?: AbortSignal, meter?: TranslationMeter | undefined): Promise<SpeakerTranscript>;
58
61
  voices(): Promise<LiveVoiceChoice[]>;
59
62
  /** A 90-second capability for one channel, never the viewer's account credential. */
60
63
  grant(by: string, channel: string): Promise<{
@@ -67,5 +70,5 @@ export declare class LiveVoice {
67
70
  private charge;
68
71
  private checkRequest;
69
72
  /** Stream the first request immediately; concurrent listeners share its cached result. */
70
- stream(ask: VoiceRequest, by: string, signal?: AbortSignal): Promise<Response>;
73
+ stream(ask: VoiceRequest, by: string, signal?: AbortSignal, meter?: TranslationMeter | undefined): Promise<Response>;
71
74
  }
@@ -22,6 +22,7 @@ export class LiveVoice {
22
22
  grants = new Map();
23
23
  activeBy = new Map();
24
24
  db;
25
+ billing;
25
26
  schema = null;
26
27
  dailyChars;
27
28
  cleanupAt = 0;
@@ -40,14 +41,16 @@ export class LiveVoice {
40
41
  this.userDailyAudioSeconds = budget(options.userDailyAudioSeconds, 43_200);
41
42
  this.requests = new Guard(this.now);
42
43
  this.db = options.db;
44
+ this.billing = options.billing;
43
45
  }
44
46
  available() { return this.key !== ""; }
45
47
  /** Optional diarization, billed only while a signed-in listener requests it.
46
48
  * Rolling audio is bounded to 15 seconds, including overlap. Every second
47
49
  * submitted (also repeated context) consumes the persistent provider budget. */
48
- async hear(bytes, by, signal) {
50
+ async hear(bytes, by, signal, meter = this.billing) {
49
51
  if (!this.available())
50
52
  throw new SpeechError("speaker voices are unavailable", 503);
53
+ await meter?.require(by);
51
54
  const wav = decodeWav(bytes);
52
55
  const seconds = wav.samples.length / wav.rate;
53
56
  if (wav.rate !== 16000 || wav.channels !== 1 || seconds < 0.2 || seconds > 15.1)
@@ -61,6 +64,8 @@ export class LiveVoice {
61
64
  if (this.hearing.has(by) || this.hearing.size >= 4)
62
65
  throw new SpeechError("speaker transcription is busy", 429);
63
66
  this.hearing.add(by);
67
+ let reservation;
68
+ let accepted = false;
64
69
  try {
65
70
  const billed = Math.ceil(seconds);
66
71
  await this.reserve(`scribe:user:${by}`, billed, 300, 60_000);
@@ -77,14 +82,23 @@ export class LiveVoice {
77
82
  form.set("tag_audio_events", "false");
78
83
  form.set("timestamps_granularity", "word");
79
84
  // No language_code: preserve the source language, including Spanish.
85
+ reservation = await meter?.reserve(by, "transcription", wav.samples.length);
80
86
  const answer = await this.fetcher("https://api.elevenlabs.io/v1/speech-to-text", {
81
87
  method: "POST", headers: { "xi-api-key": this.key }, body: form,
82
88
  signal: signal ? AbortSignal.any([signal, AbortSignal.timeout(10_000)]) : AbortSignal.timeout(10_000),
83
89
  });
84
90
  if (!answer.ok)
85
91
  throw new SpeechError("speaker transcription could not run; check provider quota and permissions", answer.status === 429 ? 429 : 502);
92
+ accepted = true;
93
+ if (reservation)
94
+ await meter.commit(reservation);
86
95
  return speakerTurns(await answer.json(), wav);
87
96
  }
97
+ catch (error) {
98
+ if (reservation && !accepted)
99
+ await meter.refund(reservation);
100
+ throw error;
101
+ }
88
102
  finally {
89
103
  this.hearing.delete(by);
90
104
  }
@@ -113,6 +127,7 @@ export class LiveVoice {
113
127
  throw new SpeechError("choose a playback session or live channel", 400);
114
128
  if (!this.requests.check(`grant:${by}`, { allowed: 10, windowMs: 60_000 }).ok)
115
129
  throw new SpeechError("too many audio authorization requests", 429);
130
+ await this.billing?.require(by);
116
131
  for (const [token, grant] of this.grants)
117
132
  if (grant.expires <= this.now())
118
133
  this.grants.delete(token);
@@ -200,8 +215,9 @@ export class LiveVoice {
200
215
  throw new SpeechError("too many translated audio requests", 429);
201
216
  }
202
217
  /** Stream the first request immediately; concurrent listeners share its cached result. */
203
- async stream(ask, by, signal) {
218
+ async stream(ask, by, signal, meter = this.billing) {
204
219
  this.checkRequest(by);
220
+ await meter?.require(by);
205
221
  const text = typeof ask.text === "string" ? ask.text.trim() : "";
206
222
  if (!text || text.length > 600)
207
223
  throw new SpeechError("translated audio needs a caption of 1–600 characters", 400);
@@ -222,11 +238,15 @@ export class LiveVoice {
222
238
  if (item.until < this.now())
223
239
  this.cache.delete(key);
224
240
  const cached = this.cache.get(id);
225
- if (cached)
226
- return new Response(new Uint8Array(cached.bytes), { headers: HEADERS });
227
241
  const pending = this.pending.get(id);
228
- if (pending)
229
- return new Response(new Uint8Array(await pending), { headers: HEADERS });
242
+ if (cached || pending) {
243
+ const bytes = cached?.bytes ?? await pending;
244
+ signal?.throwIfAborted();
245
+ const paid = await meter?.reserve(by, "voice", text.length);
246
+ if (paid)
247
+ await meter.commit(paid);
248
+ return new Response(new Uint8Array(bytes), { headers: HEADERS });
249
+ }
230
250
  if (this.pending.size >= 4)
231
251
  throw new SpeechError("translated audio is busy; waiting for the next caption", 429);
232
252
  if ((this.activeBy.get(by) ?? 0) >= 2)
@@ -240,9 +260,12 @@ export class LiveVoice {
240
260
  const release = () => { this.pending.delete(id); this.activeBy.set(by, Math.max(0, (this.activeBy.get(by) ?? 1) - 1)); if (!this.activeBy.get(by))
241
261
  this.activeBy.delete(by); };
242
262
  void finished.catch(() => undefined);
263
+ let reservation;
264
+ let accepted = false;
243
265
  try {
244
266
  await this.charge(by, ask.channel ?? "direct", text.length);
245
267
  signal?.throwIfAborted();
268
+ reservation = await meter?.reserve(by, "voice", text.length);
246
269
  const response = await this.fetcher(`https://api.elevenlabs.io/v1/text-to-speech/${voice.id}/stream?output_format=pcm_16000`, {
247
270
  method: "POST",
248
271
  headers: { "xi-api-key": this.key, "content-type": "application/json" },
@@ -251,6 +274,10 @@ export class LiveVoice {
251
274
  });
252
275
  if (!response.ok || !response.body)
253
276
  throw new SpeechError(response.status === 429 ? "ElevenLabs audio quota is temporarily exhausted" : "ElevenLabs could not generate audio; check the server key and quota", response.status === 429 ? 429 : 502);
277
+ // Once the provider accepts, aborting playback cannot refund heard audio.
278
+ accepted = true;
279
+ if (reservation)
280
+ await meter.commit(reservation);
254
281
  const [play, keep] = response.body.tee();
255
282
  void (async () => {
256
283
  const reader = keep.getReader();
@@ -295,6 +322,8 @@ export class LiveVoice {
295
322
  catch (error) {
296
323
  fail(error);
297
324
  release();
325
+ if (reservation && !accepted)
326
+ await meter.refund(reservation);
298
327
  throw error;
299
328
  }
300
329
  }
@@ -0,0 +1,2 @@
1
+ /** 16-bit mono PCM WAV bytes from samples in [-1, 1]. */
2
+ export declare function encodeWav(samples: Float32Array, rate?: number): Uint8Array;
@@ -0,0 +1,27 @@
1
+ /** 16-bit mono PCM WAV bytes from samples in [-1, 1]. */
2
+ export function encodeWav(samples, rate = 16000) {
3
+ const bytes = new Uint8Array(44 + samples.length * 2);
4
+ const view = new DataView(bytes.buffer);
5
+ const ascii = (at, text) => {
6
+ for (let i = 0; i < text.length; i++)
7
+ bytes[at + i] = text.charCodeAt(i);
8
+ };
9
+ ascii(0, "RIFF");
10
+ view.setUint32(4, 36 + samples.length * 2, true);
11
+ ascii(8, "WAVE");
12
+ ascii(12, "fmt ");
13
+ view.setUint32(16, 16, true);
14
+ view.setUint16(20, 1, true);
15
+ view.setUint16(22, 1, true);
16
+ view.setUint32(24, rate, true);
17
+ view.setUint32(28, rate * 2, true);
18
+ view.setUint16(32, 2, true);
19
+ view.setUint16(34, 16, true);
20
+ ascii(36, "data");
21
+ view.setUint32(40, samples.length * 2, true);
22
+ for (let i = 0; i < samples.length; i++) {
23
+ const clipped = Math.max(-1, Math.min(1, samples[i]));
24
+ view.setInt16(44 + i * 2, clipped < 0 ? clipped * 32768 : clipped * 32767, true);
25
+ }
26
+ return bytes;
27
+ }
package/dist/server.d.ts CHANGED
@@ -1,4 +1,6 @@
1
1
  import { type IncomingMessage, type Server, type ServerResponse } from "node:http";
2
+ import { SharedTranslations } from "./shared-translation.ts";
3
+ import { TranslationPasses } from "./translation-passes.ts";
2
4
  import { LiveVoice } from "./live-voice.ts";
3
5
  import { Connections } from "./connections.ts";
4
6
  import { Broadcaster, type Destination, type EncoderSettings } from "./broadcast.ts";
@@ -525,6 +527,8 @@ export declare function joinSubject(url: URL, options: Pick<HandlerOptions, "dir
525
527
  export declare function joinDocument(shell: string, subject: JoinSubject, site: string): string;
526
528
  export interface HandlerOptions {
527
529
  web: string | null;
530
+ /** Branded clients on this same backend, keyed by an explicitly configured host. */
531
+ webSites?: ReadonlyMap<string, import("./web-sites.ts").WebSite>;
528
532
  media: boolean;
529
533
  version: string;
530
534
  /** The key from the share link, or null to serve to anyone who can connect. */
@@ -705,6 +709,8 @@ export interface HandlerOptions {
705
709
  /** Speech to text: a line said out loud, heard here. Needs the optional model. */
706
710
  speech?: Speech;
707
711
  liveVoice?: LiveVoice;
712
+ translationPasses?: TranslationPasses;
713
+ sharedTranslations?: SharedTranslations;
708
714
  /** Translation: texts in another language, by a model here. The same optional library. */
709
715
  translator?: Translator;
710
716
  /** The transcript store: what was heard, kept under the media's identity. Where the accounts are. */