nixamp 0.26.7 → 0.27.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +78 -6
- package/dist/live-interpreter.d.ts +70 -0
- package/dist/live-interpreter.js +243 -0
- package/dist/live-voice.d.ts +5 -2
- package/dist/live-voice.js +35 -6
- package/dist/pcm-wav.d.ts +2 -0
- package/dist/pcm-wav.js +27 -0
- package/dist/server.d.ts +6 -0
- package/dist/server.js +136 -7
- package/dist/shared-translation.d.ts +55 -0
- package/dist/shared-translation.js +335 -0
- package/dist/translation-passes.d.ts +97 -0
- package/dist/translation-passes.js +227 -0
- package/dist/web-sites.d.ts +6 -0
- package/dist/web-sites.js +26 -0
- package/package.json +1 -1
- package/src/live-interpreter.ts +194 -0
- package/src/live-voice.ts +33 -6
- package/src/pcm-wav.ts +26 -0
- package/src/server.ts +96 -6
- package/src/shared-translation.ts +218 -0
- package/src/translation-passes.ts +232 -0
- package/src/web-sites.ts +29 -0
- package/web/dist/assets/{hls-3VKVEQE3-BVogqLei.js → hls-3VKVEQE3-C1ttVP7K.js} +1 -1
- package/web/dist/assets/index-3WZwGw04.js +3 -0
- package/web/dist/assets/{index-biLbPIu_.css → index-BmR7T0zX.css} +1 -1
- package/web/dist/assets/{mpegts-mpVD5UG-.js → mpegts-CVYPgepU.js} +1 -1
- package/web/dist/assets/{mpegts-LO6RVLD6-CsNrXUYy.js → mpegts-LO6RVLD6-Dw2p8qGf.js} +1 -1
- package/web/dist/index.html +20 -2
- package/web/dist/sw.js +6 -6
- package/web/dist/assets/index-DMn-qhqH.js +0 -1
package/README.md
CHANGED
|
@@ -219,6 +219,18 @@ BackToSchool.help is a branded, mobile-first client for NixAmp live events. It
|
|
|
219
219
|
uses the same NixAmp accounts, PostgreSQL data, rooms, invitations, layouts, and
|
|
220
220
|
channel transport as the main app; it has no separate backend or user store.
|
|
221
221
|
|
|
222
|
+
The production Docker image builds both clients and serves the BackToSchool
|
|
223
|
+
client for `backtoschool.help` and `www.backtoschool.help`. Attach both domains
|
|
224
|
+
to the existing NixAmp service and point their DNS at the hosting provider's
|
|
225
|
+
targets. Accounts, event APIs, and live audio stay in that same process. FFmpeg
|
|
226
|
+
is installed in the image so hosts can broadcast from their browser.
|
|
227
|
+
|
|
228
|
+
`NIXAMP_WEB_SITES` maps public origins to built client directories, for example
|
|
229
|
+
`{"https://backtoschool.help":"/app/backtoschool/dist"}`. The configured origin
|
|
230
|
+
also supplies event metadata and invitation links. Other hosts use `--web`.
|
|
231
|
+
An invalid mapping or missing build stops startup rather than serving the wrong
|
|
232
|
+
client. `NIXAMP_SITE` continues to identify the shared NixAmp account service.
|
|
233
|
+
|
|
222
234
|
Build the server and both web clients from the repository root:
|
|
223
235
|
|
|
224
236
|
```
|
|
@@ -545,6 +557,61 @@ uses its direct model; it does not first translate the audio into English.
|
|
|
545
557
|
|
|
546
558
|
### Hear it in your language
|
|
547
559
|
|
|
560
|
+
**Buy translated audio** (`$` in the player or Transcript title bar) offers
|
|
561
|
+
prepaid, account-bound passes: **$5 / 24 hours**, **$25 / 7 days**, or **$100 /
|
|
562
|
+
30 days**. Each purchase provides that many dollars of usage credit, not
|
|
563
|
+
unlimited listening. There is no automatic renewal. Credit expires; buying
|
|
564
|
+
before expiry adds the credit and keeps the later expiry. At 1,000 translated
|
|
565
|
+
characters/minute with normal recognition overlap, the passes provide about
|
|
566
|
+
16, 81, or 327 minutes respectively. Actual speech density changes the allowance.
|
|
567
|
+
|
|
568
|
+
Paid access is **5× base speech API cost (400% markup)**: $0.25 per 1,000 Flash
|
|
569
|
+
characters and $1.10 per submitted Scribe audio hour. Recognition includes
|
|
570
|
+
repeated context, normally three submitted hours per listening hour. The price
|
|
571
|
+
is the same for every listener, including reused audio; reuse reduces provider
|
|
572
|
+
spending. Captions and self-hosted text translation retain their existing free
|
|
573
|
+
access and throttles.
|
|
574
|
+
|
|
575
|
+
CoinPay hosts crypto checkout with the merchant's configured currencies. Network
|
|
576
|
+
fees are shown separately at checkout. Nixamp creates fixed-price orders on the
|
|
577
|
+
server and verifies the stored payment ID, confirmed status, USD currency, and
|
|
578
|
+
exact price before crediting the account. Returning from checkout or sending a
|
|
579
|
+
client-side `paid` flag never unlocks access. Pending purchases can be resumed
|
|
580
|
+
from the panel on another device signed into the same account.
|
|
581
|
+
|
|
582
|
+
PostgreSQL atomically reserves usage credit before paid calls, refunds rejected
|
|
583
|
+
provider requests, and credits a confirmed payment once across concurrent checks.
|
|
584
|
+
Accepted speech is charged even if playback is canceled. Money is stored as
|
|
585
|
+
integer micro-USD. The ledger uses base cost rounded up to a micro-dollar, then
|
|
586
|
+
multiplied by five. Credentials and balances never travel in checkout URLs.
|
|
587
|
+
New checkout creation is capped at five per account and fifty per account server
|
|
588
|
+
per UTC day, plus IP and request throttles; retries reuse the original invoice.
|
|
589
|
+
|
|
590
|
+
Account servers require a paid pass by default. Configure `COINPAY_X402_KEY`
|
|
591
|
+
with `payments:create` permission and at least one business wallet; the scoped
|
|
592
|
+
key supplies the merchant identity. Existing credit still works during a
|
|
593
|
+
checkout outage. A self-hosted operator explicitly sponsoring API usage may set
|
|
594
|
+
`NIXAMP_TRANSLATION_BILLING=off`.
|
|
595
|
+
|
|
596
|
+
Live Nixamp channels share **one recognition, translation, and voice pipeline
|
|
597
|
+
per source and target language** on the account server. Every listening account
|
|
598
|
+
pays the same access rate; joining adds no extra recognition or voice generation.
|
|
599
|
+
The pipeline persists while anyone remains and closes its source and pending
|
|
600
|
+
work when the last listener leaves. Disconnecting one viewer does not stop the
|
|
601
|
+
others. Two connections per account, four active source/language pipelines, and
|
|
602
|
+
1,000 connections per pipeline bound resource use. A slow or unfunded listener
|
|
603
|
+
is disconnected independently. Background sound stays local and independently
|
|
604
|
+
switchable. Public source addresses are resolved and pinned before fetching;
|
|
605
|
+
redirects and ffmpeg network/file fetches are disabled.
|
|
606
|
+
|
|
607
|
+
Live pipeline sharing currently runs within one account-server process (as
|
|
608
|
+
nixamp.com's deployment does). Multiple replicas need stream affinity before
|
|
609
|
+
scaling this path; the payment ledger already works across replicas. Files and
|
|
610
|
+
individually timed browser media retain local capture, because viewers can be
|
|
611
|
+
at different playback positions. Shared live streams use the same speaker voices
|
|
612
|
+
for everyone; individual playback retains voice overrides.
|
|
613
|
+
|
|
614
|
+
|
|
548
615
|
Use **Translate audio** beside the player's language menu to hear whatever
|
|
549
616
|
Nixamp is playing in your language. One click starts translation; it selects
|
|
550
617
|
your preferred supported language if the menu is still on Original. Turn it off
|
|
@@ -574,7 +641,7 @@ recognition. Native captions never translate to English as
|
|
|
574
641
|
an intermediate recognition step.
|
|
575
642
|
|
|
576
643
|
Recognition, text translation, and streaming voice playback run as separate
|
|
577
|
-
stages. Each stage has at most one active request per
|
|
644
|
+
stages. Each stage has at most one active request per shared pipeline or individual playback session. Overlapping
|
|
578
645
|
recognition windows recover unprocessed words; unfinished phrases briefly stay
|
|
579
646
|
in context instead of translating every two-second fragment separately. The
|
|
580
647
|
voice player preserves pending speaker turns and fetches the next phrase with
|
|
@@ -585,8 +652,9 @@ queued speech; errors restore the original audio. This is a delayed live
|
|
|
585
652
|
interpreter, not a promise of exact lip sync or word-by-word streaming captions.
|
|
586
653
|
|
|
587
654
|
OpenStream currently compresses server-to-server relays, not this browser
|
|
588
|
-
translation path.
|
|
589
|
-
|
|
655
|
+
translation path. Individual playback uploads bounded mono 16 kHz WAV clips. Shared live channels
|
|
656
|
+
are decoded on the account server and distribute the generated PCM over one
|
|
657
|
+
authenticated event stream per viewer. Ordinary media playback already uses its audio/video
|
|
590
658
|
codecs. The short-window overlap ratio and audio-second spending limits remain
|
|
591
659
|
unchanged; smaller windows do not increase the steady-state audio submitted.
|
|
592
660
|
|
|
@@ -626,8 +694,8 @@ limits, and cached duplicate voice generation. Native speech and local
|
|
|
626
694
|
translation retain their existing account and queue limits.
|
|
627
695
|
|
|
628
696
|
Postgres stores atomic usage reservations and hashed grants, so the feature's
|
|
629
|
-
budgets survive restarts and are shared between replicas. Provider failures
|
|
630
|
-
|
|
697
|
+
budgets survive restarts and are shared between replicas. Provider failures still consume the abuse budgets conservatively;
|
|
698
|
+
the separate paid balance refunds requests rejected before provider acceptance. The configurable daily limits are:
|
|
631
699
|
|
|
632
700
|
| Setting | Default | Counts |
|
|
633
701
|
| --- | ---: | --- |
|
|
@@ -648,7 +716,11 @@ unlimited fallback provider.
|
|
|
648
716
|
|
|
649
717
|
```
|
|
650
718
|
GET /api/v1/speech/voices authenticated stock voices and supported audio languages
|
|
651
|
-
POST /api/v1/speech/
|
|
719
|
+
POST /api/v1/speech/shared paid {source: liveChannelUrl, language} -> shared captions and PCM events
|
|
720
|
+
GET /api/v1/translation-passes plans, balance and pending purchases
|
|
721
|
+
POST /api/v1/translation-passes/checkout authenticated {plan, coin, requestKey} -> hosted checkout
|
|
722
|
+
GET /api/v1/translation-passes/orders/:id authenticated owner payment verification
|
|
723
|
+
POST /api/v1/speech/speakers paid, bounded mono 16 kHz WAV -> native speaker turns
|
|
652
724
|
POST /api/v1/speech/grant authenticated {channel: playbackScope} -> short-lived grant
|
|
653
725
|
POST /api/v1/speech/synthesize scoped grant + {channel, text, language, voice, profile} -> streaming PCM
|
|
654
726
|
```
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import type { SpeakerTurn } from "./speaker-turns.ts";
|
|
2
|
+
export interface Caption {
|
|
3
|
+
channel: string;
|
|
4
|
+
at: number;
|
|
5
|
+
until: number;
|
|
6
|
+
text: string;
|
|
7
|
+
language?: string;
|
|
8
|
+
original?: string;
|
|
9
|
+
sourceLanguage?: string;
|
|
10
|
+
speaker?: string;
|
|
11
|
+
voiceProfile?: "lower" | "higher" | "unknown";
|
|
12
|
+
}
|
|
13
|
+
export interface VoiceChoice {
|
|
14
|
+
id: string;
|
|
15
|
+
name: string;
|
|
16
|
+
gender: string;
|
|
17
|
+
language: string;
|
|
18
|
+
}
|
|
19
|
+
export interface AudioWindow {
|
|
20
|
+
samples: Float32Array;
|
|
21
|
+
at: number;
|
|
22
|
+
until: number;
|
|
23
|
+
freshAt: number;
|
|
24
|
+
}
|
|
25
|
+
export interface Speaker {
|
|
26
|
+
id: string;
|
|
27
|
+
profile: string;
|
|
28
|
+
voice: string;
|
|
29
|
+
}
|
|
30
|
+
/** Match provider labels across overlapping timestamps. A speaker absent from
|
|
31
|
+
* the rolling context gets a new label; pitch alone never identifies someone. */
|
|
32
|
+
export declare class SpeakerTracker {
|
|
33
|
+
private readonly random;
|
|
34
|
+
private previous;
|
|
35
|
+
private sequence;
|
|
36
|
+
readonly speakers: Map<string, Speaker>;
|
|
37
|
+
constructor(random?: () => number);
|
|
38
|
+
reset(): void;
|
|
39
|
+
reconcile(turns: SpeakerTurn[], at: number, voices: VoiceChoice[]): Map<string, Speaker>;
|
|
40
|
+
}
|
|
41
|
+
/** Listener-local interpretation for the playing media. One request in flight and only the latest pending audio window. */
|
|
42
|
+
export declare class Interpreter {
|
|
43
|
+
private readonly options;
|
|
44
|
+
readonly tracker: SpeakerTracker;
|
|
45
|
+
private generation;
|
|
46
|
+
private pending;
|
|
47
|
+
private committedUntil;
|
|
48
|
+
private running;
|
|
49
|
+
private controller;
|
|
50
|
+
private translating;
|
|
51
|
+
private translationController;
|
|
52
|
+
private translations;
|
|
53
|
+
constructor(options: {
|
|
54
|
+
language: () => string;
|
|
55
|
+
speakers: () => boolean;
|
|
56
|
+
voices: () => VoiceChoice[];
|
|
57
|
+
channel: () => string;
|
|
58
|
+
lines: (lines: Caption[]) => void;
|
|
59
|
+
status: (text: string) => void;
|
|
60
|
+
failed: () => void;
|
|
61
|
+
fetcher?: typeof fetch;
|
|
62
|
+
});
|
|
63
|
+
reset(): void;
|
|
64
|
+
push(window: AudioWindow): void;
|
|
65
|
+
private json;
|
|
66
|
+
private run;
|
|
67
|
+
private translate;
|
|
68
|
+
}
|
|
69
|
+
/** Preserve all words while respecting the speech API's character limit. */
|
|
70
|
+
export declare function splitCaption(line: Caption): Caption[];
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
import { encodeWav } from "./pcm-wav.js";
|
|
2
|
+
/** Match provider labels across overlapping timestamps. A speaker absent from
|
|
3
|
+
* the rolling context gets a new label; pitch alone never identifies someone. */
|
|
4
|
+
export class SpeakerTracker {
|
|
5
|
+
random;
|
|
6
|
+
previous = [];
|
|
7
|
+
sequence = 0;
|
|
8
|
+
speakers = new Map();
|
|
9
|
+
constructor(random = Math.random) {
|
|
10
|
+
this.random = random;
|
|
11
|
+
}
|
|
12
|
+
reset() { this.previous = []; this.sequence = 0; this.speakers.clear(); }
|
|
13
|
+
reconcile(turns, at, voices) {
|
|
14
|
+
const links = new Map();
|
|
15
|
+
const used = new Set();
|
|
16
|
+
const candidates = [];
|
|
17
|
+
const totals = new Map();
|
|
18
|
+
for (const turn of turns)
|
|
19
|
+
for (const old of this.previous) {
|
|
20
|
+
const overlap = Math.min(at + turn.end * 1000, old.end) - Math.max(at + turn.start * 1000, old.start);
|
|
21
|
+
if (overlap > 120) {
|
|
22
|
+
const key = `${turn.speaker}|${old.speaker}`;
|
|
23
|
+
totals.set(key, (totals.get(key) ?? 0) + overlap);
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
for (const [key, overlap] of totals) {
|
|
27
|
+
const [local, global] = key.split("|");
|
|
28
|
+
candidates.push({ local: local, global: global, overlap });
|
|
29
|
+
}
|
|
30
|
+
for (const one of candidates.sort((a, b) => b.overlap - a.overlap)) {
|
|
31
|
+
const speaker = this.speakers.get(one.global);
|
|
32
|
+
if (speaker && !links.has(one.local) && !used.has(one.global)) {
|
|
33
|
+
links.set(one.local, speaker);
|
|
34
|
+
used.add(one.global);
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
for (const turn of turns) {
|
|
38
|
+
let speaker = links.get(turn.speaker);
|
|
39
|
+
if (!speaker) {
|
|
40
|
+
// Assign contrasting stock voices, without guessing a person's
|
|
41
|
+
// gender from pitch or a noisy, short opening phrase.
|
|
42
|
+
const taken = new Set([...this.speakers.values()].map(one => one.voice));
|
|
43
|
+
const available = voices.filter(voice => !taken.has(voice.id));
|
|
44
|
+
const pool = available.length ? available : voices;
|
|
45
|
+
const voice = pool[Math.floor(this.random() * pool.length)];
|
|
46
|
+
speaker = { id: `speaker-${++this.sequence}`, profile: turn.profile, voice: voice?.id ?? "auto" };
|
|
47
|
+
this.speakers.set(speaker.id, speaker);
|
|
48
|
+
links.set(turn.speaker, speaker);
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
this.previous = turns.map(turn => ({ start: at + turn.start * 1000, end: at + turn.end * 1000, speaker: links.get(turn.speaker).id }));
|
|
52
|
+
const present = new Set(this.previous.map(turn => turn.speaker));
|
|
53
|
+
for (const key of this.speakers.keys())
|
|
54
|
+
if (this.speakers.size > 64 && !present.has(key))
|
|
55
|
+
this.speakers.delete(key);
|
|
56
|
+
return links;
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
/** Listener-local interpretation for the playing media. One request in flight and only the latest pending audio window. */
|
|
60
|
+
export class Interpreter {
|
|
61
|
+
options;
|
|
62
|
+
tracker = new SpeakerTracker();
|
|
63
|
+
generation = 0;
|
|
64
|
+
pending = null;
|
|
65
|
+
committedUntil = null;
|
|
66
|
+
running = null;
|
|
67
|
+
controller = null;
|
|
68
|
+
translating = null;
|
|
69
|
+
translationController = null;
|
|
70
|
+
translations = [];
|
|
71
|
+
constructor(options) {
|
|
72
|
+
this.options = options;
|
|
73
|
+
}
|
|
74
|
+
reset() {
|
|
75
|
+
this.generation++;
|
|
76
|
+
this.controller?.abort();
|
|
77
|
+
this.translationController?.abort();
|
|
78
|
+
this.pending = null;
|
|
79
|
+
this.running = null;
|
|
80
|
+
this.translating = null;
|
|
81
|
+
this.translations = [];
|
|
82
|
+
this.committedUntil = null;
|
|
83
|
+
this.tracker.reset();
|
|
84
|
+
}
|
|
85
|
+
push(window) {
|
|
86
|
+
this.committedUntil ??= window.freshAt;
|
|
87
|
+
this.pending = window;
|
|
88
|
+
if (this.running === null)
|
|
89
|
+
void this.run(this.generation);
|
|
90
|
+
}
|
|
91
|
+
async json(path, init, signal) {
|
|
92
|
+
const response = await (this.options.fetcher ?? fetch)(path, { ...init, signal });
|
|
93
|
+
const body = await response.json();
|
|
94
|
+
if (!response.ok)
|
|
95
|
+
throw new Error(body.error || "Audio translation is unavailable.");
|
|
96
|
+
return body;
|
|
97
|
+
}
|
|
98
|
+
async run(generation) {
|
|
99
|
+
this.running = generation;
|
|
100
|
+
try {
|
|
101
|
+
while (this.pending && generation === this.generation) {
|
|
102
|
+
const window = this.pending;
|
|
103
|
+
this.pending = null;
|
|
104
|
+
if (Date.now() - window.until > 12_000)
|
|
105
|
+
continue;
|
|
106
|
+
const controller = new AbortController();
|
|
107
|
+
this.controller = controller;
|
|
108
|
+
const signal = AbortSignal.any([controller.signal, AbortSignal.timeout(12_000)]);
|
|
109
|
+
const speakers = this.options.speakers(), target = this.options.language();
|
|
110
|
+
const heard = await this.json(speakers ? "/api/v1/speech/speakers" : "/api/v1/speech/transcribe?live=1", {
|
|
111
|
+
method: "POST", headers: { "content-type": "audio/wav" }, body: new Uint8Array(encodeWav(speakers ? window.samples : window.samples.slice(-80_000))),
|
|
112
|
+
}, signal);
|
|
113
|
+
if (generation !== this.generation)
|
|
114
|
+
return;
|
|
115
|
+
const language = heard.language;
|
|
116
|
+
let lines = [];
|
|
117
|
+
if (speakers) {
|
|
118
|
+
const links = this.tracker.reconcile(heard.turns ?? [], window.at, this.options.voices());
|
|
119
|
+
// The watermark follows emitted words, not the newest capture
|
|
120
|
+
// interval: an overlapping window can recover audio that arrived
|
|
121
|
+
// while recognition/translation was busy. Keep unfinished phrases
|
|
122
|
+
// in that overlap so short capture intervals do not split every
|
|
123
|
+
// sentence (especially damaging when German changes word order).
|
|
124
|
+
const cutoff = this.committedUntil ?? window.freshAt;
|
|
125
|
+
const turns = heard.turns ?? [];
|
|
126
|
+
for (const [index, turn] of turns.entries()) {
|
|
127
|
+
const speaker = links.get(turn.speaker);
|
|
128
|
+
const words = turn.words.filter(word => window.at + (word.start + word.end) * 500 >= cutoff &&
|
|
129
|
+
(window.at + word.end * 1000 <= window.until - 100 || /[.!?。!?]$/.test(word.text)));
|
|
130
|
+
let phrase = [];
|
|
131
|
+
const emit = () => {
|
|
132
|
+
if (!phrase.length)
|
|
133
|
+
return;
|
|
134
|
+
lines.push({ channel: this.options.channel(), at: window.at + phrase[0].start * 1000,
|
|
135
|
+
until: window.at + phrase.at(-1).end * 1000, text: phrase.map(word => word.text).join(" ").trim(),
|
|
136
|
+
language, speaker: speaker.id, voiceProfile: turn.profile });
|
|
137
|
+
phrase = [];
|
|
138
|
+
};
|
|
139
|
+
for (const word of words) {
|
|
140
|
+
if (phrase.length && word.start - phrase.at(-1).end >= 0.45)
|
|
141
|
+
emit();
|
|
142
|
+
phrase.push(word);
|
|
143
|
+
if (/[.!?。!?]$/.test(word.text) || phrase.map(word => word.text).join(" ").length >= 400)
|
|
144
|
+
emit();
|
|
145
|
+
}
|
|
146
|
+
if (phrase.length && (index < turns.length - 1 ||
|
|
147
|
+
window.until - (window.at + phrase.at(-1).end * 1000) >= 350 ||
|
|
148
|
+
window.until - (window.at + phrase[0].start * 1000) >= 3000))
|
|
149
|
+
emit();
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
else if (heard.text) {
|
|
153
|
+
lines = [{ channel: this.options.channel(), at: window.freshAt, until: window.until, text: heard.text, language }];
|
|
154
|
+
}
|
|
155
|
+
if (!lines.length)
|
|
156
|
+
continue;
|
|
157
|
+
this.committedUntil = Math.max(this.committedUntil ?? window.freshAt, ...lines.map(line => line.until));
|
|
158
|
+
if (this.translations.length >= 12)
|
|
159
|
+
throw new Error("Translation fell behind. Try again after the language model has warmed up.");
|
|
160
|
+
this.translations.push({ lines, language, target, until: window.until });
|
|
161
|
+
// Recognition of the next audio window can proceed while the text
|
|
162
|
+
// model translates this phrase. Voice synthesis is a third stage.
|
|
163
|
+
if (this.translating === null)
|
|
164
|
+
void this.translate(this.generation);
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
catch (error) {
|
|
168
|
+
if (generation === this.generation) {
|
|
169
|
+
this.reset();
|
|
170
|
+
this.options.failed();
|
|
171
|
+
this.options.status(error instanceof Error ? error.message : "Live translation stopped.");
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
finally {
|
|
175
|
+
if (this.running === generation)
|
|
176
|
+
this.running = null;
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
async translate(generation) {
|
|
180
|
+
this.translating = generation;
|
|
181
|
+
try {
|
|
182
|
+
while (this.translations.length && generation === this.generation) {
|
|
183
|
+
const item = this.translations.shift();
|
|
184
|
+
let { lines } = item;
|
|
185
|
+
const { language, target, until } = item;
|
|
186
|
+
if (Date.now() - until > 12_000)
|
|
187
|
+
throw new Error("Translation fell behind. Try again after the language model has warmed up.");
|
|
188
|
+
const controller = new AbortController();
|
|
189
|
+
this.translationController = controller;
|
|
190
|
+
if (target && language !== target) {
|
|
191
|
+
if (!language)
|
|
192
|
+
throw new Error("The audio language could not be detected. Waiting for clearer speech.");
|
|
193
|
+
const translated = await this.json("/api/v1/translate?live=1", {
|
|
194
|
+
method: "POST", headers: { "content-type": "application/json" },
|
|
195
|
+
body: JSON.stringify({ texts: lines.map(line => line.text), from: language, to: target }),
|
|
196
|
+
}, AbortSignal.any([controller.signal, AbortSignal.timeout(12_000)]));
|
|
197
|
+
if (translated.texts.length !== lines.length || translated.texts.some(text => !text.trim()))
|
|
198
|
+
throw new Error("Translation omitted a phrase. Enable translated audio again to retry.");
|
|
199
|
+
lines = lines.map((line, i) => ({ ...line, original: line.text, sourceLanguage: language, language: target, text: translated.texts[i] }));
|
|
200
|
+
}
|
|
201
|
+
if (generation !== this.generation)
|
|
202
|
+
return;
|
|
203
|
+
if (Date.now() - until > 12_000)
|
|
204
|
+
throw new Error("Translation fell behind. Try again after the language model has warmed up.");
|
|
205
|
+
this.options.lines(lines.flatMap(line => splitCaption(line)));
|
|
206
|
+
this.options.status(target ? `${language} → ${target} · live translation` : `Original audio language: ${language || "detecting…"}`);
|
|
207
|
+
}
|
|
208
|
+
}
|
|
209
|
+
catch (error) {
|
|
210
|
+
if (generation === this.generation) {
|
|
211
|
+
this.reset();
|
|
212
|
+
this.options.failed();
|
|
213
|
+
this.options.status(error instanceof Error ? error.message : "Live translation stopped.");
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
finally {
|
|
217
|
+
if (this.translating === generation)
|
|
218
|
+
this.translating = null;
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
/** Preserve all words while respecting the speech API's character limit. */
|
|
223
|
+
export function splitCaption(line) {
|
|
224
|
+
const text = line.text.trim();
|
|
225
|
+
if (!text)
|
|
226
|
+
return [];
|
|
227
|
+
const parts = [];
|
|
228
|
+
let rest = text;
|
|
229
|
+
while (rest.length > 600) {
|
|
230
|
+
const space = rest.lastIndexOf(" ", 600);
|
|
231
|
+
const end = space > 300 ? space : 600;
|
|
232
|
+
parts.push(rest.slice(0, end));
|
|
233
|
+
rest = rest.slice(end).trimStart();
|
|
234
|
+
}
|
|
235
|
+
if (rest)
|
|
236
|
+
parts.push(rest);
|
|
237
|
+
let offset = 0;
|
|
238
|
+
return parts.map(part => {
|
|
239
|
+
const at = line.at + (line.until - line.at) * offset / text.length;
|
|
240
|
+
offset += part.length + 1;
|
|
241
|
+
return { ...line, text: part, at, until: Math.min(line.until, line.at + (line.until - line.at) * offset / text.length) };
|
|
242
|
+
});
|
|
243
|
+
}
|
package/dist/live-voice.d.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { type SpeakerTranscript } from "./speaker-turns.ts";
|
|
2
2
|
import type { VoiceProfile } from "./voice-profile.ts";
|
|
3
|
+
import type { TranslationMeter } from "./translation-passes.ts";
|
|
3
4
|
import type { Queryable } from "./follows.ts";
|
|
4
5
|
export declare const LIVE_VOICE_MODEL = "eleven_flash_v2_5";
|
|
5
6
|
export declare const LIVE_VOICE_RATE = 16000;
|
|
@@ -32,6 +33,7 @@ export declare class LiveVoice {
|
|
|
32
33
|
private readonly grants;
|
|
33
34
|
private readonly activeBy;
|
|
34
35
|
private readonly db?;
|
|
36
|
+
private readonly billing?;
|
|
35
37
|
private schema;
|
|
36
38
|
private readonly dailyChars;
|
|
37
39
|
private cleanupAt;
|
|
@@ -49,12 +51,13 @@ export declare class LiveVoice {
|
|
|
49
51
|
userDailyChars?: number;
|
|
50
52
|
userDailyAudioSeconds?: number;
|
|
51
53
|
db?: Queryable;
|
|
54
|
+
billing?: TranslationMeter;
|
|
52
55
|
});
|
|
53
56
|
available(): boolean;
|
|
54
57
|
/** Optional diarization, billed only while a signed-in listener requests it.
|
|
55
58
|
* Rolling audio is bounded to 15 seconds, including overlap. Every second
|
|
56
59
|
* submitted (also repeated context) consumes the persistent provider budget. */
|
|
57
|
-
hear(bytes: Uint8Array, by: string, signal?: AbortSignal): Promise<SpeakerTranscript>;
|
|
60
|
+
hear(bytes: Uint8Array, by: string, signal?: AbortSignal, meter?: TranslationMeter | undefined): Promise<SpeakerTranscript>;
|
|
58
61
|
voices(): Promise<LiveVoiceChoice[]>;
|
|
59
62
|
/** A 90-second capability for one channel, never the viewer's account credential. */
|
|
60
63
|
grant(by: string, channel: string): Promise<{
|
|
@@ -67,5 +70,5 @@ export declare class LiveVoice {
|
|
|
67
70
|
private charge;
|
|
68
71
|
private checkRequest;
|
|
69
72
|
/** Stream the first request immediately; concurrent listeners share its cached result. */
|
|
70
|
-
stream(ask: VoiceRequest, by: string, signal?: AbortSignal): Promise<Response>;
|
|
73
|
+
stream(ask: VoiceRequest, by: string, signal?: AbortSignal, meter?: TranslationMeter | undefined): Promise<Response>;
|
|
71
74
|
}
|
package/dist/live-voice.js
CHANGED
|
@@ -22,6 +22,7 @@ export class LiveVoice {
|
|
|
22
22
|
grants = new Map();
|
|
23
23
|
activeBy = new Map();
|
|
24
24
|
db;
|
|
25
|
+
billing;
|
|
25
26
|
schema = null;
|
|
26
27
|
dailyChars;
|
|
27
28
|
cleanupAt = 0;
|
|
@@ -40,14 +41,16 @@ export class LiveVoice {
|
|
|
40
41
|
this.userDailyAudioSeconds = budget(options.userDailyAudioSeconds, 43_200);
|
|
41
42
|
this.requests = new Guard(this.now);
|
|
42
43
|
this.db = options.db;
|
|
44
|
+
this.billing = options.billing;
|
|
43
45
|
}
|
|
44
46
|
available() { return this.key !== ""; }
|
|
45
47
|
/** Optional diarization, billed only while a signed-in listener requests it.
|
|
46
48
|
* Rolling audio is bounded to 15 seconds, including overlap. Every second
|
|
47
49
|
* submitted (also repeated context) consumes the persistent provider budget. */
|
|
48
|
-
async hear(bytes, by, signal) {
|
|
50
|
+
async hear(bytes, by, signal, meter = this.billing) {
|
|
49
51
|
if (!this.available())
|
|
50
52
|
throw new SpeechError("speaker voices are unavailable", 503);
|
|
53
|
+
await meter?.require(by);
|
|
51
54
|
const wav = decodeWav(bytes);
|
|
52
55
|
const seconds = wav.samples.length / wav.rate;
|
|
53
56
|
if (wav.rate !== 16000 || wav.channels !== 1 || seconds < 0.2 || seconds > 15.1)
|
|
@@ -61,6 +64,8 @@ export class LiveVoice {
|
|
|
61
64
|
if (this.hearing.has(by) || this.hearing.size >= 4)
|
|
62
65
|
throw new SpeechError("speaker transcription is busy", 429);
|
|
63
66
|
this.hearing.add(by);
|
|
67
|
+
let reservation;
|
|
68
|
+
let accepted = false;
|
|
64
69
|
try {
|
|
65
70
|
const billed = Math.ceil(seconds);
|
|
66
71
|
await this.reserve(`scribe:user:${by}`, billed, 300, 60_000);
|
|
@@ -77,14 +82,23 @@ export class LiveVoice {
|
|
|
77
82
|
form.set("tag_audio_events", "false");
|
|
78
83
|
form.set("timestamps_granularity", "word");
|
|
79
84
|
// No language_code: preserve the source language, including Spanish.
|
|
85
|
+
reservation = await meter?.reserve(by, "transcription", wav.samples.length);
|
|
80
86
|
const answer = await this.fetcher("https://api.elevenlabs.io/v1/speech-to-text", {
|
|
81
87
|
method: "POST", headers: { "xi-api-key": this.key }, body: form,
|
|
82
88
|
signal: signal ? AbortSignal.any([signal, AbortSignal.timeout(10_000)]) : AbortSignal.timeout(10_000),
|
|
83
89
|
});
|
|
84
90
|
if (!answer.ok)
|
|
85
91
|
throw new SpeechError("speaker transcription could not run; check provider quota and permissions", answer.status === 429 ? 429 : 502);
|
|
92
|
+
accepted = true;
|
|
93
|
+
if (reservation)
|
|
94
|
+
await meter.commit(reservation);
|
|
86
95
|
return speakerTurns(await answer.json(), wav);
|
|
87
96
|
}
|
|
97
|
+
catch (error) {
|
|
98
|
+
if (reservation && !accepted)
|
|
99
|
+
await meter.refund(reservation);
|
|
100
|
+
throw error;
|
|
101
|
+
}
|
|
88
102
|
finally {
|
|
89
103
|
this.hearing.delete(by);
|
|
90
104
|
}
|
|
@@ -113,6 +127,7 @@ export class LiveVoice {
|
|
|
113
127
|
throw new SpeechError("choose a playback session or live channel", 400);
|
|
114
128
|
if (!this.requests.check(`grant:${by}`, { allowed: 10, windowMs: 60_000 }).ok)
|
|
115
129
|
throw new SpeechError("too many audio authorization requests", 429);
|
|
130
|
+
await this.billing?.require(by);
|
|
116
131
|
for (const [token, grant] of this.grants)
|
|
117
132
|
if (grant.expires <= this.now())
|
|
118
133
|
this.grants.delete(token);
|
|
@@ -200,8 +215,9 @@ export class LiveVoice {
|
|
|
200
215
|
throw new SpeechError("too many translated audio requests", 429);
|
|
201
216
|
}
|
|
202
217
|
/** Stream the first request immediately; concurrent listeners share its cached result. */
|
|
203
|
-
async stream(ask, by, signal) {
|
|
218
|
+
async stream(ask, by, signal, meter = this.billing) {
|
|
204
219
|
this.checkRequest(by);
|
|
220
|
+
await meter?.require(by);
|
|
205
221
|
const text = typeof ask.text === "string" ? ask.text.trim() : "";
|
|
206
222
|
if (!text || text.length > 600)
|
|
207
223
|
throw new SpeechError("translated audio needs a caption of 1–600 characters", 400);
|
|
@@ -222,11 +238,15 @@ export class LiveVoice {
|
|
|
222
238
|
if (item.until < this.now())
|
|
223
239
|
this.cache.delete(key);
|
|
224
240
|
const cached = this.cache.get(id);
|
|
225
|
-
if (cached)
|
|
226
|
-
return new Response(new Uint8Array(cached.bytes), { headers: HEADERS });
|
|
227
241
|
const pending = this.pending.get(id);
|
|
228
|
-
if (pending)
|
|
229
|
-
|
|
242
|
+
if (cached || pending) {
|
|
243
|
+
const bytes = cached?.bytes ?? await pending;
|
|
244
|
+
signal?.throwIfAborted();
|
|
245
|
+
const paid = await meter?.reserve(by, "voice", text.length);
|
|
246
|
+
if (paid)
|
|
247
|
+
await meter.commit(paid);
|
|
248
|
+
return new Response(new Uint8Array(bytes), { headers: HEADERS });
|
|
249
|
+
}
|
|
230
250
|
if (this.pending.size >= 4)
|
|
231
251
|
throw new SpeechError("translated audio is busy; waiting for the next caption", 429);
|
|
232
252
|
if ((this.activeBy.get(by) ?? 0) >= 2)
|
|
@@ -240,9 +260,12 @@ export class LiveVoice {
|
|
|
240
260
|
const release = () => { this.pending.delete(id); this.activeBy.set(by, Math.max(0, (this.activeBy.get(by) ?? 1) - 1)); if (!this.activeBy.get(by))
|
|
241
261
|
this.activeBy.delete(by); };
|
|
242
262
|
void finished.catch(() => undefined);
|
|
263
|
+
let reservation;
|
|
264
|
+
let accepted = false;
|
|
243
265
|
try {
|
|
244
266
|
await this.charge(by, ask.channel ?? "direct", text.length);
|
|
245
267
|
signal?.throwIfAborted();
|
|
268
|
+
reservation = await meter?.reserve(by, "voice", text.length);
|
|
246
269
|
const response = await this.fetcher(`https://api.elevenlabs.io/v1/text-to-speech/${voice.id}/stream?output_format=pcm_16000`, {
|
|
247
270
|
method: "POST",
|
|
248
271
|
headers: { "xi-api-key": this.key, "content-type": "application/json" },
|
|
@@ -251,6 +274,10 @@ export class LiveVoice {
|
|
|
251
274
|
});
|
|
252
275
|
if (!response.ok || !response.body)
|
|
253
276
|
throw new SpeechError(response.status === 429 ? "ElevenLabs audio quota is temporarily exhausted" : "ElevenLabs could not generate audio; check the server key and quota", response.status === 429 ? 429 : 502);
|
|
277
|
+
// Once the provider accepts, aborting playback cannot refund heard audio.
|
|
278
|
+
accepted = true;
|
|
279
|
+
if (reservation)
|
|
280
|
+
await meter.commit(reservation);
|
|
254
281
|
const [play, keep] = response.body.tee();
|
|
255
282
|
void (async () => {
|
|
256
283
|
const reader = keep.getReader();
|
|
@@ -295,6 +322,8 @@ export class LiveVoice {
|
|
|
295
322
|
catch (error) {
|
|
296
323
|
fail(error);
|
|
297
324
|
release();
|
|
325
|
+
if (reservation && !accepted)
|
|
326
|
+
await meter.refund(reservation);
|
|
298
327
|
throw error;
|
|
299
328
|
}
|
|
300
329
|
}
|
package/dist/pcm-wav.js
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/** 16-bit mono PCM WAV bytes from samples in [-1, 1]. */
|
|
2
|
+
export function encodeWav(samples, rate = 16000) {
|
|
3
|
+
const bytes = new Uint8Array(44 + samples.length * 2);
|
|
4
|
+
const view = new DataView(bytes.buffer);
|
|
5
|
+
const ascii = (at, text) => {
|
|
6
|
+
for (let i = 0; i < text.length; i++)
|
|
7
|
+
bytes[at + i] = text.charCodeAt(i);
|
|
8
|
+
};
|
|
9
|
+
ascii(0, "RIFF");
|
|
10
|
+
view.setUint32(4, 36 + samples.length * 2, true);
|
|
11
|
+
ascii(8, "WAVE");
|
|
12
|
+
ascii(12, "fmt ");
|
|
13
|
+
view.setUint32(16, 16, true);
|
|
14
|
+
view.setUint16(20, 1, true);
|
|
15
|
+
view.setUint16(22, 1, true);
|
|
16
|
+
view.setUint32(24, rate, true);
|
|
17
|
+
view.setUint32(28, rate * 2, true);
|
|
18
|
+
view.setUint16(32, 2, true);
|
|
19
|
+
view.setUint16(34, 16, true);
|
|
20
|
+
ascii(36, "data");
|
|
21
|
+
view.setUint32(40, samples.length * 2, true);
|
|
22
|
+
for (let i = 0; i < samples.length; i++) {
|
|
23
|
+
const clipped = Math.max(-1, Math.min(1, samples[i]));
|
|
24
|
+
view.setInt16(44 + i * 2, clipped < 0 ? clipped * 32768 : clipped * 32767, true);
|
|
25
|
+
}
|
|
26
|
+
return bytes;
|
|
27
|
+
}
|
package/dist/server.d.ts
CHANGED
|
@@ -1,4 +1,6 @@
|
|
|
1
1
|
import { type IncomingMessage, type Server, type ServerResponse } from "node:http";
|
|
2
|
+
import { SharedTranslations } from "./shared-translation.ts";
|
|
3
|
+
import { TranslationPasses } from "./translation-passes.ts";
|
|
2
4
|
import { LiveVoice } from "./live-voice.ts";
|
|
3
5
|
import { Connections } from "./connections.ts";
|
|
4
6
|
import { Broadcaster, type Destination, type EncoderSettings } from "./broadcast.ts";
|
|
@@ -525,6 +527,8 @@ export declare function joinSubject(url: URL, options: Pick<HandlerOptions, "dir
|
|
|
525
527
|
export declare function joinDocument(shell: string, subject: JoinSubject, site: string): string;
|
|
526
528
|
export interface HandlerOptions {
|
|
527
529
|
web: string | null;
|
|
530
|
+
/** Branded clients on this same backend, keyed by an explicitly configured host. */
|
|
531
|
+
webSites?: ReadonlyMap<string, import("./web-sites.ts").WebSite>;
|
|
528
532
|
media: boolean;
|
|
529
533
|
version: string;
|
|
530
534
|
/** The key from the share link, or null to serve to anyone who can connect. */
|
|
@@ -705,6 +709,8 @@ export interface HandlerOptions {
|
|
|
705
709
|
/** Speech to text: a line said out loud, heard here. Needs the optional model. */
|
|
706
710
|
speech?: Speech;
|
|
707
711
|
liveVoice?: LiveVoice;
|
|
712
|
+
translationPasses?: TranslationPasses;
|
|
713
|
+
sharedTranslations?: SharedTranslations;
|
|
708
714
|
/** Translation: texts in another language, by a model here. The same optional library. */
|
|
709
715
|
translator?: Translator;
|
|
710
716
|
/** The transcript store: what was heard, kept under the media's identity. Where the accounts are. */
|