nixamp 0.25.0 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +108 -7
- package/dist/captions.d.ts +7 -0
- package/dist/captions.js +92 -28
- package/dist/live-voice.d.ts +70 -0
- package/dist/live-voice.js +299 -0
- package/dist/server.d.ts +2 -0
- package/dist/server.js +154 -5
- package/dist/speaker-turns.d.ts +32 -0
- package/dist/speaker-turns.js +30 -0
- package/dist/speech.d.ts +47 -0
- package/dist/speech.js +49 -4
- package/dist/transcript-client.d.ts +2 -2
- package/dist/transcript-client.js +5 -2
- package/dist/transcripts.d.ts +5 -0
- package/dist/transcripts.js +8 -3
- package/dist/translate-jobs.js +13 -7
- package/dist/translate.d.ts +2 -0
- package/dist/translate.js +9 -0
- package/dist/voice-profile.d.ts +7 -0
- package/dist/voice-profile.js +53 -0
- package/dist/warm.js +3 -3
- package/package.json +1 -1
- package/src/captions.ts +86 -25
- package/src/live-voice.ts +255 -0
- package/src/server.ts +104 -5
- package/src/speaker-turns.ts +34 -0
- package/src/speech.ts +46 -7
- package/src/transcript-client.ts +4 -0
- package/src/transcripts.ts +13 -3
- package/src/translate-jobs.ts +13 -7
- package/src/translate.ts +7 -1
- package/src/voice-profile.ts +46 -0
- package/src/warm.ts +3 -3
- package/web/dist/assets/{hls-3VKVEQE3-B4ltbKDh.js → hls-3VKVEQE3-vgax_tk1.js} +1 -1
- package/web/dist/assets/index-BzjrTOLf.js +1 -0
- package/web/dist/assets/index-D2Iy07pG.css +1 -0
- package/web/dist/assets/{mpegts-Byy3EkfT.js → mpegts-DmcUOiHq.js} +1 -1
- package/web/dist/assets/{mpegts-LO6RVLD6-C9qqolrW.js → mpegts-LO6RVLD6-CH3EQi6L.js} +1 -1
- package/web/dist/index.html +42 -22
- package/web/dist/sw.js +6 -6
- package/web/dist/assets/index-d7TvpeFZ.js +0 -1
- package/web/dist/assets/index-oyp61Kly.css +0 -1
package/README.md
CHANGED
|
@@ -421,11 +421,18 @@ nixamp.com instead. `NIXAMP_STT_MODEL` picks another Whisper
|
|
|
421
421
|
takes twice as long), `NIXAMP_STT_CACHE` says where its files are kept, and
|
|
422
422
|
`NIXAMP_STT=off` leaves the ear out of a deployment altogether.
|
|
423
423
|
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
424
|
+
Captions default to **Original (auto-detect)**. Each audio window detects its
|
|
425
|
+
own language and explicitly transcribes it. A language selected in the menu
|
|
426
|
+
only affects translation; neither it nor a cached transcript can force the
|
|
427
|
+
recognizer into English. Short windows have a decoding limit, and silent
|
|
428
|
+
audio and repetitive hallucinations are discarded. Language is stored on
|
|
429
|
+
each line, so an interview can switch languages. Legacy live-caption cache
|
|
430
|
+
entries are heard again instead of replaying their incorrect words.
|
|
431
|
+
|
|
432
|
+
At most four channels are captioned per server, with two recognition requests
|
|
433
|
+
per channel in flight. Live work expires after twelve seconds. Translation
|
|
434
|
+
keeps one active request and the latest pending line per target; joining a
|
|
435
|
+
live reads cached translations without starting a whole-transcript job.
|
|
429
436
|
|
|
430
437
|
### Kept: written down once, for everybody
|
|
431
438
|
|
|
@@ -523,8 +530,89 @@ The page has the same choice beside the Captions switch, remembered per
|
|
|
523
530
|
device; a translated line is marked with its language and shows what was
|
|
524
531
|
heard under the pointer. An agent has `translate_text`. `NIXAMP_MT_WARM`
|
|
525
532
|
names pairs to load at boot (`en-de,en-sv`), `NIXAMP_MT=off` leaves
|
|
526
|
-
translation out
|
|
527
|
-
|
|
533
|
+
translation out. The Docker image includes the ear, German/Swedish pairs
|
|
534
|
+
with English, and Spanish pairs with English and German. Spanish-to-German
|
|
535
|
+
uses its direct model; it does not first translate the audio into English.
|
|
536
|
+
|
|
537
|
+
|
|
538
|
+
### Hear it in your language
|
|
539
|
+
|
|
540
|
+
The browser's Transcript panel works with movies, shows, sports, courses,
|
|
541
|
+
podcasts, live channels, and rooms. Choose a language and enable **Play
|
|
542
|
+
translated audio**. This is a listener preference: the video and the room's
|
|
543
|
+
shared playback clock keep running. Turn it off to restore the original sound.
|
|
544
|
+
**Start player captions** transcribes other playback in its detected source
|
|
545
|
+
language. Files are played locally; these explicit controls opt into sending
|
|
546
|
+
short audio clips for processing. Ordinary playback uploads no audio.
|
|
547
|
+
|
|
548
|
+
**Translate another tab** opens the browser's audio-sharing chooser, so a
|
|
549
|
+
watch party or video hosted on another site can be interpreted too. Select a
|
|
550
|
+
browser tab with **Share audio** enabled; **Stop listening** releases sharing.
|
|
551
|
+
Only audio is sent, even though the browser requires a video track to select
|
|
552
|
+
its source. Sharing support depends on the browser and the source's permissions;
|
|
553
|
+
protected media and sources whose audio cannot be captured remain unavailable.
|
|
554
|
+
The control requests suppression of the source tab's local sound. If a browser
|
|
555
|
+
ignores that option, mute the source tab to avoid hearing both languages.
|
|
556
|
+
|
|
557
|
+
Native captions use local Whisper Base through Transformers.js. Optional
|
|
558
|
+
speaker-aware audio uses ElevenLabs **Scribe v2** for native transcription with
|
|
559
|
+
speaker turns, local **OPUS-MT** for the selected translation, and ElevenLabs
|
|
560
|
+
**Flash v2.5** HTTP streaming for natural stock voices. The application sends
|
|
561
|
+
short audio clips and translated text to ElevenLabs only for this optional
|
|
562
|
+
feature. This uses the direct API; it does not need an MCP server or clone voices.
|
|
563
|
+
Supported translation pairs come from `/api/v1/translate`; voice languages are
|
|
564
|
+
also checked before enabling the audio toggle.
|
|
565
|
+
|
|
566
|
+
A rolling 15-second audio window advances every 5 seconds. Speaker labels are
|
|
567
|
+
reconciled using overlapping timestamps, with different voices assigned to
|
|
568
|
+
separate speakers. Lower/higher pitch suggests a male/female stock voice;
|
|
569
|
+
ambiguous audio uses a default. Each speaker's voice can be changed in the
|
|
570
|
+
panel. Pitch is not gender identity, and a speaker returning after leaving the
|
|
571
|
+
rolling context may receive a new label. Simultaneous speech and noisy crowds
|
|
572
|
+
can still confuse recognition. Native captions never translate to English as
|
|
573
|
+
an intermediate recognition step.
|
|
574
|
+
|
|
575
|
+
Processing has one active request and only the latest pending window per
|
|
576
|
+
listener; speech queues and response sizes are bounded. Old transcript history
|
|
577
|
+
is never spoken. Pause, seek, source changes, and disabling the feature cancel
|
|
578
|
+
queued speech; errors restore the original audio. This is a delayed live
|
|
579
|
+
interpreter, not a promise of exact lip sync or background-music separation.
|
|
580
|
+
|
|
581
|
+
The account server needs `ELEVENLABS_API_KEY`; `NIXAMP_DUBBING=off` disables
|
|
582
|
+
this feature. The key stays on the server. Sign-in is required for speaker
|
|
583
|
+
analysis, voice selection, and short-lived playback grants. Grants expire after
|
|
584
|
+
90 seconds, authorize at most 2,000 characters, and are scoped to one playback
|
|
585
|
+
session or channel. Provider requests also have account/IP throttles, concurrency
|
|
586
|
+
limits, and cached duplicate voice generation. Native speech and local
|
|
587
|
+
translation retain their existing account and queue limits.
|
|
588
|
+
|
|
589
|
+
Postgres stores atomic usage reservations and hashed grants, so the feature's
|
|
590
|
+
budgets survive restarts and are shared between replicas. Provider failures
|
|
591
|
+
still consume reservations conservatively. The configurable daily limits are:
|
|
592
|
+
|
|
593
|
+
| Setting | Default | Counts |
|
|
594
|
+
| --- | ---: | --- |
|
|
595
|
+
| `NIXAMP_DUB_DAILY_CHARS` | 200,000 | New voice characters across this server |
|
|
596
|
+
| `NIXAMP_DUB_USER_DAILY_CHARS` | 120,000 | New voice characters per account |
|
|
597
|
+
| `NIXAMP_DUB_DAILY_AUDIO_SECONDS` | 86,400 | Scribe audio seconds across this server |
|
|
598
|
+
| `NIXAMP_DUB_USER_DAILY_AUDIO_SECONDS` | 43,200 | Scribe audio seconds per account |
|
|
599
|
+
|
|
600
|
+
Audio limits count overlapping context too: a continuous hour of speaker-aware
|
|
601
|
+
listening submits about three hours of Scribe audio. At the [published API
|
|
602
|
+
rates](https://elevenlabs.io/pricing/api) of $0.05 per 1,000 Flash characters and
|
|
603
|
+
$0.22 per Scribe audio hour, a listener producing 1,000 translated characters
|
|
604
|
+
per minute costs about $3.66/hour, before plan minimums or discounts. These default
|
|
605
|
+
server quotas limit this feature to about $15.28/day at those rates; they do not
|
|
606
|
+
cover other applications using the same provider key. Set a daily limit to zero
|
|
607
|
+
to block new use of that resource. Limits return 429 and never trigger an
|
|
608
|
+
unlimited fallback provider.
|
|
609
|
+
|
|
610
|
+
```
|
|
611
|
+
GET /api/v1/speech/voices authenticated stock voices and supported audio languages
|
|
612
|
+
POST /api/v1/speech/speakers authenticated, bounded mono 16 kHz WAV -> native speaker turns
|
|
613
|
+
POST /api/v1/speech/grant authenticated {channel: playbackScope} -> short-lived grant
|
|
614
|
+
POST /api/v1/speech/synthesize scoped grant + {channel, text, language, voice, profile} -> streaming PCM
|
|
615
|
+
```
|
|
528
616
|
|
|
529
617
|
## Several streams at once
|
|
530
618
|
|
|
@@ -908,3 +996,16 @@ make the name honest.
|
|
|
908
996
|
## Licence
|
|
909
997
|
|
|
910
998
|
MIT
|
|
999
|
+
|
|
1000
|
+
### Accessibility and interaction
|
|
1001
|
+
|
|
1002
|
+
Keyboard access, named controls, headings, skip links, visible focus, descriptive
|
|
1003
|
+
slider values, reduced motion, and concise screen-reader status announcements
|
|
1004
|
+
are built into the web player. Incoming transcripts, chat, and playback updates
|
|
1005
|
+
preserve scrolling, focus, caret, and the browsing page. The
|
|
1006
|
+
[UX and accessibility baseline](docs/ux.md) applies to all interface changes.
|
|
1007
|
+
|
|
1008
|
+
When an older broadcaster still provides unversioned English-first captions,
|
|
1009
|
+
a signed-in listener uses native-language recognition of the playing audio
|
|
1010
|
+
instead of that stale transcript cache. Update the broadcaster for shared native
|
|
1011
|
+
captions; translated audio uses the listener's current audio and selected language.
|
package/dist/captions.d.ts
CHANGED
|
@@ -12,6 +12,8 @@ export interface CaptionLine {
|
|
|
12
12
|
language?: string;
|
|
13
13
|
/** What was heard, when this line is a translation of it. */
|
|
14
14
|
original?: string;
|
|
15
|
+
sourceLanguage?: string;
|
|
16
|
+
voiceProfile?: "lower" | "higher" | "unknown";
|
|
15
17
|
}
|
|
16
18
|
/** What turns a channel's bytes into 16 kHz mono 16-bit PCM. ffmpeg, or a test's stand-in. */
|
|
17
19
|
export interface Decoder {
|
|
@@ -71,6 +73,8 @@ export declare const IDLE_MS = 60000;
|
|
|
71
73
|
export declare const QUIET = 0.004;
|
|
72
74
|
/** Windows waiting on the ear at once. Past this the sound is dropped, not queued: late words are worse than none. */
|
|
73
75
|
export declare const IN_FLIGHT = 2;
|
|
76
|
+
export declare const LIVE_DEADLINE_MS = 12000;
|
|
77
|
+
export declare const MAX_CAPTIONERS = 4;
|
|
74
78
|
/** Heard lines wait this long, at most, before they are kept. */
|
|
75
79
|
export declare const FLUSH_MS = 20000;
|
|
76
80
|
/** Or this many. */
|
|
@@ -118,6 +122,9 @@ export declare class Captions {
|
|
|
118
122
|
constructor(options: CaptionsOptions);
|
|
119
123
|
/** Whether this server can caption at all: it has to be signed in for the ear to answer it. */
|
|
120
124
|
available(): boolean;
|
|
125
|
+
capacity(id: string): boolean;
|
|
126
|
+
/** Voice requests are constrained to captions this channel actually produced. */
|
|
127
|
+
voiceRequest(id: string, at: number | null, language: string, voice: string, signal: AbortSignal, grant?: string): Promise<Response>;
|
|
121
128
|
/**
|
|
122
129
|
* Lines for a channel as they are heard, starting the captioner if it is
|
|
123
130
|
* not running. Null when there is no such channel. The returned function
|
package/dist/captions.js
CHANGED
|
@@ -31,7 +31,8 @@
|
|
|
31
31
|
* handed to whoever wanted that language, and kept beside the original.
|
|
32
32
|
*/
|
|
33
33
|
import { spawn } from "node:child_process";
|
|
34
|
-
import { RATE } from "./speech.js";
|
|
34
|
+
import { NATIVE_REVISION, RATE, SpeechError, reliableText } from "./speech.js";
|
|
35
|
+
import { voiceProfile } from "./voice-profile.js";
|
|
35
36
|
import { fetchTranscript, keepLines, keepMedia, translateTexts } from "./transcript-client.js";
|
|
36
37
|
import { covered, lineAt, transcriptIdOf } from "./transcripts.js";
|
|
37
38
|
export const WINDOW_MS = 5000;
|
|
@@ -42,6 +43,8 @@ export const IDLE_MS = 60_000;
|
|
|
42
43
|
export const QUIET = 0.004;
|
|
43
44
|
/** Windows waiting on the ear at once. Past this the sound is dropped, not queued: late words are worse than none. */
|
|
44
45
|
export const IN_FLIGHT = 2;
|
|
46
|
+
export const LIVE_DEADLINE_MS = 12_000;
|
|
47
|
+
export const MAX_CAPTIONERS = 4;
|
|
45
48
|
/** Heard lines wait this long, at most, before they are kept. */
|
|
46
49
|
export const FLUSH_MS = 20_000;
|
|
47
50
|
/** Or this many. */
|
|
@@ -174,6 +177,9 @@ class Captioner {
|
|
|
174
177
|
stopped = false;
|
|
175
178
|
complainedAt = 0;
|
|
176
179
|
windows = 0;
|
|
180
|
+
audioUntil = 0;
|
|
181
|
+
lastEmittedAt = -Infinity;
|
|
182
|
+
requests = new Set();
|
|
177
183
|
/** What the channel is playing, when the store is to be told. */
|
|
178
184
|
media;
|
|
179
185
|
transcriptId;
|
|
@@ -186,6 +192,8 @@ class Captioner {
|
|
|
186
192
|
flush = null;
|
|
187
193
|
/** Translations in order, per language: a slow one must not overtake the next. */
|
|
188
194
|
chains = new Map();
|
|
195
|
+
translating = new Set();
|
|
196
|
+
nextTranslation = new Map();
|
|
189
197
|
constructor(id, options, onStop) {
|
|
190
198
|
this.id = id;
|
|
191
199
|
this.options = options;
|
|
@@ -248,7 +256,7 @@ class Captioner {
|
|
|
248
256
|
const session = this.options.session();
|
|
249
257
|
if (session === null)
|
|
250
258
|
return;
|
|
251
|
-
const got = await fetchTranscript(session, this.transcriptId, language, this.options.fetcher ?? fetch);
|
|
259
|
+
const got = await fetchTranscript(session, this.transcriptId, language, this.options.fetcher ?? fetch, true);
|
|
252
260
|
if (this.stopped)
|
|
253
261
|
return;
|
|
254
262
|
if (!got.ok) {
|
|
@@ -256,11 +264,12 @@ class Captioner {
|
|
|
256
264
|
this.complain(`the store did not answer: ${got.error}`);
|
|
257
265
|
return;
|
|
258
266
|
}
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
267
|
+
// A row's model/language could have been updated while its old bad lines remained.
|
|
268
|
+
// Trust individual lines made with the corrected native pipeline only.
|
|
269
|
+
const usable = got.body.lines.filter((line) => line.revision === NATIVE_REVISION && reliableText(line.text, line.end - line.start));
|
|
270
|
+
this.known.set(language, usable);
|
|
271
|
+
if (usable.length > 0) {
|
|
272
|
+
this.options.onEvent?.(`captions for "${this.id}": the store knows ${usable.length} lines of this${language ? ` in ${language}` : ""}`);
|
|
264
273
|
}
|
|
265
274
|
}
|
|
266
275
|
onPcm(pcm) {
|
|
@@ -275,7 +284,9 @@ class Captioner {
|
|
|
275
284
|
const rest = all.subarray(size);
|
|
276
285
|
this.pending = rest.length > 0 ? [Buffer.from(rest)] : [];
|
|
277
286
|
this.pendingBytes = rest.length;
|
|
278
|
-
const
|
|
287
|
+
const duration = this.options.windowMs ?? WINDOW_MS;
|
|
288
|
+
const until = Math.max(this.audioUntil + duration, this.now() - rest.length / (RATE * 2) * 1000);
|
|
289
|
+
this.audioUntil = until;
|
|
279
290
|
const index = this.windows;
|
|
280
291
|
this.windows += 1;
|
|
281
292
|
void this.hear(Buffer.from(window), until - (this.options.windowMs ?? WINDOW_MS), until, index);
|
|
@@ -294,7 +305,8 @@ class Captioner {
|
|
|
294
305
|
continue;
|
|
295
306
|
this.readOut.add(line.start);
|
|
296
307
|
const lineAt = at + (line.start - span.start) * 1000;
|
|
297
|
-
this.emit({ channel: this.id, at: lineAt, until: lineAt + (line.end - line.start) * 1000, text: line.text,
|
|
308
|
+
this.emit({ channel: this.id, at: lineAt, until: lineAt + (line.end - line.start) * 1000, text: line.text,
|
|
309
|
+
...(line.language ? { language: line.language } : {}), ...(line.voiceProfile ? { voiceProfile: line.voiceProfile } : {}) }, line.start);
|
|
298
310
|
}
|
|
299
311
|
return;
|
|
300
312
|
}
|
|
@@ -307,43 +319,52 @@ class Captioner {
|
|
|
307
319
|
return;
|
|
308
320
|
}
|
|
309
321
|
this.inFlight += 1;
|
|
322
|
+
const controller = new AbortController();
|
|
323
|
+
this.requests.add(controller);
|
|
324
|
+
const timeout = setTimeout(() => controller.abort(), LIVE_DEADLINE_MS);
|
|
310
325
|
try {
|
|
311
326
|
const wav = wavAround(pcm);
|
|
312
327
|
const url = new URL(`${session.site.replace(/\/+$/, "")}/api/v1/speech/transcribe`);
|
|
313
|
-
|
|
314
|
-
|
|
328
|
+
// Let each audio window detect its own language. Cached text and a viewer's
|
|
329
|
+
// translation selection must never constrain the recognizer.
|
|
330
|
+
url.searchParams.set("live", "1");
|
|
315
331
|
const response = await (this.options.fetcher ?? fetch)(url.toString(), {
|
|
316
332
|
method: "POST",
|
|
317
333
|
headers: { authorization: `Bearer ${session.token}`, "content-type": "audio/wav" },
|
|
318
334
|
body: new Blob([wav.buffer.slice(wav.byteOffset, wav.byteOffset + wav.byteLength)]),
|
|
335
|
+
signal: controller.signal,
|
|
319
336
|
});
|
|
320
337
|
const body = (await response.json().catch(() => ({})));
|
|
321
338
|
if (!response.ok) {
|
|
322
339
|
this.complain(body.error ?? `nixamp.com answered ${response.status}`);
|
|
323
340
|
return;
|
|
324
341
|
}
|
|
325
|
-
const text = (body.text ?? ""
|
|
326
|
-
if (text === "" || this.stopped)
|
|
342
|
+
const text = reliableText(body.text ?? "", (until - at) / 1000);
|
|
343
|
+
if (text === "" || this.stopped || controller.signal.aborted || this.now() - until > LIVE_DEADLINE_MS)
|
|
327
344
|
return;
|
|
328
345
|
this.error = "";
|
|
329
|
-
if (body.language && this.language === "")
|
|
330
|
-
this.language = body.language;
|
|
331
346
|
if (body.model)
|
|
332
347
|
this.model = body.model;
|
|
333
|
-
const line = { channel: this.id, at, until, text, ...(
|
|
348
|
+
const line = { channel: this.id, at, until, text, ...(body.language ? { language: body.language } : {}), voiceProfile: voiceProfile(pcm) };
|
|
334
349
|
this.emit(line, span?.start ?? null);
|
|
335
350
|
if (span)
|
|
336
|
-
this.keep("", { start: span.start, end: span.end, text });
|
|
351
|
+
this.keep("", { start: span.start, end: span.end, text, language: line.language, voiceProfile: line.voiceProfile, revision: NATIVE_REVISION });
|
|
337
352
|
}
|
|
338
353
|
catch (error) {
|
|
339
354
|
this.complain(`could not reach the ear: ${error.message}`);
|
|
340
355
|
}
|
|
341
356
|
finally {
|
|
357
|
+
clearTimeout(timeout);
|
|
358
|
+
this.requests.delete(controller);
|
|
342
359
|
this.inFlight -= 1;
|
|
343
360
|
}
|
|
344
361
|
}
|
|
345
362
|
/** A line as heard, to whoever wants the original, and translated to whoever wants another language. */
|
|
346
363
|
emit(line, mediaStart) {
|
|
364
|
+
if (line.at <= this.lastEmittedAt)
|
|
365
|
+
return;
|
|
366
|
+
this.lastEmittedAt = line.at;
|
|
367
|
+
this.language = line.language ?? "";
|
|
347
368
|
this.lines.push(line);
|
|
348
369
|
while (this.lines.length > KEEP)
|
|
349
370
|
this.lines.shift();
|
|
@@ -372,32 +393,41 @@ class Captioner {
|
|
|
372
393
|
}
|
|
373
394
|
/** The line in another language: from the store when it has been through this moment, from nixamp.com otherwise. */
|
|
374
395
|
translated(language, line, mediaStart) {
|
|
375
|
-
|
|
376
|
-
|
|
396
|
+
// One active request and the latest pending line per target, never a promise
|
|
397
|
+
// chain containing minutes of stale commentary.
|
|
398
|
+
if (this.translating.has(language)) {
|
|
399
|
+
this.nextTranslation.set(language, { line, mediaStart });
|
|
400
|
+
return;
|
|
401
|
+
}
|
|
402
|
+
this.translating.add(language);
|
|
403
|
+
const chain = Promise.resolve().then(async () => {
|
|
404
|
+
if (this.stopped || !this.wanted().has(language) || this.now() - line.until > LIVE_DEADLINE_MS)
|
|
377
405
|
return;
|
|
378
406
|
let text = "";
|
|
379
407
|
const stored = mediaStart === null ? null : lineAt(this.known.get(language) ?? [], mediaStart);
|
|
380
|
-
if (stored) {
|
|
408
|
+
if (stored && stored.original === line.text) {
|
|
381
409
|
text = stored.text;
|
|
382
410
|
}
|
|
383
411
|
else {
|
|
384
412
|
const session = this.options.session();
|
|
385
413
|
if (session === null)
|
|
386
414
|
return;
|
|
387
|
-
|
|
415
|
+
if (!line.language)
|
|
416
|
+
return;
|
|
417
|
+
const got = await translateTexts(session, [line.text], line.language, language, this.options.fetcher ?? fetch, AbortSignal.timeout(LIVE_DEADLINE_MS));
|
|
388
418
|
if (!got.ok) {
|
|
389
419
|
this.complain(`could not translate to ${language}: ${got.error}`);
|
|
390
420
|
return;
|
|
391
421
|
}
|
|
392
|
-
text = (got.body.texts[0] ?? ""
|
|
422
|
+
text = reliableText(got.body.texts[0] ?? "", (line.until - line.at) / 1000);
|
|
393
423
|
if (text === "")
|
|
394
424
|
return;
|
|
395
425
|
if (mediaStart !== null)
|
|
396
|
-
this.keep(language, { start: mediaStart, end: round(mediaStart + (line.until - line.at) / 1000), text });
|
|
426
|
+
this.keep(language, { start: mediaStart, end: round(mediaStart + (line.until - line.at) / 1000), text, original: line.text, revision: NATIVE_REVISION });
|
|
397
427
|
}
|
|
398
|
-
if (this.stopped)
|
|
428
|
+
if (this.stopped || !this.wanted().has(language) || this.now() - line.until > LIVE_DEADLINE_MS)
|
|
399
429
|
return;
|
|
400
|
-
const said = { ...line, text, language, original: line.text };
|
|
430
|
+
const said = { ...line, text, language, sourceLanguage: line.language, original: line.text };
|
|
401
431
|
const lines = this.linesBy.get(language) ?? [];
|
|
402
432
|
lines.push(said);
|
|
403
433
|
while (lines.length > KEEP)
|
|
@@ -407,7 +437,13 @@ class Captioner {
|
|
|
407
437
|
if (wanted === language)
|
|
408
438
|
this.tell(subscriber, said);
|
|
409
439
|
});
|
|
410
|
-
this.chains.set(language, chain.catch(() => undefined))
|
|
440
|
+
this.chains.set(language, chain.catch(() => undefined).finally(() => {
|
|
441
|
+
this.translating.delete(language);
|
|
442
|
+
const next = this.nextTranslation.get(language);
|
|
443
|
+
this.nextTranslation.delete(language);
|
|
444
|
+
if (next && !this.stopped)
|
|
445
|
+
this.translated(language, next.line, next.mediaStart);
|
|
446
|
+
}));
|
|
411
447
|
}
|
|
412
448
|
/** A line for the store, kept with the others of its language until the next flush. */
|
|
413
449
|
keep(language, line) {
|
|
@@ -441,7 +477,7 @@ class Captioner {
|
|
|
441
477
|
const got = await keepLines(session, this.transcriptId, {
|
|
442
478
|
media: this.media.media,
|
|
443
479
|
title: this.media.title,
|
|
444
|
-
language
|
|
480
|
+
language,
|
|
445
481
|
...(language === "" ? {} : { translatedFrom: this.language }),
|
|
446
482
|
...(this.model ? { model: this.model } : {}),
|
|
447
483
|
lines,
|
|
@@ -478,7 +514,7 @@ class Captioner {
|
|
|
478
514
|
};
|
|
479
515
|
}
|
|
480
516
|
recent(after, language = "") {
|
|
481
|
-
const lines = language === ""
|
|
517
|
+
const lines = language === "" ? this.lines : [...this.lines.filter((line) => line.language === language), ...(this.linesBy.get(language) ?? [])].sort((a, b) => a.at - b.at);
|
|
482
518
|
return after > 0 ? lines.filter((line) => line.at > after) : [...lines];
|
|
483
519
|
}
|
|
484
520
|
status() {
|
|
@@ -496,6 +532,10 @@ class Captioner {
|
|
|
496
532
|
if (this.stopped)
|
|
497
533
|
return;
|
|
498
534
|
this.stopped = true;
|
|
535
|
+
for (const request of this.requests)
|
|
536
|
+
request.abort();
|
|
537
|
+
this.requests.clear();
|
|
538
|
+
this.nextTranslation.clear();
|
|
499
539
|
if (this.idle)
|
|
500
540
|
clearTimeout(this.idle);
|
|
501
541
|
this.idle = null;
|
|
@@ -520,6 +560,28 @@ export class Captions {
|
|
|
520
560
|
available() {
|
|
521
561
|
return this.options.session() !== null;
|
|
522
562
|
}
|
|
563
|
+
capacity(id) { return this.running.has(id) || this.running.size < MAX_CAPTIONERS; }
|
|
564
|
+
/** Voice requests are constrained to captions this channel actually produced. */
|
|
565
|
+
async voiceRequest(id, at, language, voice, signal, grant = "") {
|
|
566
|
+
const session = this.options.session();
|
|
567
|
+
if (!session)
|
|
568
|
+
throw new SpeechError("this server must sign in to use translated audio", 503);
|
|
569
|
+
if (at === null)
|
|
570
|
+
return (this.options.fetcher ?? fetch)(`${session.site.replace(/\/+$/, "")}/api/v1/speech/voices`, {
|
|
571
|
+
headers: { authorization: `Bearer ${session.token}` }, signal,
|
|
572
|
+
});
|
|
573
|
+
if (!language)
|
|
574
|
+
throw new SpeechError("choose a translation language first", 400);
|
|
575
|
+
if (!/^nxd_[A-Za-z0-9_-]{43}$/.test(grant))
|
|
576
|
+
throw new SpeechError("sign in to enable translated audio", 401);
|
|
577
|
+
const line = this.recent(id, 0, language).find(line => line.at === at);
|
|
578
|
+
if (!line || (this.options.now ?? Date.now)() - line.until > 30_000)
|
|
579
|
+
throw new SpeechError("that live caption is no longer available for audio", 404);
|
|
580
|
+
return (this.options.fetcher ?? fetch)(`${session.site.replace(/\/+$/, "")}/api/v1/speech/synthesize`, {
|
|
581
|
+
method: "POST", headers: { authorization: `Bearer ${grant}`, "content-type": "application/json" },
|
|
582
|
+
body: JSON.stringify({ text: line.text, language, voice, profile: line.voiceProfile, channel: id }), signal,
|
|
583
|
+
});
|
|
584
|
+
}
|
|
523
585
|
/**
|
|
524
586
|
* Lines for a channel as they are heard, starting the captioner if it is
|
|
525
587
|
* not running. Null when there is no such channel. The returned function
|
|
@@ -528,6 +590,8 @@ export class Captions {
|
|
|
528
590
|
* "" is the original.
|
|
529
591
|
*/
|
|
530
592
|
subscribe(id, subscriber, language = "") {
|
|
593
|
+
if (!this.capacity(id))
|
|
594
|
+
return null;
|
|
531
595
|
let captioner = this.running.get(id);
|
|
532
596
|
if (!captioner) {
|
|
533
597
|
const made = new Captioner(id, this.options, () => {
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import { type SpeakerTranscript } from "./speaker-turns.ts";
|
|
2
|
+
import type { VoiceProfile } from "./voice-profile.ts";
|
|
3
|
+
import type { Queryable } from "./follows.ts";
|
|
4
|
+
export declare const LIVE_VOICE_MODEL = "eleven_flash_v2_5";
|
|
5
|
+
export declare const LIVE_VOICE_RATE = 16000;
|
|
6
|
+
export declare const LIVE_VOICE_LANGUAGES: Set<string>;
|
|
7
|
+
export interface LiveVoiceChoice {
|
|
8
|
+
id: string;
|
|
9
|
+
name: string;
|
|
10
|
+
gender: string;
|
|
11
|
+
language: string;
|
|
12
|
+
}
|
|
13
|
+
export interface VoiceRequest {
|
|
14
|
+
text: string;
|
|
15
|
+
language: string;
|
|
16
|
+
voice?: string;
|
|
17
|
+
profile?: VoiceProfile;
|
|
18
|
+
channel?: string;
|
|
19
|
+
}
|
|
20
|
+
export declare class LiveVoice {
|
|
21
|
+
private readonly key;
|
|
22
|
+
private readonly fetcher;
|
|
23
|
+
private readonly now;
|
|
24
|
+
private catalog;
|
|
25
|
+
private catalogUntil;
|
|
26
|
+
private readonly cache;
|
|
27
|
+
private readonly pending;
|
|
28
|
+
private readonly usage;
|
|
29
|
+
private readonly charsPerMinute;
|
|
30
|
+
private readonly requests;
|
|
31
|
+
private readonly grants;
|
|
32
|
+
private readonly activeBy;
|
|
33
|
+
private readonly db?;
|
|
34
|
+
private schema;
|
|
35
|
+
private readonly dailyChars;
|
|
36
|
+
private cleanupAt;
|
|
37
|
+
private readonly hearing;
|
|
38
|
+
private readonly dailyAudioSeconds;
|
|
39
|
+
private readonly userDailyChars;
|
|
40
|
+
private readonly userDailyAudioSeconds;
|
|
41
|
+
constructor(options?: {
|
|
42
|
+
apiKey?: string;
|
|
43
|
+
fetcher?: typeof fetch;
|
|
44
|
+
now?: () => number;
|
|
45
|
+
charsPerMinute?: number;
|
|
46
|
+
dailyChars?: number;
|
|
47
|
+
dailyAudioSeconds?: number;
|
|
48
|
+
userDailyChars?: number;
|
|
49
|
+
userDailyAudioSeconds?: number;
|
|
50
|
+
db?: Queryable;
|
|
51
|
+
});
|
|
52
|
+
available(): boolean;
|
|
53
|
+
/** Optional diarization, billed only while a signed-in listener requests it.
|
|
54
|
+
* Rolling audio is bounded to 15 seconds, including overlap. Every second
|
|
55
|
+
* submitted (also repeated context) consumes the persistent provider budget. */
|
|
56
|
+
hear(bytes: Uint8Array, by: string, signal?: AbortSignal): Promise<SpeakerTranscript>;
|
|
57
|
+
voices(): Promise<LiveVoiceChoice[]>;
|
|
58
|
+
/** A 90-second capability for one channel, never the viewer's account credential. */
|
|
59
|
+
grant(by: string, channel: string): Promise<{
|
|
60
|
+
token: string;
|
|
61
|
+
expires: number;
|
|
62
|
+
}>;
|
|
63
|
+
authorize(token: string, channel: string, chars: number): Promise<string>;
|
|
64
|
+
private ensure;
|
|
65
|
+
private reserve;
|
|
66
|
+
private charge;
|
|
67
|
+
private checkRequest;
|
|
68
|
+
/** Stream the first request immediately; concurrent listeners share its cached result. */
|
|
69
|
+
stream(ask: VoiceRequest, by: string, signal?: AbortSignal): Promise<Response>;
|
|
70
|
+
}
|