nixamp 0.26.1 → 0.26.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -22
- package/dist/live-voice.d.ts +1 -0
- package/dist/live-voice.js +4 -2
- package/package.json +1 -1
- package/src/live-voice.ts +5 -3
- package/web/dist/assets/background.worker-D9ejvk_o.js +547 -0
- package/web/dist/assets/{hls-3VKVEQE3-DYwZJLyI.js → hls-3VKVEQE3-BJCEkhkX.js} +1 -1
- package/web/dist/assets/index-BZ57qkpe.js +1 -0
- package/web/dist/assets/{index-D2Iy07pG.css → index-DJEWeNkX.css} +1 -1
- package/web/dist/assets/{mpegts-B41k3KPM.js → mpegts-DaLDFdZV.js} +1 -1
- package/web/dist/assets/{mpegts-LO6RVLD6-CPGhrqiW.js → mpegts-LO6RVLD6-DLNvQAAS.js} +1 -1
- package/web/dist/index.html +15 -18
- package/web/dist/licenses/fastenhancer-web.txt +21 -0
- package/web/dist/sw.js +8 -6
- package/web/dist/assets/index-C77VMysw.js +0 -1
package/README.md
CHANGED
|
@@ -537,22 +537,13 @@ uses its direct model; it does not first translate the audio into English.
|
|
|
537
537
|
|
|
538
538
|
### Hear it in your language
|
|
539
539
|
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
**Translate another tab** opens the browser's audio-sharing chooser, so a
|
|
549
|
-
watch party or video hosted on another site can be interpreted too. Select a
|
|
550
|
-
browser tab with **Share audio** enabled; **Stop listening** releases sharing.
|
|
551
|
-
Only audio is sent, even though the browser requires a video track to select
|
|
552
|
-
its source. Sharing support depends on the browser and the source's permissions;
|
|
553
|
-
protected media and sources whose audio cannot be captured remain unavailable.
|
|
554
|
-
The control requests suppression of the source tab's local sound. If a browser
|
|
555
|
-
ignores that option, mute the source tab to avoid hearing both languages.
|
|
540
|
+
Use **Translate audio** beside the player's language menu to hear whatever
|
|
541
|
+
Nixamp is playing in your language. One click starts translation; it selects
|
|
542
|
+
your preferred supported language if the menu is still on Original. Turn it off
|
|
543
|
+
to restore the original audio. The video and the room's shared playback clock
|
|
544
|
+
keep running. **Captions** in the Transcript panel enables text-only recognition.
|
|
545
|
+
Ordinary file playback uploads no audio; enabling captions or translated audio
|
|
546
|
+
opts into processing short clips from the playing media.
|
|
556
547
|
|
|
557
548
|
Native captions use local Whisper Base through Transformers.js. Optional
|
|
558
549
|
speaker-aware audio uses ElevenLabs **Scribe v2** for native transcription with
|
|
@@ -565,18 +556,31 @@ also checked before enabling the audio toggle.
|
|
|
565
556
|
|
|
566
557
|
A rolling 15-second audio window advances every 5 seconds. Speaker labels are
|
|
567
558
|
reconciled using overlapping timestamps, with different voices assigned to
|
|
568
|
-
separate speakers.
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
559
|
+
separate speakers. Voices are picked from the available stock catalogue without
|
|
560
|
+
inferring a person's gender from pitch; each detected speaker gets an unused
|
|
561
|
+
voice until the catalogue is exhausted. **Audio options** folds away optional
|
|
562
|
+
individual overrides. A speaker returning after leaving the rolling context may
|
|
563
|
+
receive a new label. Simultaneous speech and noisy crowds can still confuse
|
|
564
|
+
recognition. Native captions never translate to English as
|
|
573
565
|
an intermediate recognition step.
|
|
574
566
|
|
|
575
567
|
Processing has one active request and only the latest pending window per
|
|
576
568
|
listener; speech queues and response sizes are bounded. Old transcript history
|
|
577
569
|
is never spoken. Pause, seek, source changes, and disabling the feature cancel
|
|
578
570
|
queued speech; errors restore the original audio. This is a delayed live
|
|
579
|
-
interpreter, not a promise of exact lip sync
|
|
571
|
+
interpreter, not a promise of exact lip sync.
|
|
572
|
+
|
|
573
|
+
Translated playback keeps an approximate version of the original background
|
|
574
|
+
sound. FastEnhancer Web's Tiny model estimates speech locally in a dedicated
|
|
575
|
+
browser worker; the player subtracts that estimate from the aligned source in
|
|
576
|
+
each stereo channel and mixes the remainder with translated voices. Background
|
|
577
|
+
processing adds no API calls or provider charges. It stops with translation;
|
|
578
|
+
recognition starts without waiting for it. If the device cannot keep up or load the model, that
|
|
579
|
+
background branch is silenced while translated speech continues. Separation can
|
|
580
|
+
leave some original speech or remove parts of music and crowd noise; disable
|
|
581
|
+
**Keep background sound** under **Audio options** when needed. This uses
|
|
582
|
+
[FastEnhancer Web](https://github.com/ryyr-ry/fastenhancer-web), under the MIT
|
|
583
|
+
license.
|
|
580
584
|
|
|
581
585
|
The account server needs `ELEVENLABS_API_KEY`; `NIXAMP_DUBBING=off` disables
|
|
582
586
|
this feature. The key stays on the server. Sign-in is required for speaker
|
package/dist/live-voice.d.ts
CHANGED
package/dist/live-voice.js
CHANGED
|
@@ -209,10 +209,12 @@ export class LiveVoice {
|
|
|
209
209
|
throw new SpeechError("this language is not supported by Flash voices", 400);
|
|
210
210
|
const voices = await this.voices();
|
|
211
211
|
signal?.throwIfAborted();
|
|
212
|
-
|
|
212
|
+
// The browser chooses unique voices for speakers. Legacy callers get a
|
|
213
|
+
// stable stock voice; pitch is not used to infer gender.
|
|
214
|
+
const seed = createHash("sha256").update(`${ask.channel ?? ""}|${ask.speaker ?? ""}`).digest().readUInt32BE(0);
|
|
213
215
|
const voice = ask.voice && ask.voice !== "auto"
|
|
214
216
|
? voices.find(voice => voice.id === ask.voice)
|
|
215
|
-
: voices
|
|
217
|
+
: voices[seed % voices.length];
|
|
216
218
|
if (!voice)
|
|
217
219
|
throw new SpeechError("choose an available voice", 400);
|
|
218
220
|
const id = createHash("sha256").update(JSON.stringify([LIVE_VOICE_MODEL, voice.id, ask.language, text])).digest("hex");
|
package/package.json
CHANGED
package/src/live-voice.ts
CHANGED
|
@@ -10,7 +10,7 @@ export const LIVE_VOICE_MODEL = "eleven_flash_v2_5";
|
|
|
10
10
|
export const LIVE_VOICE_RATE = 16_000;
|
|
11
11
|
export const LIVE_VOICE_LANGUAGES = new Set("en ja zh de hi fr ko pt it es id nl tr fil pl sv bg ro ar cs el fi hr ms sk da ta uk ru hu no vi".split(" "));
|
|
12
12
|
export interface LiveVoiceChoice { id: string; name: string; gender: string; language: string; }
|
|
13
|
-
export interface VoiceRequest { text: string; language: string; voice?: string; profile?: VoiceProfile; channel?: string; }
|
|
13
|
+
export interface VoiceRequest { text: string; language: string; voice?: string; profile?: VoiceProfile; channel?: string; speaker?: string; }
|
|
14
14
|
const HEADERS = { "content-type": "audio/pcm", "cache-control": "no-store", "x-audio-sample-rate": String(LIVE_VOICE_RATE) };
|
|
15
15
|
const budget = (value: number | undefined, fallback: number): number => Number.isFinite(value) && value! >= 0 ? Math.floor(value!) : fallback;
|
|
16
16
|
|
|
@@ -192,10 +192,12 @@ export class LiveVoice {
|
|
|
192
192
|
if (!LIVE_VOICE_LANGUAGES.has(ask.language)) throw new SpeechError("this language is not supported by Flash voices", 400);
|
|
193
193
|
const voices = await this.voices();
|
|
194
194
|
signal?.throwIfAborted();
|
|
195
|
-
|
|
195
|
+
// The browser chooses unique voices for speakers. Legacy callers get a
|
|
196
|
+
// stable stock voice; pitch is not used to infer gender.
|
|
197
|
+
const seed = createHash("sha256").update(`${ask.channel ?? ""}|${ask.speaker ?? ""}`).digest().readUInt32BE(0);
|
|
196
198
|
const voice = ask.voice && ask.voice !== "auto"
|
|
197
199
|
? voices.find(voice => voice.id === ask.voice)
|
|
198
|
-
: voices
|
|
200
|
+
: voices[seed % voices.length];
|
|
199
201
|
if (!voice) throw new SpeechError("choose an available voice", 400);
|
|
200
202
|
const id = createHash("sha256").update(JSON.stringify([LIVE_VOICE_MODEL, voice.id, ask.language, text])).digest("hex");
|
|
201
203
|
for (const [key, item] of this.cache) if (item.until < this.now()) this.cache.delete(key);
|