nixamp 0.26.1 → 0.26.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -537,22 +537,13 @@ uses its direct model; it does not first translate the audio into English.
537
537
 
538
538
  ### Hear it in your language
539
539
 
540
- The browser's Transcript panel works with movies, shows, sports, courses,
541
- podcasts, live channels, and rooms. Choose a language and enable **Play
542
- translated audio**. This is a listener preference: the video and the room's
543
- shared playback clock keep running. Turn it off to restore the original sound.
544
- **Start player captions** transcribes other playback in its detected source
545
- language. Files are played locally; these explicit controls opt into sending
546
- short audio clips for processing. Ordinary playback uploads no audio.
547
-
548
- **Translate another tab** opens the browser's audio-sharing chooser, so a
549
- watch party or video hosted on another site can be interpreted too. Select a
550
- browser tab with **Share audio** enabled; **Stop listening** releases sharing.
551
- Only audio is sent, even though the browser requires a video track to select
552
- its source. Sharing support depends on the browser and the source's permissions;
553
- protected media and sources whose audio cannot be captured remain unavailable.
554
- The control requests suppression of the source tab's local sound. If a browser
555
- ignores that option, mute the source tab to avoid hearing both languages.
540
+ Use **Translate audio** beside the player's language menu to hear whatever
541
+ Nixamp is playing in your language. One click starts translation; it selects
542
+ your preferred supported language if the menu is still on Original. Turn it off
543
+ to restore the original audio. The video and the room's shared playback clock
544
+ keep running. **Captions** in the Transcript panel enables text-only recognition.
545
+ Ordinary file playback uploads no audio; enabling captions or translated audio
546
+ opts into processing short clips from the playing media.
556
547
 
557
548
  Native captions use local Whisper Base through Transformers.js. Optional
558
549
  speaker-aware audio uses ElevenLabs **Scribe v2** for native transcription with
@@ -565,18 +556,31 @@ also checked before enabling the audio toggle.
565
556
 
566
557
  A rolling 15-second audio window advances every 5 seconds. Speaker labels are
567
558
  reconciled using overlapping timestamps, with different voices assigned to
568
- separate speakers. Lower/higher pitch suggests a male/female stock voice;
569
- ambiguous audio uses a default. Each speaker's voice can be changed in the
570
- panel. Pitch is not gender identity, and a speaker returning after leaving the
571
- rolling context may receive a new label. Simultaneous speech and noisy crowds
572
- can still confuse recognition. Native captions never translate to English as
559
+ separate speakers. Voices are picked from the available stock catalogue without
560
+ inferring a person's gender from pitch; each detected speaker gets an unused
561
+ voice until the catalogue is exhausted. **Audio options** folds away optional
562
+ individual overrides. A speaker returning after leaving the rolling context may
563
+ receive a new label. Simultaneous speech and noisy crowds can still confuse
564
+ recognition. Native captions never translate to English as
573
565
  an intermediate recognition step.
574
566
 
575
567
  Processing has one active request and only the latest pending window per
576
568
  listener; speech queues and response sizes are bounded. Old transcript history
577
569
  is never spoken. Pause, seek, source changes, and disabling the feature cancel
578
570
  queued speech; errors restore the original audio. This is a delayed live
579
- interpreter, not a promise of exact lip sync or background-music separation.
571
+ interpreter, not a promise of exact lip sync.
572
+
573
+ Translated playback keeps an approximate version of the original background
574
+ sound. FastEnhancer Web's Tiny model estimates speech locally in a dedicated
575
+ browser worker; the player subtracts that estimate from the aligned source in
576
+ each stereo channel and mixes the remainder with translated voices. Background
577
+ processing adds no API calls or provider charges. It stops with translation;
578
+ recognition starts without waiting for it. If the device cannot keep up or load the model, that
579
+ background branch is silenced while translated speech continues. Separation can
580
+ leave some original speech or remove parts of music and crowd noise; disable
581
+ **Keep background sound** under **Audio options** when needed. This uses
582
+ [FastEnhancer Web](https://github.com/ryyr-ry/fastenhancer-web), under the MIT
583
+ license.
580
584
 
581
585
  The account server needs `ELEVENLABS_API_KEY`; `NIXAMP_DUBBING=off` disables
582
586
  this feature. The key stays on the server. Sign-in is required for speaker
@@ -16,6 +16,7 @@ export interface VoiceRequest {
16
16
  voice?: string;
17
17
  profile?: VoiceProfile;
18
18
  channel?: string;
19
+ speaker?: string;
19
20
  }
20
21
  export declare class LiveVoice {
21
22
  private readonly key;
@@ -209,10 +209,12 @@ export class LiveVoice {
209
209
  throw new SpeechError("this language is not supported by Flash voices", 400);
210
210
  const voices = await this.voices();
211
211
  signal?.throwIfAborted();
212
- const gender = ask.profile === "lower" ? "male" : ask.profile === "higher" ? "female" : "neutral";
212
+ // The browser chooses unique voices for speakers. Legacy callers get a
213
+ // stable stock voice; pitch is not used to infer gender.
214
+ const seed = createHash("sha256").update(`${ask.channel ?? ""}|${ask.speaker ?? ""}`).digest().readUInt32BE(0);
213
215
  const voice = ask.voice && ask.voice !== "auto"
214
216
  ? voices.find(voice => voice.id === ask.voice)
215
- : voices.find(voice => voice.gender === gender) ?? voices[0];
217
+ : voices[seed % voices.length];
216
218
  if (!voice)
217
219
  throw new SpeechError("choose an available voice", 400);
218
220
  const id = createHash("sha256").update(JSON.stringify([LIVE_VOICE_MODEL, voice.id, ask.language, text])).digest("hex");
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "nixamp",
3
- "version": "0.26.1",
3
+ "version": "0.26.3",
4
4
  "description": "It really whips the terminal's ass. A Winamp-shaped audio player for your terminal.",
5
5
  "license": "MIT",
6
6
  "type": "module",
package/src/live-voice.ts CHANGED
@@ -10,7 +10,7 @@ export const LIVE_VOICE_MODEL = "eleven_flash_v2_5";
10
10
  export const LIVE_VOICE_RATE = 16_000;
11
11
  export const LIVE_VOICE_LANGUAGES = new Set("en ja zh de hi fr ko pt it es id nl tr fil pl sv bg ro ar cs el fi hr ms sk da ta uk ru hu no vi".split(" "));
12
12
  export interface LiveVoiceChoice { id: string; name: string; gender: string; language: string; }
13
- export interface VoiceRequest { text: string; language: string; voice?: string; profile?: VoiceProfile; channel?: string; }
13
+ export interface VoiceRequest { text: string; language: string; voice?: string; profile?: VoiceProfile; channel?: string; speaker?: string; }
14
14
  const HEADERS = { "content-type": "audio/pcm", "cache-control": "no-store", "x-audio-sample-rate": String(LIVE_VOICE_RATE) };
15
15
  const budget = (value: number | undefined, fallback: number): number => Number.isFinite(value) && value! >= 0 ? Math.floor(value!) : fallback;
16
16
 
@@ -192,10 +192,12 @@ export class LiveVoice {
192
192
  if (!LIVE_VOICE_LANGUAGES.has(ask.language)) throw new SpeechError("this language is not supported by Flash voices", 400);
193
193
  const voices = await this.voices();
194
194
  signal?.throwIfAborted();
195
- const gender = ask.profile === "lower" ? "male" : ask.profile === "higher" ? "female" : "neutral";
195
+ // The browser chooses unique voices for speakers. Legacy callers get a
196
+ // stable stock voice; pitch is not used to infer gender.
197
+ const seed = createHash("sha256").update(`${ask.channel ?? ""}|${ask.speaker ?? ""}`).digest().readUInt32BE(0);
196
198
  const voice = ask.voice && ask.voice !== "auto"
197
199
  ? voices.find(voice => voice.id === ask.voice)
198
- : voices.find(voice => voice.gender === gender) ?? voices[0];
200
+ : voices[seed % voices.length];
199
201
  if (!voice) throw new SpeechError("choose an available voice", 400);
200
202
  const id = createHash("sha256").update(JSON.stringify([LIVE_VOICE_MODEL, voice.id, ask.language, text])).digest("hex");
201
203
  for (const [key, item] of this.cache) if (item.until < this.now()) this.cache.delete(key);