nixamp 0.25.1 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +108 -7
  2. package/dist/captions.d.ts +7 -0
  3. package/dist/captions.js +92 -28
  4. package/dist/live-voice.d.ts +70 -0
  5. package/dist/live-voice.js +299 -0
  6. package/dist/server.d.ts +2 -0
  7. package/dist/server.js +154 -5
  8. package/dist/speaker-turns.d.ts +32 -0
  9. package/dist/speaker-turns.js +30 -0
  10. package/dist/speech.d.ts +47 -0
  11. package/dist/speech.js +49 -4
  12. package/dist/transcript-client.d.ts +2 -2
  13. package/dist/transcript-client.js +5 -2
  14. package/dist/transcripts.d.ts +5 -0
  15. package/dist/transcripts.js +8 -3
  16. package/dist/translate-jobs.js +13 -7
  17. package/dist/translate.d.ts +2 -0
  18. package/dist/translate.js +9 -0
  19. package/dist/voice-profile.d.ts +7 -0
  20. package/dist/voice-profile.js +53 -0
  21. package/dist/warm.js +3 -3
  22. package/package.json +1 -1
  23. package/src/captions.ts +86 -25
  24. package/src/live-voice.ts +255 -0
  25. package/src/server.ts +104 -5
  26. package/src/speaker-turns.ts +34 -0
  27. package/src/speech.ts +46 -7
  28. package/src/transcript-client.ts +4 -0
  29. package/src/transcripts.ts +13 -3
  30. package/src/translate-jobs.ts +13 -7
  31. package/src/translate.ts +7 -1
  32. package/src/voice-profile.ts +46 -0
  33. package/src/warm.ts +3 -3
  34. package/web/dist/assets/{hls-3VKVEQE3-GlzSYi0P.js → hls-3VKVEQE3-vgax_tk1.js} +1 -1
  35. package/web/dist/assets/index-BzjrTOLf.js +1 -0
  36. package/web/dist/assets/index-D2Iy07pG.css +1 -0
  37. package/web/dist/assets/{mpegts-KIbyW_RL.js → mpegts-DmcUOiHq.js} +1 -1
  38. package/web/dist/assets/{mpegts-LO6RVLD6-Fs_vbGcB.js → mpegts-LO6RVLD6-CH3EQi6L.js} +1 -1
  39. package/web/dist/index.html +42 -22
  40. package/web/dist/sw.js +6 -6
  41. package/web/dist/assets/index-DJHHB_63.js +0 -1
  42. package/web/dist/assets/index-oyp61Kly.css +0 -1
package/src/server.ts CHANGED
@@ -16,6 +16,9 @@ import { randomBytes } from "node:crypto";
16
16
  import { hostname, networkInterfaces } from "node:os";
17
17
  import { spawn, spawnSync } from "node:child_process";
18
18
  import { readFileSync } from "node:fs";
19
+ import { Readable } from "node:stream";
20
+ import { pipeline as pipeStream } from "node:stream/promises";
21
+ import { LiveVoice, LIVE_VOICE_LANGUAGES, LIVE_VOICE_MODEL, type VoiceRequest } from "./live-voice.ts";
19
22
  import { Connections, type Kind } from "./connections.ts";
20
23
  import {
21
24
  Broadcaster,
@@ -86,7 +89,7 @@ import { Tickets, needsTicket, ticketFrom, ticketsFromEnv } from "./tickets.ts";
86
89
  import { Layouts } from "./layouts.ts";
87
90
  import { Rooms } from "./rooms.ts";
88
91
  import { Trollbox, TrollboxError, fallbackHandle, roomFor } from "./trollbox.ts";
89
- import { MAX_BYTES as SPEECH_BYTES, Speech, SpeechError, isWav, languageOf } from "./speech.ts";
92
+ import { MAX_BYTES as SPEECH_BYTES, NATIVE_REVISION, Speech, SpeechError, isWav, languageOf } from "./speech.ts";
90
93
  import { Captions } from "./captions.ts";
91
94
  import { LANGUAGES, Translator } from "./translate.ts";
92
95
  import { StoredTranslations } from "./translate-jobs.ts";
@@ -1350,7 +1353,7 @@ const CORS: Record<string, string> = {
1350
1353
  // paths and takes six commands; binding to 127.0.0.1 is what keeps it shut.
1351
1354
  "access-control-allow-origin": "*",
1352
1355
  "access-control-allow-methods": "GET, POST, PUT, DELETE, OPTIONS",
1353
- "access-control-allow-headers": "content-type",
1356
+ "access-control-allow-headers": "content-type, authorization",
1354
1357
  "access-control-max-age": "86400",
1355
1358
  };
1356
1359
 
@@ -1548,6 +1551,7 @@ export function joinDocument(shell: string, subject: JoinSubject, site: string):
1548
1551
  }
1549
1552
 
1550
1553
  function json(response: ServerResponse, code: number, body: unknown): void {
1554
+ if (code === 429) response.setHeader("retry-after", "60");
1551
1555
  const text = JSON.stringify(body);
1552
1556
  response.writeHead(code, {
1553
1557
  ...CORS,
@@ -1558,6 +1562,16 @@ function json(response: ServerResponse, code: number, body: unknown): void {
1558
1562
  response.end(text);
1559
1563
  }
1560
1564
 
1565
+ /** Stream PCM without waiting for a complete utterance, respecting browser backpressure. */
1566
+ async function voiceResponse(response: ServerResponse, upstream: Response): Promise<void> {
1567
+ response.writeHead(upstream.status, {
1568
+ ...CORS, "content-type": upstream.headers.get("content-type") ?? "application/json",
1569
+ "cache-control": "no-store", "x-accel-buffering": "no",
1570
+ });
1571
+ if (!upstream.body) { response.end(); return; }
1572
+ await pipeStream(Readable.fromWeb(upstream.body as import("node:stream/web").ReadableStream), response);
1573
+ }
1574
+
1561
1575
  async function readBody(request: IncomingMessage, limit = 64 * 1024): Promise<string> {
1562
1576
  const chunks: Buffer[] = [];
1563
1577
  let size = 0;
@@ -1745,6 +1759,7 @@ export interface HandlerOptions {
1745
1759
  trollbox?: Trollbox;
1746
1760
  /** Speech to text: a line said out loud, heard here. Needs the optional model. */
1747
1761
  speech?: Speech;
1762
+ liveVoice?: LiveVoice;
1748
1763
  /** Translation: texts in another language, by a model here. The same optional library. */
1749
1764
  translator?: Translator;
1750
1765
  /** The transcript store: what was heard, kept under the media's identity. Where the accounts are. */
@@ -2843,6 +2858,45 @@ export function createHandler(engine: Engine, options: HandlerOptions) {
2843
2858
  * room, the words go straight into that room's trollbox as a line by
2844
2859
  * whoever spoke them. The page, the CLI and the MCP tools all come here.
2845
2860
  */
2861
+ if (["/api/v1/speech/voices", "/api/v1/speech/grant", "/api/v1/speech/synthesize", "/api/v1/speech/speakers"].includes(path) && options.accounts) {
2862
+ const controller = new AbortController();
2863
+ response.once("close", () => controller.abort());
2864
+ try {
2865
+ const caller = callerOf(request.headers, request.socket.remoteAddress, options.behindProxy ?? false);
2866
+ if (!guard.check(`voice-ip:${caller}`, { allowed: 600, windowMs: 60_000 }).ok) { json(response, 429, { error: "too many voice requests" }); return; }
2867
+ const service = options.liveVoice;
2868
+ if (!service?.available()) { json(response, 503, { error: "translated audio needs ELEVENLABS_API_KEY on the account server" }); return; }
2869
+ if (path.endsWith("/synthesize") && request.method === "POST") {
2870
+ let body: VoiceRequest;
2871
+ try { body = JSON.parse(await readBody(request, 4096)) as VoiceRequest; }
2872
+ catch { json(response, 400, { error: "send a caption as JSON" }); return; }
2873
+ if (!body || typeof body !== "object") { json(response, 400, { error: "send a caption as JSON" }); return; }
2874
+ const by = await service.authorize(tokenFrom(request.headers), body.channel ?? "", typeof body.text === "string" ? body.text.length : 0);
2875
+ await voiceResponse(response, await service.stream(body, by, controller.signal));
2876
+ } else {
2877
+ const who = await options.accounts.whoIs(tokenFrom(request.headers));
2878
+ if (!who) { json(response, 401, { error: "sign in to use translated audio" }); return; }
2879
+ if (path.endsWith("/voices") && request.method === "GET") {
2880
+ json(response, 200, { voices: await service.voices(), model: LIVE_VOICE_MODEL, languages: [...LIVE_VOICE_LANGUAGES] });
2881
+ } else if (path.endsWith("/speakers") && request.method === "POST") {
2882
+ if (!request.headers["content-type"]?.startsWith("audio/wav")) { json(response, 415, { error: "send a mono 16 kHz WAV" }); return; }
2883
+ const bytes = await readBytes(request, 484_000);
2884
+ json(response, 200, await service.hear(bytes, who.id, controller.signal));
2885
+ } else if (path.endsWith("/grant") && request.method === "POST") {
2886
+ if (!request.headers["content-type"]?.startsWith("application/json")) { json(response, 415, { error: "send JSON" }); return; }
2887
+ let body: { channel?: unknown };
2888
+ try { body = JSON.parse(await readBody(request, 1024)) as typeof body; }
2889
+ catch { json(response, 400, { error: "send a channel as JSON" }); return; }
2890
+ json(response, 200, await service.grant(who.id, typeof body?.channel === "string" ? body.channel : ""));
2891
+ } else json(response, 405, { error: "GET voices or POST an audio request" });
2892
+ }
2893
+ } catch (error) {
2894
+ if (response.headersSent) response.destroy();
2895
+ else json(response, error instanceof SpeechError ? error.status : 502, { error: error instanceof SpeechError ? error.message : "translated audio is unavailable" });
2896
+ }
2897
+ return;
2898
+ }
2899
+
2846
2900
  if (path === "/api/v1/speech/transcribe" && options.speech && options.accounts) {
2847
2901
  const speech = options.speech;
2848
2902
  try {
@@ -2872,6 +2926,7 @@ export function createHandler(engine: Engine, options: HandlerOptions) {
2872
2926
  // Pieces with their timing, for a whole file being written down.
2873
2927
  timestamps: url.searchParams.get("timestamps") === "1",
2874
2928
  by: who.id,
2929
+ ...(url.searchParams.get("live") === "1" ? { deadline: Date.now() + 10_000 } : {}),
2875
2930
  });
2876
2931
  // The language travels back: as told, or as the ear guessed it, so
2877
2932
  // a captioner can say it next time and a transcript can be kept as it.
@@ -3082,7 +3137,7 @@ export function createHandler(engine: Engine, options: HandlerOptions) {
3082
3137
  return;
3083
3138
  }
3084
3139
  try {
3085
- json(response, 200, await translator.translate(texts, from, to, { by: who.id }));
3140
+ json(response, 200, await translator.translate(texts, from, to, { by: who.id, ...(url.searchParams.get("live") === "1" ? { deadline: Date.now() + 10_000 } : {}) }));
3086
3141
  } catch (error) {
3087
3142
  if (error instanceof SpeechError) json(response, error.status, { error: error.message });
3088
3143
  else throw error;
@@ -3136,6 +3191,12 @@ export function createHandler(engine: Engine, options: HandlerOptions) {
3136
3191
  json(response, 400, { error: "format is json, srt, vtt or txt" });
3137
3192
  return;
3138
3193
  }
3194
+ // Joining a live must not start translating an entire old transcript.
3195
+ if (url.searchParams.get("cached") === "1") {
3196
+ const saved = await store.get(id, language);
3197
+ json(response, saved ? 200 : 404, saved ? wire(saved) : { error: "no cached transcript" });
3198
+ return;
3199
+ }
3139
3200
  const answer = await options.translations.get(id, language, who.id);
3140
3201
  if (answer.status !== 200 && answer.status !== 202) {
3141
3202
  json(response, answer.status, { error: answer.error });
@@ -4486,6 +4547,31 @@ export function createHandler(engine: Engine, options: HandlerOptions) {
4486
4547
  * started by the first person asking. `captions` is the live stream
4487
4548
  * of lines; `transcript` is the recent ones as JSON, for a poll.
4488
4549
  */
4550
+ if ((action === "voice" || action === "voice-options") && request.method === "GET") {
4551
+ const controller = new AbortController();
4552
+ response.once("close", () => controller.abort());
4553
+ const timeout = setTimeout(() => controller.abort(), 12_000);
4554
+ try {
4555
+ const caller = callerOf(request.headers, request.socket.remoteAddress, options.behindProxy ?? false);
4556
+ if (!guard.check(`channel-voice-ip:${caller}`, { allowed: 60, windowMs: 60_000 }).ok || !guard.check(`channel-voice:${id}`, { allowed: 240, windowMs: 60_000 }).ok) {
4557
+ json(response, 429, { error: "too many translated audio requests" }); return;
4558
+ }
4559
+ if (!channels.has(id)) { json(response, 404, { error: "nothing is playing on that channel" }); return; }
4560
+ if (!options.captions) { json(response, 503, { error: "this server cannot caption" }); return; }
4561
+ const at = action === "voice-options" ? null : Number(url.searchParams.get("at"));
4562
+ const language = languageCode(url.searchParams.get("language"));
4563
+ if ((at !== null && (!url.searchParams.has("at") || !Number.isFinite(at))) || language === null) {
4564
+ json(response, 400, { error: "audio needs a caption time and language" }); return;
4565
+ }
4566
+ const upstream = await options.captions.voiceRequest(id, at, language, url.searchParams.get("voice") ?? "auto", controller.signal, tokenFrom(request.headers));
4567
+ await voiceResponse(response, upstream);
4568
+ } catch (error) {
4569
+ if (response.headersSent) response.destroy();
4570
+ else json(response, error instanceof SpeechError ? error.status : 502, { error: error instanceof SpeechError ? error.message : "translated audio is unavailable" });
4571
+ } finally { clearTimeout(timeout); }
4572
+ return;
4573
+ }
4574
+
4489
4575
  if ((action === "captions" || action === "transcript") && request.method === "GET") {
4490
4576
  const captions = options.captions;
4491
4577
  if (!captions) {
@@ -4496,6 +4582,11 @@ export function createHandler(engine: Engine, options: HandlerOptions) {
4496
4582
  json(response, 503, { error: "this server is not signed in to nixamp.com, so it cannot caption; run `nixamp login` on it" });
4497
4583
  return;
4498
4584
  }
4585
+ const caller = callerOf(request.headers, request.socket.remoteAddress, options.behindProxy ?? false);
4586
+ if (!guard.check(`captions:${caller}`, { allowed: 120, windowMs: 60_000 }).ok || !captions.capacity(id)) {
4587
+ json(response, 429, { error: "captioning is at capacity; try again in a moment" });
4588
+ return;
4589
+ }
4499
4590
  if (!channels.has(id)) {
4500
4591
  json(response, 404, { error: "nothing is playing on that channel" });
4501
4592
  return;
@@ -4514,7 +4605,7 @@ export function createHandler(engine: Engine, options: HandlerOptions) {
4514
4605
  // Asking keeps the captioner up: it stops a minute after the last ask.
4515
4606
  captions.subscribe(id, () => undefined, wanted)?.();
4516
4607
  json(response, 200, {
4517
- channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted),
4608
+ channel: id, recognitionRevision: NATIVE_REVISION, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted),
4518
4609
  });
4519
4610
  return;
4520
4611
  }
@@ -4535,7 +4626,7 @@ export function createHandler(engine: Engine, options: HandlerOptions) {
4535
4626
  }
4536
4627
  // How far behind the live edge a newcomer's playback starts, so the
4537
4628
  // page can hold each line until its own sound gets there.
4538
- write("hello", { channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted) });
4629
+ write("hello", { channel: id, recognitionRevision: NATIVE_REVISION, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted) });
4539
4630
  const beat = setInterval(() => response.write(": beat\n\n"), 20_000);
4540
4631
  beat.unref?.();
4541
4632
  const done = (): void => {
@@ -5909,6 +6000,13 @@ export async function serve(argv: string[], version = "0.1.0"): Promise<void> {
5909
6000
  // the first person to speak does not wait for the model to arrive; a box
5910
6001
  // without the optional model never says it is ready, and answers 503.
5911
6002
  const speech = accounts && process.env["NIXAMP_STT"] !== "off" ? new Speech() : undefined;
6003
+ const liveVoice = accounts && process.env["NIXAMP_DUBBING"] !== "off" ? new LiveVoice({
6004
+ ...(pool ? { db: pool } : {}),
6005
+ dailyChars: Number(process.env["NIXAMP_DUB_DAILY_CHARS"] ?? 200_000),
6006
+ dailyAudioSeconds: Number(process.env["NIXAMP_DUB_DAILY_AUDIO_SECONDS"] ?? 86_400),
6007
+ userDailyChars: Number(process.env["NIXAMP_DUB_USER_DAILY_CHARS"] ?? 120_000),
6008
+ userDailyAudioSeconds: Number(process.env["NIXAMP_DUB_USER_DAILY_AUDIO_SECONDS"] ?? 43_200),
6009
+ }) : undefined;
5912
6010
  if (speech) {
5913
6011
  void speech.warm().then((ready) => {
5914
6012
  console.error(ready ? `nixamp: hearing with ${speech.model}` : `nixamp: not hearing: ${speech.lastFailure}`);
@@ -6400,6 +6498,7 @@ export async function serve(argv: string[], version = "0.1.0"): Promise<void> {
6400
6498
  ...(rooms ? { rooms } : {}),
6401
6499
  ...(trollbox ? { trollbox } : {}),
6402
6500
  ...(speech ? { speech } : {}),
6501
+ ...(liveVoice ? { liveVoice } : {}),
6403
6502
  ...(translator ? { translator } : {}),
6404
6503
  ...(transcripts ? { transcripts } : {}),
6405
6504
  ...(translations ? { translations } : {}),
@@ -0,0 +1,34 @@
1
+ /** Provider speaker IDs belong to a clip. Clients reconcile the overlapping
2
+ * word timestamps before assigning a voice; these are not people's identities. */
3
+ import { reliableText, type Wav } from "./speech.ts";
4
+ import { voiceProfile, type VoiceProfile } from "./voice-profile.ts";
5
+
6
+ export interface SpeakerTurn { start: number; end: number; text: string; speaker: string; profile: VoiceProfile; words: { text: string; start: number; end: number }[]; }
7
+ export interface SpeakerTranscript { language: string; seconds: number; turns: SpeakerTurn[]; }
8
+ export interface ScribeResult {
9
+ language_code?: string;
10
+ words?: { text: string; start: number; end: number; type: string; speaker_id?: string | null }[];
11
+ }
12
+ const ISO: Record<string, string> = Object.fromEntries("eng:en spa:es deu:de ger:de fra:fr fre:fr por:pt ita:it nld:nl dut:nl swe:sv dan:da fin:fi rus:ru ukr:uk ces:cs cze:cs hun:hu cmn:zh zho:zh ara:ar hin:hi vie:vi ind:id jpn:ja kor:ko pol:pl tur:tr ron:ro bul:bg ell:el nor:no nob:no nno:no".split(" ").map(pair => pair.split(":")));
13
+
14
+ export function speakerTurns(body: ScribeResult, wav: Wav): SpeakerTranscript {
15
+ const seconds = wav.samples.length / wav.rate;
16
+ const language = ISO[body.language_code ?? ""] ?? body.language_code ?? "";
17
+ const turns: SpeakerTurn[] = [];
18
+ for (const word of (body.words ?? []).slice(0, 1500)) {
19
+ if (word.type !== "word" || typeof word.text !== "string" || !Number.isFinite(word.start) || !Number.isFinite(word.end) || word.start < 0 || word.end < word.start || word.end > seconds + 0.5) continue;
20
+ const speaker = typeof word.speaker_id === "string" && /^[\w-]{1,80}$/.test(word.speaker_id) ? word.speaker_id : "unknown";
21
+ const last = turns.at(-1);
22
+ if (last && last.speaker === speaker && word.start - last.end < 0.8 && last.text.length + word.text.length < 400) {
23
+ last.text += ` ${word.text}`; last.end = Math.max(last.end, word.end);
24
+ last.words.push({ text: word.text, start: word.start, end: word.end });
25
+ } else turns.push({ start: word.start, end: word.end, text: word.text, speaker, profile: "unknown", words: [{ text: word.text, start: word.start, end: word.end }] });
26
+ }
27
+ const pcm = Buffer.alloc(wav.samples.length * 2);
28
+ wav.samples.forEach((sample, i) => pcm.writeInt16LE(Math.round(Math.max(-1, Math.min(1, sample)) * 32767), i * 2));
29
+ for (const turn of turns) {
30
+ turn.text = reliableText(turn.text, Math.max(1, turn.end - turn.start));
31
+ turn.profile = voiceProfile(pcm.subarray(Math.floor(turn.start * 16000) * 2, Math.ceil(turn.end * 16000) * 2));
32
+ }
33
+ return { language, seconds, turns: turns.filter(turn => turn.text !== "") };
34
+ }
package/src/speech.ts CHANGED
@@ -40,6 +40,8 @@ export const SECONDS_PER_MINUTE = 300;
40
40
  /** How many may wait for the one CPU. Past this, the honest answer is "later". */
41
41
  export const QUEUE_LIMIT = 8;
42
42
  export const DEFAULT_MODEL = "onnx-community/whisper-base";
43
+ /** Stored live lines from older recognition settings must be heard again. */
44
+ export const NATIVE_REVISION = "native-v2";
43
45
 
44
46
  export class SpeechError extends Error {
45
47
  constructor(message: string, readonly status: number) {
@@ -215,6 +217,29 @@ export function tidy(text: string): string {
215
217
  return text.replace(/\[[A-Z_ ]+\]|\([A-Za-z ]+\)/g, " ").replace(/\s+/g, " ").trim();
216
218
  }
217
219
 
220
+ /** Reject decoding loops, without removing normal emphasis like “no, no, no”. */
221
+ export function reliableText(text: string, seconds: number): string {
222
+ const clean = text.replace(/\s+/g, " ").trim();
223
+ if (clean.length > Math.max(120, seconds * 55)) return "";
224
+ const words = clean.toLowerCase().match(/[\p{L}\p{N}']+/gu) ?? [];
225
+ for (let size = 1; size <= Math.min(20, Math.floor(words.length / 4)); size++) {
226
+ for (let start = 0; start + size * 4 <= words.length; start++) {
227
+ let repeated = true;
228
+ for (let i = size; i < size * 4; i++) {
229
+ if (words[start + i] !== words[start + i % size]) { repeated = false; break; }
230
+ }
231
+ if (repeated) return "";
232
+ }
233
+ }
234
+ return clean;
235
+ }
236
+
237
+ export function quietSamples(pcm: Float32Array): boolean {
238
+ let energy = 0;
239
+ for (const sample of pcm) energy += sample * sample;
240
+ return pcm.length === 0 || Math.sqrt(energy / pcm.length) < 0.004;
241
+ }
242
+
218
243
  /** A two-letter language code, or nothing: Whisper guesses when not told. */
219
244
  export function languageOf(value: unknown): string | undefined {
220
245
  if (typeof value !== "string") return undefined;
@@ -229,7 +254,7 @@ interface AsrOutput {
229
254
  }
230
255
 
231
256
  /** The pipeline, and the parts under it that language detection needs. */
232
- interface AsrPipeline {
257
+ export interface AsrPipeline {
233
258
  (audio: Float32Array, options: Record<string, unknown>): Promise<AsrOutput | AsrOutput[]>;
234
259
  model: {
235
260
  (inputs: Record<string, unknown>): Promise<{ logits: { data: Float32Array | number[] } }>;
@@ -239,7 +264,7 @@ interface AsrPipeline {
239
264
  }
240
265
 
241
266
  /** The module, as much of it as this file touches. Typed here so the import can be by name. */
242
- interface Transformers {
267
+ export interface Transformers {
243
268
  env: { cacheDir?: string; allowLocalModels?: boolean };
244
269
  Tensor: new (type: string, data: BigInt64Array, dims: number[]) => unknown;
245
270
  pipeline(task: "automatic-speech-recognition", model: string, options: { dtype: string }): Promise<AsrPipeline>;
@@ -296,13 +321,25 @@ async function loadWhisper(model: string, cacheDir: string): Promise<Recognizer>
296
321
  const transformers = await loadTransformers<Transformers>();
297
322
  transformers.env.cacheDir = cacheDir;
298
323
  const recognize = await transformers.pipeline("automatic-speech-recognition", model, { dtype: "q8" });
324
+ return whisperRecognizer(transformers, recognize);
325
+ }
326
+
327
+ /** Detection is from each recording, never from the caption translation selection. */
328
+ export function whisperRecognizer(transformers: Transformers, recognize: AsrPipeline): Recognizer {
299
329
  return async (pcm, { language, timestamps }) => {
300
- const spoken = language ?? (await detectLanguage(transformers, recognize, pcm));
330
+ const multilingual = recognize.model.generation_config.is_multilingual !== false;
331
+ const spoken = language ?? (multilingual ? await detectLanguage(transformers, recognize, pcm) : "en");
332
+ if (multilingual && !spoken) throw new SpeechError("could not detect the audio language; try another speech segment", 422);
301
333
  const heard = await recognize(pcm, {
302
334
  // Whisper hears thirty seconds at a time; longer is heard in overlapping pieces.
303
335
  chunk_length_s: 30,
304
336
  stride_length_s: 5,
305
- ...(spoken ? { language: spoken, task: "transcribe" } : {}),
337
+ ...(multilingual ? { language: spoken, task: "transcribe" } : {}),
338
+ // A five-second clip used to be allowed hundreds of tokens of hallucinated loops.
339
+ max_new_tokens: Math.min(384, Math.max(32, Math.ceil(Math.min(30, pcm.length / RATE) * 10) + 16)),
340
+ do_sample: false,
341
+ num_beams: 1,
342
+ no_repeat_ngram_size: 6,
306
343
  ...(timestamps ? { return_timestamps: true } : {}),
307
344
  });
308
345
  const pieces = Array.isArray(heard) ? heard : [heard];
@@ -385,7 +422,7 @@ export class Speech {
385
422
  * The words in a WAV. Refuses what is not a WAV, or is too long, or more
386
423
  * than the account may have heard this minute, with a status.
387
424
  */
388
- async transcribe(bytes: Uint8Array, options: { language?: string; timestamps?: boolean; by?: string } = {}): Promise<Heard> {
425
+ async transcribe(bytes: Uint8Array, options: { language?: string; timestamps?: boolean; by?: string; deadline?: number } = {}): Promise<Heard> {
389
426
  if (bytes.length > MAX_BYTES) throw new SpeechError(`that is too much sound: ${MAX_SECONDS} seconds at most`, 413);
390
427
  const wav = decodeWav(bytes);
391
428
  const seconds = wav.samples.length / wav.rate;
@@ -393,9 +430,11 @@ export class Speech {
393
430
  if (options.by) this.allow(options.by, seconds);
394
431
  if (wav.samples.length < wav.rate / 10) return { text: "", seconds };
395
432
  const pcm = resample(wav.samples, wav.rate, RATE);
433
+ if (quietSamples(pcm)) return { text: "", seconds };
396
434
  if (this.waiting >= QUEUE_LIMIT) throw new SpeechError("too many people are talking at once; try again in a moment", 503);
397
435
  this.waiting += 1;
398
436
  const turn = this.tail.then(async () => {
437
+ if (options.deadline && this.now() > options.deadline) throw new SpeechError("this live audio is too old; waiting for the next segment", 408);
399
438
  const recognize = await this.ear();
400
439
  return recognize(pcm, {
401
440
  ...(options.language ? { language: options.language } : {}),
@@ -407,10 +446,10 @@ export class Speech {
407
446
  try {
408
447
  const heard = await turn;
409
448
  const segments = heard.segments
410
- ?.map((segment) => ({ start: segment.start, end: segment.end, text: tidy(segment.text) }))
449
+ ?.map((segment) => ({ start: segment.start, end: segment.end, text: reliableText(tidy(segment.text), segment.end - segment.start) }))
411
450
  .filter((segment) => segment.text !== "");
412
451
  return {
413
- text: tidy(heard.text),
452
+ text: reliableText(tidy(heard.text), seconds),
414
453
  seconds,
415
454
  ...(heard.language ? { language: heard.language } : {}),
416
455
  ...(segments ? { segments } : {}),
@@ -53,9 +53,11 @@ export async function fetchTranscript(
53
53
  id: string,
54
54
  language = "",
55
55
  fetcher: typeof fetch = fetch,
56
+ cachedOnly = false,
56
57
  ): Promise<Got<StoredTranscript>> {
57
58
  const url = new URL(`${base(signed.site)}/api/v1/transcripts/${encodeURIComponent(id)}`);
58
59
  if (language) url.searchParams.set("language", language);
60
+ if (cachedOnly) url.searchParams.set("cached", "1");
59
61
  try {
60
62
  const response = await fetcher(url.toString(), { headers: { authorization: `Bearer ${signed.token}` } });
61
63
  return await asJson<StoredTranscript>(response);
@@ -107,12 +109,14 @@ export async function translateTexts(
107
109
  from: string,
108
110
  to: string,
109
111
  fetcher: typeof fetch = fetch,
112
+ signal?: AbortSignal,
110
113
  ): Promise<Got<Translated>> {
111
114
  try {
112
115
  const response = await fetcher(`${base(signed.site)}/api/v1/translate`, {
113
116
  method: "POST",
114
117
  headers: { authorization: `Bearer ${signed.token}`, "content-type": "application/json" },
115
118
  body: JSON.stringify({ texts, from, to }),
119
+ signal,
116
120
  });
117
121
  return await asJson<Translated>(response);
118
122
  } catch (error) {
@@ -30,6 +30,11 @@ export interface TranscriptLine {
30
30
  start: number;
31
31
  end: number;
32
32
  text: string;
33
+ /** Live transcripts may change language within one programme. */
34
+ language?: string;
35
+ revision?: string;
36
+ original?: string;
37
+ voiceProfile?: "lower" | "higher" | "unknown";
33
38
  }
34
39
 
35
40
  export type MediaKind = "file" | "url" | "live";
@@ -159,12 +164,17 @@ export function linesFrom(value: unknown): TranscriptLine[] {
159
164
  const lines: TranscriptLine[] = [];
160
165
  for (const one of parsed) {
161
166
  if (!one || typeof one !== "object") continue;
162
- const { start, end, text } = one as Record<string, unknown>;
167
+ const { start, end, text, language, revision, original, voiceProfile } = one as Record<string, unknown>;
163
168
  if (typeof text !== "string" || typeof start !== "number" || !Number.isFinite(start) || start < 0) continue;
164
169
  const words = text.replace(/\s+/g, " ").trim().slice(0, MAX_LINE_CHARS);
165
170
  if (words === "") continue;
166
171
  const until = typeof end === "number" && Number.isFinite(end) && end > start ? end : start;
167
- lines.push({ start: round(start), end: round(until), text: words });
172
+ lines.push({ start: round(start), end: round(until), text: words,
173
+ ...(typeof language === "string" && /^[a-z]{2,3}$/.test(language) ? { language } : {}),
174
+ ...(typeof revision === "string" ? { revision: revision.slice(0, 32) } : {}),
175
+ ...(typeof original === "string" ? { original: original.slice(0, MAX_LINE_CHARS) } : {}),
176
+ ...(voiceProfile === "lower" || voiceProfile === "higher" || voiceProfile === "unknown" ? { voiceProfile } : {}),
177
+ });
168
178
  }
169
179
  return lines;
170
180
  }
@@ -371,7 +381,7 @@ export class Transcripts {
371
381
  const { rows } = language === ""
372
382
  ? await this.db.query(
373
383
  `SELECT * FROM transcripts WHERE id = $1 AND translated_from IS NULL
374
- ORDER BY complete DESC, updated_at DESC LIMIT 1`,
384
+ ORDER BY (language = '') DESC, complete DESC, updated_at DESC LIMIT 1`,
375
385
  [id],
376
386
  )
377
387
  : await this.db.query("SELECT * FROM transcripts WHERE id = $1 AND language = $2 LIMIT 1", [id, language]);
@@ -60,9 +60,10 @@ export class StoredTranslations {
60
60
  const translation = await this.store.get(id, language);
61
61
  const missing = StoredTranslations.missing(original, translation);
62
62
  if (translation && missing.length === 0) return { status: 200, transcript: translation };
63
- if (original.language === "") return { status: 409, error: "the language this was heard in is not known, so it cannot be translated" };
63
+ const sources = new Set(missing.map((line) => line.language || original.language));
64
+ if (sources.has("")) return { status: 409, error: "the language this was heard in is not known, so it cannot be translated" };
64
65
  if (!this.translator) return { status: 503, error: "this nixamp cannot translate: no model here. nixamp.com can." };
65
- if (!this.translator.can(original.language, language)) {
66
+ if ([...sources].some((source) => !this.translator!.can(source, language))) {
66
67
  return { status: 409, error: `there is no model here from ${original.language} to ${language}` };
67
68
  }
68
69
  const key = `${id}|${language}`;
@@ -107,19 +108,24 @@ export class StoredTranslations {
107
108
  let model = "";
108
109
  for (let at = 0; at < missing.length; at += BATCH) {
109
110
  const batch = missing.slice(at, at + BATCH);
110
- const done = await translator.translate(batch.map((line) => line.text), original.language, language);
111
- model = done.model;
112
- const lines = batch.map((line, i) => ({ start: line.start, end: line.end, text: done.texts[i] ?? "" })).filter((line) => line.text !== "");
111
+ const lines: TranscriptLine[] = [];
112
+ for (const source of new Set(batch.map((line) => line.language || original.language))) {
113
+ const group = batch.filter((line) => (line.language || original.language) === source);
114
+ const done = await translator.translate(group.map((line) => line.text), source, language);
115
+ model = done.model;
116
+ lines.push(...group.map((line, i) => ({ ...line, language, original: line.text, text: done.texts[i] ?? "" })).filter((line) => line.text !== ""));
117
+ }
118
+ lines.sort((a, b) => a.start - b.start);
113
119
  made.push(...lines);
114
120
  await this.store.save({
115
- media: original.media, language, translatedFrom: original.language, model, title: original.title, by, lines,
121
+ media: original.media, language, translatedFrom: original.language || "mul", model, title: original.title, by, lines,
116
122
  });
117
123
  job.progress.done += batch.length;
118
124
  }
119
125
  if (original.complete) {
120
126
  // Whole, like the original: the pieces are replaced by the lot, and nothing partial touches it again.
121
127
  await this.store.save({
122
- media: original.media, language, translatedFrom: original.language, model, title: original.title, by, lines: made, complete: true,
128
+ media: original.media, language, translatedFrom: original.language || "mul", model, title: original.title, by, lines: made, complete: true,
123
129
  });
124
130
  }
125
131
  if (missing.length > 0) this.onEvent(` translated ${missing.length} lines of ${original.title || original.media} to ${language}`);
package/src/translate.ts CHANGED
@@ -164,6 +164,7 @@ export class Translator {
164
164
  private readonly keep: number;
165
165
  private readonly pairs: ReadonlySet<string>;
166
166
  private readonly loaded = new Map<string, { pair: Promise<Pair>; usedAt: number }>();
167
+ private readonly downloadCooldown = new Map<string, number>();
167
168
  private tail: Promise<unknown> = Promise.resolve();
168
169
  private waiting = 0;
169
170
  private readonly asked = new Map<string, { minute: number; chars: number }>();
@@ -189,6 +190,7 @@ export class Translator {
189
190
 
190
191
  private pair(from: string, to: string): Promise<Pair> {
191
192
  const model = modelFor(from, to);
193
+ if ((this.downloadCooldown.get(model) ?? 0) > this.now()) return Promise.reject(new SpeechError("the translation model download was rate limited; retry in a minute", 503));
192
194
  const held = this.loaded.get(model);
193
195
  if (held) {
194
196
  held.usedAt = this.now();
@@ -197,6 +199,7 @@ export class Translator {
197
199
  const pair = this.load(model, this.cacheDir).catch((error: unknown) => {
198
200
  // A failed load is tried again next time, not remembered forever.
199
201
  this.loaded.delete(model);
202
+ if (/\b429\b|rate.?limit/i.test(String(error))) this.downloadCooldown.set(model, this.now() + 60_000);
200
203
  throw error;
201
204
  });
202
205
  this.loaded.set(model, { pair, usedAt: this.now() });
@@ -255,7 +258,7 @@ export class Translator {
255
258
  * one account, or too many waiting, each with a status. Empty texts come
256
259
  * back empty and cost nothing.
257
260
  */
258
- async translate(texts: string[], from: string, to: string, options: { by?: string } = {}): Promise<Translated> {
261
+ async translate(texts: string[], from: string, to: string, options: { by?: string; deadline?: number } = {}): Promise<Translated> {
259
262
  const hops = route(from, to, this.pairs);
260
263
  if (hops === null) {
261
264
  const known = Object.keys(LANGUAGES).includes(from) && Object.keys(LANGUAGES).includes(to);
@@ -268,12 +271,15 @@ export class Translator {
268
271
  if (this.waiting >= QUEUE_LIMIT) throw new SpeechError("too much is being translated at once; try again in a moment", 503);
269
272
  this.waiting += 1;
270
273
  const turn = this.tail.then(async () => {
274
+ const fresh = (): void => { if (options.deadline && this.now() > options.deadline) throw new SpeechError("live translation expired while waiting; try the next line", 503); };
275
+ fresh();
271
276
  // Only the lines with something in them go to the model; the rest keep their place.
272
277
  const spoken = texts.map((text) => text.trim());
273
278
  const which = spoken.map((text, i) => (text === "" ? -1 : i)).filter((i) => i >= 0);
274
279
  let current = which.map((i) => spoken[i] as string);
275
280
  for (const [a, b] of hops) {
276
281
  const pair = await this.pair(a, b);
282
+ fresh();
277
283
  current = await pair.translate(current);
278
284
  }
279
285
  const out = [...spoken];
@@ -0,0 +1,46 @@
1
+ /** Acoustic matching only: pitch does not establish a speaker's gender or identity. */
2
+ export type VoiceProfile = "lower" | "higher" | "unknown";
3
+
4
+ /** Median fundamental frequency from periodic voiced frames, sampled at 8 kHz.
5
+ * Ambiguous pitch, overlapping voices and unvoiced sound use the selected default.
6
+ * This is deliberately inexpensive and does not load another model beside ASR.
7
+ */
8
+ export function voiceProfile(pcm: Buffer): VoiceProfile {
9
+ const pitches: number[] = [];
10
+ const frame = 320;
11
+ const samples = Math.floor(pcm.length / 4);
12
+ for (let start = 0; start + frame < samples; start += 1600) {
13
+ const values = new Float32Array(frame);
14
+ let mean = 0;
15
+ for (let i = 0; i < frame; i++) mean += values[i] = pcm.readInt16LE((start + i) * 4) / 32768;
16
+ mean /= frame;
17
+ let energy = 0;
18
+ for (let i = 0; i < frame; i++) { values[i] = (values[i] as number) - mean; energy += (values[i] as number) ** 2; }
19
+ if (Math.sqrt(energy / frame) < 0.01) continue;
20
+ let best = 0;
21
+ let period = 0;
22
+ // Prefer the first strong peak; a later multiple of the period has the same correlation.
23
+ const correlations: number[] = [];
24
+ for (let lag = 20; lag <= 114; lag++) {
25
+ let cross = 0, left = 0, right = 0;
26
+ for (let i = 0; i < frame - lag; i++) {
27
+ const a = values[i] as number, b = values[i + lag] as number;
28
+ cross += a * b; left += a * a; right += b * b;
29
+ }
30
+ correlations[lag] = cross / Math.sqrt(left * right || 1);
31
+ }
32
+ for (let lag = 21; lag < 114; lag++) {
33
+ const score = correlations[lag] as number;
34
+ if (score > 0.8 && score > (correlations[lag - 1] as number) && score >= (correlations[lag + 1] as number) && score > best + 0.03) {
35
+ best = score; period = lag;
36
+ }
37
+ }
38
+ if (period) pitches.push(8000 / period);
39
+ }
40
+ if (pitches.length < 3) return "unknown";
41
+ pitches.sort((a, b) => a - b);
42
+ const median = pitches[Math.floor(pitches.length / 2)] as number;
43
+ const lower = pitches.filter((pitch) => pitch < 155).length / pitches.length;
44
+ const higher = pitches.filter((pitch) => pitch > 185).length / pitches.length;
45
+ return median < 155 && lower >= 0.75 ? "lower" : median > 185 && higher >= 0.75 ? "higher" : "unknown";
46
+ }
package/src/warm.ts CHANGED
@@ -7,15 +7,15 @@
7
7
  * ask after each one, and the first person to speak waited for it. The
8
8
  * models land in NIXAMP_STT_CACHE and the image keeps them.
9
9
  *
10
- * NIXAMP_MT_WARM names the pairs (en-de,en-sv,de-en,sv-en by default):
11
- * German and Swedish both ways, which are the two that were asked for.
10
+ * NIXAMP_MT_WARM names the pairs: German and Swedish both ways with
11
+ * English, and Spanish both ways with English and German by default.
12
12
  * Exits non-zero when anything could not be fetched, so a build does not
13
13
  * quietly ship without its ear.
14
14
  */
15
15
  import { Speech } from "./speech.ts";
16
16
  import { Translator } from "./translate.ts";
17
17
 
18
- const DEFAULT_PAIRS = "en-de,en-sv,de-en,sv-en";
18
+ const DEFAULT_PAIRS = "en-de,en-sv,de-en,sv-en,es-en,en-es,es-de,de-es";
19
19
 
20
20
  async function main(): Promise<number> {
21
21
  const speech = new Speech();
@@ -1 +1 @@
1
- import{t as e}from"./index-DJHHB_63.js";var t=3;async function n(n){let{media:r,src:i,isTv:a}=n,{default:o}=await e(async()=>{let{default:e}=await import(`./hls-n74Cnh8A.js`);return{default:e}},[]);if(!o.isSupported())return n.onError(`This browser cannot play HLS streams.`),{destroy:()=>void 0,levels:()=>[]};let s=new o({...a?{maxBufferLength:60,maxMaxBufferLength:120,backBufferLength:30,liveSyncDurationCount:4}:{backBufferLength:90},enableWorker:!0}),c=0,l=!1;s.on(o.Events.ERROR,(e,r)=>{if(!l&&r.fatal){if(c>=t){n.onError(`This stream kept failing and has been stopped.`),s.destroy();return}switch(c+=1,r.type){case o.ErrorTypes.NETWORK_ERROR:n.onNotice(`Reconnecting…`),s.startLoad();break;case o.ErrorTypes.MEDIA_ERROR:n.onNotice(`Recovering…`),s.recoverMediaError();break;default:n.onError(`This stream could not be played.`),s.destroy()}}}),s.on(o.Events.MANIFEST_PARSED,()=>{l||(n.onNotice(null),n.onReady?.({live:s.levels.length>0&&!Number.isFinite(r.duration),levels:u()}))}),s.on(o.Events.LEVEL_LOADED,(e,t)=>{l||n.onReady?.({live:t.details.live,levels:u()})}),s.on(o.Events.FRAG_BUFFERED,()=>{l||n.onNotice(null)});function u(){return s.levels.map((e,t)=>({index:t,height:e.height||null,bitrate:e.bitrate||null,label:e.height?`${String(e.height)}p`:`${String(Math.round((e.bitrate||0)/1e3))}k`}))}return s.loadSource(i),s.attachMedia(r),{destroy(){l=!0,s.destroy()},levels:u,setLevel(e){s.currentLevel=e},currentLevel:()=>s.autoLevelEnabled?-1:s.currentLevel}}export{n as createHlsEngine};
1
+ import{t as e}from"./index-BzjrTOLf.js";var t=3;async function n(n){let{media:r,src:i,isTv:a}=n,{default:o}=await e(async()=>{let{default:e}=await import(`./hls-n74Cnh8A.js`);return{default:e}},[]);if(!o.isSupported())return n.onError(`This browser cannot play HLS streams.`),{destroy:()=>void 0,levels:()=>[]};let s=new o({...a?{maxBufferLength:60,maxMaxBufferLength:120,backBufferLength:30,liveSyncDurationCount:4}:{backBufferLength:90},enableWorker:!0}),c=0,l=!1;s.on(o.Events.ERROR,(e,r)=>{if(!l&&r.fatal){if(c>=t){n.onError(`This stream kept failing and has been stopped.`),s.destroy();return}switch(c+=1,r.type){case o.ErrorTypes.NETWORK_ERROR:n.onNotice(`Reconnecting…`),s.startLoad();break;case o.ErrorTypes.MEDIA_ERROR:n.onNotice(`Recovering…`),s.recoverMediaError();break;default:n.onError(`This stream could not be played.`),s.destroy()}}}),s.on(o.Events.MANIFEST_PARSED,()=>{l||(n.onNotice(null),n.onReady?.({live:s.levels.length>0&&!Number.isFinite(r.duration),levels:u()}))}),s.on(o.Events.LEVEL_LOADED,(e,t)=>{l||n.onReady?.({live:t.details.live,levels:u()})}),s.on(o.Events.FRAG_BUFFERED,()=>{l||n.onNotice(null)});function u(){return s.levels.map((e,t)=>({index:t,height:e.height||null,bitrate:e.bitrate||null,label:e.height?`${String(e.height)}p`:`${String(Math.round((e.bitrate||0)/1e3))}k`}))}return s.loadSource(i),s.attachMedia(r),{destroy(){l=!0,s.destroy()},levels:u,setLevel(e){s.currentLevel=e},currentLevel:()=>s.autoLevelEnabled?-1:s.currentLevel}}export{n as createHlsEngine};