nixamp 0.23.7 → 0.24.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +79 -3
  2. package/dist/captions.d.ts +57 -10
  3. package/dist/captions.js +253 -29
  4. package/dist/main.js +63 -19
  5. package/dist/mcp.d.ts +7 -1
  6. package/dist/mcp.js +159 -24
  7. package/dist/server.d.ts +11 -0
  8. package/dist/server.js +353 -8
  9. package/dist/speech.d.ts +21 -3
  10. package/dist/speech.js +65 -8
  11. package/dist/transcribe.d.ts +70 -0
  12. package/dist/transcribe.js +337 -41
  13. package/dist/transcript-client.d.ts +81 -0
  14. package/dist/transcript-client.js +76 -0
  15. package/dist/transcript.d.ts +7 -2
  16. package/dist/transcript.js +73 -5
  17. package/dist/transcripts.d.ts +135 -0
  18. package/dist/transcripts.js +384 -0
  19. package/dist/translate-cli.d.ts +16 -0
  20. package/dist/translate-cli.js +97 -0
  21. package/dist/translate-jobs.d.ts +39 -0
  22. package/dist/translate-jobs.js +120 -0
  23. package/dist/translate.d.ts +78 -0
  24. package/dist/translate.js +241 -0
  25. package/dist/warm.d.ts +1 -0
  26. package/dist/warm.js +40 -0
  27. package/package.json +1 -1
  28. package/src/captions.ts +276 -29
  29. package/src/main.ts +63 -19
  30. package/src/mcp.ts +156 -21
  31. package/src/server.ts +352 -8
  32. package/src/speech.ts +103 -13
  33. package/src/transcribe.ts +374 -41
  34. package/src/transcript-client.ts +147 -0
  35. package/src/transcript.ts +76 -4
  36. package/src/transcripts.ts +462 -0
  37. package/src/translate-cli.ts +111 -0
  38. package/src/translate-jobs.ts +132 -0
  39. package/src/translate.ts +276 -0
  40. package/src/warm.ts +43 -0
  41. package/web/dist/assets/{hls-3VKVEQE3-CI1U7kbP.js → hls-3VKVEQE3-Dtl-3mpW.js} +1 -1
  42. package/web/dist/assets/index-B0h4Nexr.js +1 -0
  43. package/web/dist/assets/index-BTGV3Pi5.css +1 -0
  44. package/web/dist/assets/{mpegts-CPOYjgRP.js → mpegts-BJC48bFV.js} +1 -1
  45. package/web/dist/assets/{mpegts-LO6RVLD6-CE8YPjx1.js → mpegts-LO6RVLD6-CUIAB9k3.js} +1 -1
  46. package/web/dist/index.html +12 -8
  47. package/web/dist/sw.js +6 -6
  48. package/web/dist/assets/index-46pwGn5-.css +0 -1
  49. package/web/dist/assets/index-5H3sUGHu.js +0 -1
package/dist/server.js CHANGED
@@ -28,7 +28,7 @@ import { Servers } from "./servers.js";
28
28
  import { DeviceGrants } from "./device.js";
29
29
  import { BAD_KEY_LIMIT, callerOf, Guard, SIGN_IN_LIMIT } from "./guard.js";
30
30
  import { deviceDonePage, devicePage, exchangeCode, providersFrom, signInFailedPage, SignIn, } from "./oauth.js";
31
- import { AuthorizationServer, clientsFrom } from "./oauth-server.js";
31
+ import { AuthorizationServer, SCOPE_NAMES, clientsFrom } from "./oauth-server.js";
32
32
  import { handleOAuthApi, oauthApiPath } from "./oauth-api.js";
33
33
  import { WatchParties } from "./watch-party.js";
34
34
  import { needsAdmin, needsMember, Owner } from "./owner.js";
@@ -67,6 +67,10 @@ import { Rooms } from "./rooms.js";
67
67
  import { Trollbox, TrollboxError, fallbackHandle, roomFor } from "./trollbox.js";
68
68
  import { MAX_BYTES as SPEECH_BYTES, Speech, SpeechError, isWav, languageOf } from "./speech.js";
69
69
  import { Captions } from "./captions.js";
70
+ import { LANGUAGES, Translator } from "./translate.js";
71
+ import { StoredTranslations } from "./translate-jobs.js";
72
+ import { Transcripts, fileFingerprint, formatOf, idFrom, languageCode, linesFrom, mediaOfLive, mediaOfUrl, toSrt, toText, toVtt, transcriptIdOf, wire, } from "./transcripts.js";
73
+ import { handleMessage as mcpMessage } from "./mcp.js";
70
74
  import { Outro } from "./outro.js";
71
75
  import { Profiles, Voices, spokenLine, spokenVoiceFor } from "./voices.js";
72
76
  import { confirm, DEFAULT_DIRECTORY, Publisher } from "./publish.js";
@@ -2350,18 +2354,30 @@ export function createHandler(engine, options) {
2350
2354
  json(response, 400, { error: "a room is a server address and a channel" });
2351
2355
  return;
2352
2356
  }
2353
- const heard = await speech.transcribe(bytes, { language: languageOf(url.searchParams.get("language")), by: who.id });
2357
+ const heard = await speech.transcribe(bytes, {
2358
+ language: languageOf(url.searchParams.get("language")),
2359
+ // Pieces with their timing, for a whole file being written down.
2360
+ timestamps: url.searchParams.get("timestamps") === "1",
2361
+ by: who.id,
2362
+ });
2363
+ // The language travels back: as told, or as the ear guessed it, so
2364
+ // a captioner can say it next time and a transcript can be kept as it.
2365
+ const said = {
2366
+ text: heard.text, seconds: heard.seconds, model: speech.model,
2367
+ ...(heard.language ? { language: heard.language } : {}),
2368
+ ...(heard.segments ? { segments: heard.segments } : {}),
2369
+ };
2354
2370
  if (where && options.trollbox && heard.text !== "") {
2355
2371
  const handle = (await options.handles?.of(who.id)) || fallbackHandle(who.id);
2356
2372
  const line = await options.trollbox.post(where, who.id, handle, heard.text);
2357
2373
  readOnThePhone(where, who.id, line.handle, line.body);
2358
2374
  json(response, 201, {
2359
- text: heard.text, seconds: heard.seconds, model: speech.model,
2375
+ ...said,
2360
2376
  message: { id: line.id, handle: line.handle, body: line.body, createdAt: line.createdAt, mine: true },
2361
2377
  });
2362
2378
  return;
2363
2379
  }
2364
- json(response, 200, { text: heard.text, seconds: heard.seconds, model: speech.model });
2380
+ json(response, 200, said);
2365
2381
  }
2366
2382
  catch (error) {
2367
2383
  if (error instanceof SpeechError || error instanceof TrollboxError)
@@ -2371,6 +2387,282 @@ export function createHandler(engine, options) {
2371
2387
  }
2372
2388
  return;
2373
2389
  }
2390
+ /*
2391
+ * Translation: texts in one language, said in another, by a model on
2392
+ * this CPU (see translate.ts). Signed in only, like the ear. GET says
2393
+ * which languages, and which each can be turned into here.
2394
+ */
2395
+ if (path === "/api/v1/translate" && options.accounts) {
2396
+ const translator = options.translator;
2397
+ if (request.method === "GET") {
2398
+ json(response, 200, {
2399
+ available: translator !== undefined,
2400
+ languages: Object.entries(LANGUAGES).map(([code, names]) => ({ code, ...names, targets: translator ? translator.targets(code) : [] })),
2401
+ });
2402
+ return;
2403
+ }
2404
+ if (request.method !== "POST") {
2405
+ json(response, 405, { error: "POST texts, or GET the languages" });
2406
+ return;
2407
+ }
2408
+ if (!translator) {
2409
+ json(response, 503, { error: "this nixamp cannot translate: the models are not here. nixamp.com can." });
2410
+ return;
2411
+ }
2412
+ const who = await options.accounts.whoIs(tokenFrom(request.headers));
2413
+ if (who === null) {
2414
+ json(response, 401, { error: "sign in to nixamp.com to translate" });
2415
+ return;
2416
+ }
2417
+ let body = {};
2418
+ try {
2419
+ body = JSON.parse(await readBody(request, 256 * 1024));
2420
+ }
2421
+ catch {
2422
+ json(response, 400, { error: "bad JSON" });
2423
+ return;
2424
+ }
2425
+ const texts = Array.isArray(body.texts)
2426
+ ? body.texts.filter((one) => typeof one === "string")
2427
+ : typeof body.text === "string" ? [body.text] : [];
2428
+ if (texts.length === 0) {
2429
+ json(response, 400, { error: "say what: texts, a list of strings" });
2430
+ return;
2431
+ }
2432
+ if (texts.length > 200) {
2433
+ json(response, 413, { error: "200 texts at a time at most" });
2434
+ return;
2435
+ }
2436
+ const from = languageCode(body.from);
2437
+ const to = languageCode(body.to);
2438
+ if (from === null || to === null || to === "") {
2439
+ json(response, 400, { error: "from and to are two-letter language codes, such as en and sv" });
2440
+ return;
2441
+ }
2442
+ if (from === "") {
2443
+ json(response, 400, { error: "say which language the texts are in: from" });
2444
+ return;
2445
+ }
2446
+ try {
2447
+ json(response, 200, await translator.translate(texts, from, to, { by: who.id }));
2448
+ }
2449
+ catch (error) {
2450
+ if (error instanceof SpeechError)
2451
+ json(response, error.status, { error: error.message });
2452
+ else
2453
+ throw error;
2454
+ }
2455
+ return;
2456
+ }
2457
+ /*
2458
+ * The transcript store: what a file, a link or a live said, kept once
2459
+ * under the media's identity (see transcripts.ts). Signed in to read
2460
+ * and to keep, the way the ear is: a captioning server, the CLI, the
2461
+ * MCP tools. A language asks for a translation, made before answering
2462
+ * when it is short and as a job when it is not; a format asks for it
2463
+ * as subtitles.
2464
+ */
2465
+ if ((path === "/api/v1/transcripts" || path.startsWith("/api/v1/transcripts/")) && options.transcripts && options.translations && options.accounts) {
2466
+ const store = options.transcripts;
2467
+ const parts = path.slice("/api/v1/transcripts".length).split("/").filter(Boolean);
2468
+ const who = await options.accounts.whoIs(tokenFrom(request.headers));
2469
+ if (who === null) {
2470
+ json(response, 401, { error: "sign in to nixamp.com to read or keep transcripts" });
2471
+ return;
2472
+ }
2473
+ try {
2474
+ if (parts.length === 0) {
2475
+ if (request.method !== "GET") {
2476
+ json(response, 405, { error: "GET: what you have had written down" });
2477
+ return;
2478
+ }
2479
+ json(response, 200, { transcripts: await store.list(who.id, Number(url.searchParams.get("limit") ?? 50) || 50) });
2480
+ return;
2481
+ }
2482
+ let named = "";
2483
+ try {
2484
+ named = decodeURIComponent(parts[0]);
2485
+ }
2486
+ catch {
2487
+ json(response, 400, { error: "bad id" });
2488
+ return;
2489
+ }
2490
+ // By id, or by the media identity itself: both name the same row.
2491
+ const id = idFrom(named);
2492
+ const action = parts[1];
2493
+ if (action === undefined && request.method === "GET") {
2494
+ const language = languageCode(url.searchParams.get("language"));
2495
+ const format = formatOf(url.searchParams.get("format"));
2496
+ if (language === null) {
2497
+ json(response, 400, { error: "language is a two-letter code, or original" });
2498
+ return;
2499
+ }
2500
+ if (format === null) {
2501
+ json(response, 400, { error: "format is json, srt, vtt or txt" });
2502
+ return;
2503
+ }
2504
+ const answer = await options.translations.get(id, language, who.id);
2505
+ if (answer.status !== 200 && answer.status !== 202) {
2506
+ json(response, answer.status, { error: answer.error });
2507
+ return;
2508
+ }
2509
+ if (answer.status === 202) {
2510
+ json(response, 202, {
2511
+ ...(answer.transcript ? wire(answer.transcript, await store.languages(id)) : { id, language, lines: [], languages: await store.languages(id) }),
2512
+ translating: answer.translating,
2513
+ });
2514
+ return;
2515
+ }
2516
+ const transcript = answer.transcript;
2517
+ if (format === "json") {
2518
+ json(response, 200, wire(transcript, await store.languages(id)));
2519
+ return;
2520
+ }
2521
+ const text = format === "srt" ? toSrt(transcript.lines) : format === "vtt" ? toVtt(transcript.lines) : toText(transcript.lines);
2522
+ const name = (transcript.title || id).replace(/[^\w.-]+/g, "_").slice(0, 60) || "transcript";
2523
+ response.writeHead(200, {
2524
+ ...CORS,
2525
+ "content-type": `${format === "vtt" ? "text/vtt" : "text/plain"}; charset=utf-8`,
2526
+ "content-length": Buffer.byteLength(text),
2527
+ "cache-control": "no-store",
2528
+ "content-disposition": `inline; filename="${name}${transcript.language ? `.${transcript.language}` : ""}.${format}"`,
2529
+ });
2530
+ response.end(text);
2531
+ return;
2532
+ }
2533
+ if (action === "lines" && request.method === "POST") {
2534
+ let body = {};
2535
+ try {
2536
+ body = JSON.parse(await readBody(request, 4 * 1024 * 1024));
2537
+ }
2538
+ catch {
2539
+ json(response, 400, { error: "bad JSON, or more than 4 MB of it" });
2540
+ return;
2541
+ }
2542
+ const media = typeof body.media === "string" ? body.media.trim() : "";
2543
+ if (media === "" || media.length > 2048 || !/^(file|url|live):/.test(media)) {
2544
+ json(response, 400, { error: "media is the identity: file:v1:<hash>, url:<address> or live:<server>/<channel>@<started>" });
2545
+ return;
2546
+ }
2547
+ if (transcriptIdOf(media) !== id) {
2548
+ json(response, 400, { error: "that media is not this transcript's id" });
2549
+ return;
2550
+ }
2551
+ const language = languageCode(body.language);
2552
+ const translatedFrom = body.translatedFrom === undefined || body.translatedFrom === null ? "" : languageCode(body.translatedFrom);
2553
+ if (language === null || translatedFrom === null) {
2554
+ json(response, 400, { error: "language and translatedFrom are two-letter codes" });
2555
+ return;
2556
+ }
2557
+ const lines = linesFrom(body.lines);
2558
+ if (lines.length === 0 && body.complete !== true) {
2559
+ json(response, 400, { error: "lines: a list of {start, end, text}, seconds into the media" });
2560
+ return;
2561
+ }
2562
+ const saved = await store.save({
2563
+ media,
2564
+ language,
2565
+ translatedFrom: translatedFrom === "" ? null : translatedFrom,
2566
+ model: typeof body.model === "string" ? body.model.slice(0, 100) : "",
2567
+ title: typeof body.title === "string" ? body.title : "",
2568
+ complete: body.complete === true,
2569
+ by: who.id,
2570
+ lines,
2571
+ });
2572
+ json(response, 200, { id, language: saved?.language ?? language, saved: lines.length, seconds: saved?.seconds ?? 0, complete: saved?.complete ?? false });
2573
+ return;
2574
+ }
2575
+ if (action === undefined && request.method === "DELETE") {
2576
+ const gone = await store.forget(id, who.id);
2577
+ json(response, gone ? 200 : 404, gone ? { ok: true } : { error: "nothing of yours under that id" });
2578
+ return;
2579
+ }
2580
+ json(response, 405, { error: "GET a transcript, POST its lines, or DELETE it" });
2581
+ }
2582
+ catch (error) {
2583
+ if (error instanceof SpeechError)
2584
+ json(response, error.status, { error: error.message });
2585
+ else
2586
+ throw error;
2587
+ }
2588
+ return;
2589
+ }
2590
+ /*
2591
+ * MCP over HTTP: the tools `nixamp mcp` offers on a pipe, at
2592
+ * nixamp.com/mcp for an agent with no nixamp installed. A JSON-RPC
2593
+ * message, or a batch of them, in a POST; the answers as JSON. The
2594
+ * bearer token says whose tools they are, as on every other route, and
2595
+ * the tools call nixamp.com as that account. No stream is offered on
2596
+ * GET: nothing here has to be pushed, so a client is told so and gets
2597
+ * on with POST.
2598
+ */
2599
+ if (path === "/mcp" && options.accounts && options.site) {
2600
+ if (request.method === "GET") {
2601
+ response.writeHead(405, { ...CORS, allow: "POST, DELETE" });
2602
+ response.end();
2603
+ return;
2604
+ }
2605
+ if (request.method === "DELETE") {
2606
+ response.writeHead(204, CORS);
2607
+ response.end();
2608
+ return;
2609
+ }
2610
+ if (request.method !== "POST") {
2611
+ json(response, 405, { error: "POST a JSON-RPC message" });
2612
+ return;
2613
+ }
2614
+ const token = tokenFrom(request.headers);
2615
+ const who = await options.accounts.whoIs(token);
2616
+ if (who === null) {
2617
+ const error = JSON.stringify({ error: "sign in: a nixamp token as the bearer (`nixamp token create`), or OAuth" });
2618
+ response.writeHead(401, {
2619
+ ...CORS,
2620
+ "content-type": "application/json; charset=utf-8",
2621
+ "content-length": Buffer.byteLength(error),
2622
+ "www-authenticate": `Bearer resource_metadata="${options.site}/.well-known/oauth-protected-resource"`,
2623
+ });
2624
+ response.end(error);
2625
+ return;
2626
+ }
2627
+ let parsed;
2628
+ try {
2629
+ parsed = JSON.parse(await readBody(request, 1024 * 1024));
2630
+ }
2631
+ catch {
2632
+ json(response, 400, { jsonrpc: "2.0", id: null, error: { code: -32700, message: "parse error" } });
2633
+ return;
2634
+ }
2635
+ const messages = Array.isArray(parsed) ? parsed : [parsed];
2636
+ const session = { site: options.site, token };
2637
+ const answers = [];
2638
+ for (const message of messages) {
2639
+ if (!message || typeof message !== "object" || typeof message.method !== "string") {
2640
+ answers.push({ jsonrpc: "2.0", id: null, error: { code: -32600, message: "invalid request" } });
2641
+ continue;
2642
+ }
2643
+ const answer = await mcpMessage(message, { session, fetcher: fetch });
2644
+ if (answer)
2645
+ answers.push(answer);
2646
+ }
2647
+ if (answers.length === 0) {
2648
+ response.writeHead(202, CORS);
2649
+ response.end();
2650
+ return;
2651
+ }
2652
+ json(response, 200, Array.isArray(parsed) ? answers : answers[0]);
2653
+ return;
2654
+ }
2655
+ /* Where a client of /mcp finds the authorization server (RFC 9728). */
2656
+ if (path === "/.well-known/oauth-protected-resource" && options.site && options.authServer) {
2657
+ json(response, 200, {
2658
+ resource: options.site,
2659
+ authorization_servers: [options.site],
2660
+ bearer_methods_supported: ["header"],
2661
+ scopes_supported: SCOPE_NAMES,
2662
+ resource_name: "nixamp",
2663
+ });
2664
+ return;
2665
+ }
2374
2666
  if (path === "/api/v1/me/handle" && options.handles && options.accounts) {
2375
2667
  const handles = options.handles;
2376
2668
  const who = await options.accounts.whoIs(tokenFrom(request.headers));
@@ -3537,11 +3829,17 @@ export function createHandler(engine, options) {
3537
3829
  // reconnecting EventSource saw, which the browser sends by itself.
3538
3830
  const lastSeen = request.headers["last-event-id"];
3539
3831
  const after = Number(url.searchParams.get("after") ?? (Array.isArray(lastSeen) ? lastSeen[0] : lastSeen) ?? 0) || 0;
3832
+ // The lines in another language, translated as they are heard; "" is what was said.
3833
+ const wanted = languageCode(url.searchParams.get("language"));
3834
+ if (wanted === null) {
3835
+ json(response, 400, { error: "language is a two-letter code, such as de or sv" });
3836
+ return;
3837
+ }
3540
3838
  if (action === "transcript") {
3541
3839
  // Asking keeps the captioner up: it stops a minute after the last ask.
3542
- captions.subscribe(id, () => undefined)?.();
3840
+ captions.subscribe(id, () => undefined, wanted)?.();
3543
3841
  json(response, 200, {
3544
- channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), ...captions.status(id), lines: captions.recent(id, after),
3842
+ channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted),
3545
3843
  });
3546
3844
  return;
3547
3845
  }
@@ -3555,14 +3853,14 @@ export function createHandler(engine, options) {
3555
3853
  const write = (event, data, eventId) => {
3556
3854
  response.write(`${eventId === undefined ? "" : `id: ${eventId}\n`}event: ${event}\ndata: ${JSON.stringify(data)}\n\n`);
3557
3855
  };
3558
- const off = captions.subscribe(id, (line) => write("line", line, line.at));
3856
+ const off = captions.subscribe(id, (line) => write("line", line, line.at), wanted);
3559
3857
  if (off === null) {
3560
3858
  response.end();
3561
3859
  return;
3562
3860
  }
3563
3861
  // How far behind the live edge a newcomer's playback starts, so the
3564
3862
  // page can hold each line until its own sound gets there.
3565
- write("hello", { channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), ...captions.status(id), lines: captions.recent(id, after) });
3863
+ write("hello", { channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted) });
3566
3864
  const beat = setInterval(() => response.write(": beat\n\n"), 20_000);
3567
3865
  beat.unref?.();
3568
3866
  const done = () => {
@@ -4864,6 +5162,22 @@ export async function serve(argv, version = "0.1.0") {
4864
5162
  console.error(ready ? `nixamp: hearing with ${speech.model}` : `nixamp: not hearing: ${speech.lastFailure}`);
4865
5163
  });
4866
5164
  }
5165
+ // Translation lives beside the ear: the same optional library, a model
5166
+ // per pair of languages, loaded when first asked for. NIXAMP_MT_WARM
5167
+ // names pairs to load at boot (en-de,en-sv) so the first line does not
5168
+ // wait; NIXAMP_MT=off turns it off.
5169
+ const translator = accounts && process.env["NIXAMP_MT"] !== "off" ? new Translator() : undefined;
5170
+ if (translator) {
5171
+ const warm = (process.env["NIXAMP_MT_WARM"] ?? "").split(",").map((one) => one.trim()).filter(Boolean);
5172
+ if (warm.length > 0) {
5173
+ void translator.warm(warm).then((ready) => {
5174
+ console.error(ready ? `nixamp: translating (${warm.join(", ")} loaded)` : `nixamp: not translating everything: ${translator.lastFailure}`);
5175
+ });
5176
+ }
5177
+ }
5178
+ // What was heard, kept: where the accounts are, under the media's identity.
5179
+ const transcripts = pool ? new Transcripts(pool, (message) => console.log(message)) : undefined;
5180
+ const translations = transcripts ? new StoredTranslations(transcripts, translator, (message) => console.log(message)) : undefined;
4867
5181
  // nixamp as an OAuth 2.1 authorization server, and the watch parties a
4868
5182
  // client site bridges through it. Both need the same three things -- a
4869
5183
  // database, accounts, and a site to be the issuer of -- so both appear or
@@ -5051,6 +5365,33 @@ export async function serve(argv, version = "0.1.0") {
5051
5365
  listen: (id, listener) => channels.listen(id, listener),
5052
5366
  session: () => readSession(),
5053
5367
  onEvent: (message) => console.log(` ${message}`),
5368
+ // What a channel is playing, for the transcript store: a film by
5369
+ // its bytes and where it has got to, a link by its address, and
5370
+ // anything live as the broadcast it is.
5371
+ mediaOf: (id) => {
5372
+ const info = channels.info(id);
5373
+ if (!info)
5374
+ return null;
5375
+ const base = { title: info.name, startedAt: info.startedAt, backlog: info.codecs?.video ? BACKLOG_SECONDS : 3 };
5376
+ if (info.via === "pull" && !info.live && typeof info.position === "number" && info.source) {
5377
+ if (/^https?:\/\//.test(info.source)) {
5378
+ // The pasted address while it is still known: it outlives a media URL with a token in it.
5379
+ let pasted = info.source;
5380
+ for (const [link] of links)
5381
+ if (linkChannelId(link) === id)
5382
+ pasted = link;
5383
+ return { ...base, media: mediaOfUrl(pasted), position: info.position };
5384
+ }
5385
+ try {
5386
+ return { ...base, media: fileFingerprint(info.source), position: info.position };
5387
+ }
5388
+ catch {
5389
+ return null;
5390
+ }
5391
+ }
5392
+ const self = publishable_?.url || options.publicUrl || `${options.host}:${options.port}`;
5393
+ return { ...base, media: mediaOfLive(self, id, info.startedAt) };
5394
+ },
5054
5395
  })
5055
5396
  : undefined;
5056
5397
  const owner = new Owner({
@@ -5254,6 +5595,10 @@ export async function serve(argv, version = "0.1.0") {
5254
5595
  ...(rooms ? { rooms } : {}),
5255
5596
  ...(trollbox ? { trollbox } : {}),
5256
5597
  ...(speech ? { speech } : {}),
5598
+ ...(translator ? { translator } : {}),
5599
+ ...(transcripts ? { transcripts } : {}),
5600
+ ...(translations ? { translations } : {}),
5601
+ site: nixampSite,
5257
5602
  // Other people's OpenProfiles, read for the voice a line is spoken in,
5258
5603
  // and the voices to speak in: ElevenLabs when the Telnyx account holds
5259
5604
  // the key for it (an integration secret named "elevenlabs"), Kokoro
package/dist/speech.d.ts CHANGED
@@ -20,16 +20,31 @@ export declare class SpeechError extends Error {
20
20
  readonly status: number;
21
21
  constructor(message: string, status: number);
22
22
  }
23
+ /** A stretch of the clip and what was said in it, seconds from the clip's start. */
24
+ export interface Segment {
25
+ start: number;
26
+ end: number;
27
+ text: string;
28
+ }
23
29
  export interface Heard {
24
30
  text: string;
25
31
  /** How long the audio was, in seconds. */
26
32
  seconds: number;
33
+ /** The language heard, as told or as guessed; absent when neither was possible. */
34
+ language?: string;
35
+ /** The clip in pieces with their timing, when asked for. */
36
+ segments?: Segment[];
27
37
  }
28
- /** Mono samples in [-1, 1] at RATE -> words. What the loaded model is, to this file. */
29
- export type Recognizer = (pcm: Float32Array, options: {
38
+ export interface RecognizeOptions {
30
39
  language?: string;
31
- }) => Promise<{
40
+ /** Also say when each piece was said. Slower: the ear takes about twice as long. */
41
+ timestamps?: boolean;
42
+ }
43
+ /** Mono samples in [-1, 1] at RATE -> words. What the loaded model is, to this file. */
44
+ export type Recognizer = (pcm: Float32Array, options: RecognizeOptions) => Promise<{
32
45
  text: string;
46
+ language?: string;
47
+ segments?: Segment[];
33
48
  }>;
34
49
  export interface SpeechOptions {
35
50
  /** A Hugging Face model id; NIXAMP_STT_MODEL otherwise; whisper-base by default. */
@@ -66,6 +81,8 @@ export declare function encodeWav(samples: Float32Array, rate?: number): Uint8Ar
66
81
  export declare function tidy(text: string): string;
67
82
  /** A two-letter language code, or nothing: Whisper guesses when not told. */
68
83
  export declare function languageOf(value: unknown): string | undefined;
84
+ /** Load the library, or say why not with a status. */
85
+ export declare function loadTransformers<T>(): Promise<T>;
69
86
  export declare class Speech {
70
87
  readonly model: string;
71
88
  private readonly cacheDir;
@@ -93,6 +110,7 @@ export declare class Speech {
93
110
  */
94
111
  transcribe(bytes: Uint8Array, options?: {
95
112
  language?: string;
113
+ timestamps?: boolean;
96
114
  by?: string;
97
115
  }): Promise<Heard>;
98
116
  }
package/dist/speech.js CHANGED
@@ -196,13 +196,13 @@ export function languageOf(value) {
196
196
  const code = value.trim().toLowerCase();
197
197
  return /^[a-z]{2}$/.test(code) ? code : undefined;
198
198
  }
199
- async function loadWhisper(model, cacheDir) {
199
+ /** Load the library, or say why not with a status. */
200
+ export async function loadTransformers() {
200
201
  // By name held in a variable: an optional dependency that is not on disk
201
202
  // must fail here, at the first ask, and not when the file is imported.
202
203
  const name = "@huggingface/transformers";
203
- let transformers;
204
204
  try {
205
- transformers = (await import(__rewriteRelativeImportExtension(name)));
205
+ return (await import(__rewriteRelativeImportExtension(name)));
206
206
  }
207
207
  catch (error) {
208
208
  // The reason travels with the refusal: "not installed" and "installed
@@ -211,16 +211,62 @@ async function loadWhisper(model, cacheDir) {
211
211
  const why = String(error.message ?? error).split("\n")[0] ?? "";
212
212
  throw new SpeechError(`this nixamp cannot hear: @huggingface/transformers did not load here (${why}). nixamp.com can.`, 503);
213
213
  }
214
+ }
215
+ /**
216
+ * Which language a clip is in, the way whisper.cpp asks: the encoder once
217
+ * over the first thirty seconds, one decoder step from the start token,
218
+ * and the loudest of the language tokens wins. Whisper told nothing
219
+ * assumes English, and a Swedish sentence heard as English comes back as
220
+ * the same three English words repeated to the end of the clip; guessing
221
+ * first costs an encoder pass and is what makes a caption in Swedish say
222
+ * something. Undefined for a model that knows one language.
223
+ */
224
+ async function detectLanguage(transformers, asr, pcm) {
225
+ const config = asr.model.generation_config;
226
+ const languages = config.lang_to_id;
227
+ if (!languages || config.is_multilingual === false)
228
+ return undefined;
229
+ const inputs = await asr.processor(pcm.subarray(0, 30 * RATE));
230
+ const start = new transformers.Tensor("int64", BigInt64Array.from([BigInt(config.decoder_start_token_id)]), [1, 1]);
231
+ const out = await asr.model({ input_features: inputs.input_features, decoder_input_ids: start });
232
+ const logits = out.logits.data;
233
+ let best;
234
+ let loudest = -Infinity;
235
+ for (const [token, id] of Object.entries(languages)) {
236
+ const score = logits[id];
237
+ if (typeof score === "number" && score > loudest) {
238
+ loudest = score;
239
+ best = token;
240
+ }
241
+ }
242
+ // The token is written <|sv|>.
243
+ const code = best?.replace(/^<\|/, "").replace(/\|>$/, "");
244
+ return code && /^[a-z]{2,3}$/.test(code) ? code : undefined;
245
+ }
246
+ async function loadWhisper(model, cacheDir) {
247
+ const transformers = await loadTransformers();
214
248
  transformers.env.cacheDir = cacheDir;
215
249
  const recognize = await transformers.pipeline("automatic-speech-recognition", model, { dtype: "q8" });
216
- return async (pcm, { language }) => {
250
+ return async (pcm, { language, timestamps }) => {
251
+ const spoken = language ?? (await detectLanguage(transformers, recognize, pcm));
217
252
  const heard = await recognize(pcm, {
218
253
  // Whisper hears thirty seconds at a time; longer is heard in overlapping pieces.
219
254
  chunk_length_s: 30,
220
255
  stride_length_s: 5,
221
- ...(language ? { language, task: "transcribe" } : {}),
256
+ ...(spoken ? { language: spoken, task: "transcribe" } : {}),
257
+ ...(timestamps ? { return_timestamps: true } : {}),
222
258
  });
223
- return { text: Array.isArray(heard) ? heard.map((piece) => piece.text).join(" ") : heard.text };
259
+ const pieces = Array.isArray(heard) ? heard : [heard];
260
+ const text = pieces.map((piece) => piece.text).join(" ");
261
+ const seconds = pcm.length / RATE;
262
+ const segments = timestamps
263
+ ? pieces.flatMap((piece) => piece.chunks ?? []).map((chunk) => ({
264
+ start: Math.max(0, chunk.timestamp[0] ?? 0),
265
+ end: Math.min(seconds, chunk.timestamp[1] ?? seconds),
266
+ text: chunk.text,
267
+ }))
268
+ : undefined;
269
+ return { text, ...(spoken ? { language: spoken } : {}), ...(segments ? { segments } : {}) };
224
270
  };
225
271
  }
226
272
  export class Speech {
@@ -303,13 +349,24 @@ export class Speech {
303
349
  this.waiting += 1;
304
350
  const turn = this.tail.then(async () => {
305
351
  const recognize = await this.ear();
306
- return recognize(pcm, options.language ? { language: options.language } : {});
352
+ return recognize(pcm, {
353
+ ...(options.language ? { language: options.language } : {}),
354
+ ...(options.timestamps ? { timestamps: true } : {}),
355
+ });
307
356
  });
308
357
  // The queue moves on whether or not this one was heard.
309
358
  this.tail = turn.catch(() => undefined);
310
359
  try {
311
360
  const heard = await turn;
312
- return { text: tidy(heard.text), seconds };
361
+ const segments = heard.segments
362
+ ?.map((segment) => ({ start: segment.start, end: segment.end, text: tidy(segment.text) }))
363
+ .filter((segment) => segment.text !== "");
364
+ return {
365
+ text: tidy(heard.text),
366
+ seconds,
367
+ ...(heard.language ? { language: heard.language } : {}),
368
+ ...(segments ? { segments } : {}),
369
+ };
313
370
  }
314
371
  finally {
315
372
  this.waiting -= 1;