nixamp 0.23.7 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +79 -3
- package/dist/captions.d.ts +57 -10
- package/dist/captions.js +253 -29
- package/dist/main.js +63 -19
- package/dist/mcp.d.ts +7 -1
- package/dist/mcp.js +159 -24
- package/dist/server.d.ts +11 -0
- package/dist/server.js +353 -8
- package/dist/speech.d.ts +21 -3
- package/dist/speech.js +65 -8
- package/dist/transcribe.d.ts +70 -0
- package/dist/transcribe.js +337 -41
- package/dist/transcript-client.d.ts +81 -0
- package/dist/transcript-client.js +76 -0
- package/dist/transcript.d.ts +7 -2
- package/dist/transcript.js +73 -5
- package/dist/transcripts.d.ts +135 -0
- package/dist/transcripts.js +384 -0
- package/dist/translate-cli.d.ts +16 -0
- package/dist/translate-cli.js +97 -0
- package/dist/translate-jobs.d.ts +39 -0
- package/dist/translate-jobs.js +120 -0
- package/dist/translate.d.ts +78 -0
- package/dist/translate.js +241 -0
- package/dist/warm.d.ts +1 -0
- package/dist/warm.js +40 -0
- package/package.json +1 -1
- package/src/captions.ts +276 -29
- package/src/main.ts +63 -19
- package/src/mcp.ts +156 -21
- package/src/server.ts +352 -8
- package/src/speech.ts +103 -13
- package/src/transcribe.ts +374 -41
- package/src/transcript-client.ts +147 -0
- package/src/transcript.ts +76 -4
- package/src/transcripts.ts +462 -0
- package/src/translate-cli.ts +111 -0
- package/src/translate-jobs.ts +132 -0
- package/src/translate.ts +276 -0
- package/src/warm.ts +43 -0
- package/web/dist/assets/{hls-3VKVEQE3-CI1U7kbP.js → hls-3VKVEQE3-Dtl-3mpW.js} +1 -1
- package/web/dist/assets/index-B0h4Nexr.js +1 -0
- package/web/dist/assets/index-BTGV3Pi5.css +1 -0
- package/web/dist/assets/{mpegts-CPOYjgRP.js → mpegts-BJC48bFV.js} +1 -1
- package/web/dist/assets/{mpegts-LO6RVLD6-CE8YPjx1.js → mpegts-LO6RVLD6-CUIAB9k3.js} +1 -1
- package/web/dist/index.html +12 -8
- package/web/dist/sw.js +6 -6
- package/web/dist/assets/index-46pwGn5-.css +0 -1
- package/web/dist/assets/index-5H3sUGHu.js +0 -1
package/dist/server.js
CHANGED
|
@@ -28,7 +28,7 @@ import { Servers } from "./servers.js";
|
|
|
28
28
|
import { DeviceGrants } from "./device.js";
|
|
29
29
|
import { BAD_KEY_LIMIT, callerOf, Guard, SIGN_IN_LIMIT } from "./guard.js";
|
|
30
30
|
import { deviceDonePage, devicePage, exchangeCode, providersFrom, signInFailedPage, SignIn, } from "./oauth.js";
|
|
31
|
-
import { AuthorizationServer, clientsFrom } from "./oauth-server.js";
|
|
31
|
+
import { AuthorizationServer, SCOPE_NAMES, clientsFrom } from "./oauth-server.js";
|
|
32
32
|
import { handleOAuthApi, oauthApiPath } from "./oauth-api.js";
|
|
33
33
|
import { WatchParties } from "./watch-party.js";
|
|
34
34
|
import { needsAdmin, needsMember, Owner } from "./owner.js";
|
|
@@ -67,6 +67,10 @@ import { Rooms } from "./rooms.js";
|
|
|
67
67
|
import { Trollbox, TrollboxError, fallbackHandle, roomFor } from "./trollbox.js";
|
|
68
68
|
import { MAX_BYTES as SPEECH_BYTES, Speech, SpeechError, isWav, languageOf } from "./speech.js";
|
|
69
69
|
import { Captions } from "./captions.js";
|
|
70
|
+
import { LANGUAGES, Translator } from "./translate.js";
|
|
71
|
+
import { StoredTranslations } from "./translate-jobs.js";
|
|
72
|
+
import { Transcripts, fileFingerprint, formatOf, idFrom, languageCode, linesFrom, mediaOfLive, mediaOfUrl, toSrt, toText, toVtt, transcriptIdOf, wire, } from "./transcripts.js";
|
|
73
|
+
import { handleMessage as mcpMessage } from "./mcp.js";
|
|
70
74
|
import { Outro } from "./outro.js";
|
|
71
75
|
import { Profiles, Voices, spokenLine, spokenVoiceFor } from "./voices.js";
|
|
72
76
|
import { confirm, DEFAULT_DIRECTORY, Publisher } from "./publish.js";
|
|
@@ -2350,18 +2354,30 @@ export function createHandler(engine, options) {
|
|
|
2350
2354
|
json(response, 400, { error: "a room is a server address and a channel" });
|
|
2351
2355
|
return;
|
|
2352
2356
|
}
|
|
2353
|
-
const heard = await speech.transcribe(bytes, {
|
|
2357
|
+
const heard = await speech.transcribe(bytes, {
|
|
2358
|
+
language: languageOf(url.searchParams.get("language")),
|
|
2359
|
+
// Pieces with their timing, for a whole file being written down.
|
|
2360
|
+
timestamps: url.searchParams.get("timestamps") === "1",
|
|
2361
|
+
by: who.id,
|
|
2362
|
+
});
|
|
2363
|
+
// The language travels back: as told, or as the ear guessed it, so
|
|
2364
|
+
// a captioner can say it next time and a transcript can be kept as it.
|
|
2365
|
+
const said = {
|
|
2366
|
+
text: heard.text, seconds: heard.seconds, model: speech.model,
|
|
2367
|
+
...(heard.language ? { language: heard.language } : {}),
|
|
2368
|
+
...(heard.segments ? { segments: heard.segments } : {}),
|
|
2369
|
+
};
|
|
2354
2370
|
if (where && options.trollbox && heard.text !== "") {
|
|
2355
2371
|
const handle = (await options.handles?.of(who.id)) || fallbackHandle(who.id);
|
|
2356
2372
|
const line = await options.trollbox.post(where, who.id, handle, heard.text);
|
|
2357
2373
|
readOnThePhone(where, who.id, line.handle, line.body);
|
|
2358
2374
|
json(response, 201, {
|
|
2359
|
-
|
|
2375
|
+
...said,
|
|
2360
2376
|
message: { id: line.id, handle: line.handle, body: line.body, createdAt: line.createdAt, mine: true },
|
|
2361
2377
|
});
|
|
2362
2378
|
return;
|
|
2363
2379
|
}
|
|
2364
|
-
json(response, 200,
|
|
2380
|
+
json(response, 200, said);
|
|
2365
2381
|
}
|
|
2366
2382
|
catch (error) {
|
|
2367
2383
|
if (error instanceof SpeechError || error instanceof TrollboxError)
|
|
@@ -2371,6 +2387,282 @@ export function createHandler(engine, options) {
|
|
|
2371
2387
|
}
|
|
2372
2388
|
return;
|
|
2373
2389
|
}
|
|
2390
|
+
/*
|
|
2391
|
+
* Translation: texts in one language, said in another, by a model on
|
|
2392
|
+
* this CPU (see translate.ts). Signed in only, like the ear. GET says
|
|
2393
|
+
* which languages, and which each can be turned into here.
|
|
2394
|
+
*/
|
|
2395
|
+
if (path === "/api/v1/translate" && options.accounts) {
|
|
2396
|
+
const translator = options.translator;
|
|
2397
|
+
if (request.method === "GET") {
|
|
2398
|
+
json(response, 200, {
|
|
2399
|
+
available: translator !== undefined,
|
|
2400
|
+
languages: Object.entries(LANGUAGES).map(([code, names]) => ({ code, ...names, targets: translator ? translator.targets(code) : [] })),
|
|
2401
|
+
});
|
|
2402
|
+
return;
|
|
2403
|
+
}
|
|
2404
|
+
if (request.method !== "POST") {
|
|
2405
|
+
json(response, 405, { error: "POST texts, or GET the languages" });
|
|
2406
|
+
return;
|
|
2407
|
+
}
|
|
2408
|
+
if (!translator) {
|
|
2409
|
+
json(response, 503, { error: "this nixamp cannot translate: the models are not here. nixamp.com can." });
|
|
2410
|
+
return;
|
|
2411
|
+
}
|
|
2412
|
+
const who = await options.accounts.whoIs(tokenFrom(request.headers));
|
|
2413
|
+
if (who === null) {
|
|
2414
|
+
json(response, 401, { error: "sign in to nixamp.com to translate" });
|
|
2415
|
+
return;
|
|
2416
|
+
}
|
|
2417
|
+
let body = {};
|
|
2418
|
+
try {
|
|
2419
|
+
body = JSON.parse(await readBody(request, 256 * 1024));
|
|
2420
|
+
}
|
|
2421
|
+
catch {
|
|
2422
|
+
json(response, 400, { error: "bad JSON" });
|
|
2423
|
+
return;
|
|
2424
|
+
}
|
|
2425
|
+
const texts = Array.isArray(body.texts)
|
|
2426
|
+
? body.texts.filter((one) => typeof one === "string")
|
|
2427
|
+
: typeof body.text === "string" ? [body.text] : [];
|
|
2428
|
+
if (texts.length === 0) {
|
|
2429
|
+
json(response, 400, { error: "say what: texts, a list of strings" });
|
|
2430
|
+
return;
|
|
2431
|
+
}
|
|
2432
|
+
if (texts.length > 200) {
|
|
2433
|
+
json(response, 413, { error: "200 texts at a time at most" });
|
|
2434
|
+
return;
|
|
2435
|
+
}
|
|
2436
|
+
const from = languageCode(body.from);
|
|
2437
|
+
const to = languageCode(body.to);
|
|
2438
|
+
if (from === null || to === null || to === "") {
|
|
2439
|
+
json(response, 400, { error: "from and to are two-letter language codes, such as en and sv" });
|
|
2440
|
+
return;
|
|
2441
|
+
}
|
|
2442
|
+
if (from === "") {
|
|
2443
|
+
json(response, 400, { error: "say which language the texts are in: from" });
|
|
2444
|
+
return;
|
|
2445
|
+
}
|
|
2446
|
+
try {
|
|
2447
|
+
json(response, 200, await translator.translate(texts, from, to, { by: who.id }));
|
|
2448
|
+
}
|
|
2449
|
+
catch (error) {
|
|
2450
|
+
if (error instanceof SpeechError)
|
|
2451
|
+
json(response, error.status, { error: error.message });
|
|
2452
|
+
else
|
|
2453
|
+
throw error;
|
|
2454
|
+
}
|
|
2455
|
+
return;
|
|
2456
|
+
}
|
|
2457
|
+
/*
|
|
2458
|
+
* The transcript store: what a file, a link or a live said, kept once
|
|
2459
|
+
* under the media's identity (see transcripts.ts). Signed in to read
|
|
2460
|
+
* and to keep, the way the ear is: a captioning server, the CLI, the
|
|
2461
|
+
* MCP tools. A language asks for a translation, made before answering
|
|
2462
|
+
* when it is short and as a job when it is not; a format asks for it
|
|
2463
|
+
* as subtitles.
|
|
2464
|
+
*/
|
|
2465
|
+
if ((path === "/api/v1/transcripts" || path.startsWith("/api/v1/transcripts/")) && options.transcripts && options.translations && options.accounts) {
|
|
2466
|
+
const store = options.transcripts;
|
|
2467
|
+
const parts = path.slice("/api/v1/transcripts".length).split("/").filter(Boolean);
|
|
2468
|
+
const who = await options.accounts.whoIs(tokenFrom(request.headers));
|
|
2469
|
+
if (who === null) {
|
|
2470
|
+
json(response, 401, { error: "sign in to nixamp.com to read or keep transcripts" });
|
|
2471
|
+
return;
|
|
2472
|
+
}
|
|
2473
|
+
try {
|
|
2474
|
+
if (parts.length === 0) {
|
|
2475
|
+
if (request.method !== "GET") {
|
|
2476
|
+
json(response, 405, { error: "GET: what you have had written down" });
|
|
2477
|
+
return;
|
|
2478
|
+
}
|
|
2479
|
+
json(response, 200, { transcripts: await store.list(who.id, Number(url.searchParams.get("limit") ?? 50) || 50) });
|
|
2480
|
+
return;
|
|
2481
|
+
}
|
|
2482
|
+
let named = "";
|
|
2483
|
+
try {
|
|
2484
|
+
named = decodeURIComponent(parts[0]);
|
|
2485
|
+
}
|
|
2486
|
+
catch {
|
|
2487
|
+
json(response, 400, { error: "bad id" });
|
|
2488
|
+
return;
|
|
2489
|
+
}
|
|
2490
|
+
// By id, or by the media identity itself: both name the same row.
|
|
2491
|
+
const id = idFrom(named);
|
|
2492
|
+
const action = parts[1];
|
|
2493
|
+
if (action === undefined && request.method === "GET") {
|
|
2494
|
+
const language = languageCode(url.searchParams.get("language"));
|
|
2495
|
+
const format = formatOf(url.searchParams.get("format"));
|
|
2496
|
+
if (language === null) {
|
|
2497
|
+
json(response, 400, { error: "language is a two-letter code, or original" });
|
|
2498
|
+
return;
|
|
2499
|
+
}
|
|
2500
|
+
if (format === null) {
|
|
2501
|
+
json(response, 400, { error: "format is json, srt, vtt or txt" });
|
|
2502
|
+
return;
|
|
2503
|
+
}
|
|
2504
|
+
const answer = await options.translations.get(id, language, who.id);
|
|
2505
|
+
if (answer.status !== 200 && answer.status !== 202) {
|
|
2506
|
+
json(response, answer.status, { error: answer.error });
|
|
2507
|
+
return;
|
|
2508
|
+
}
|
|
2509
|
+
if (answer.status === 202) {
|
|
2510
|
+
json(response, 202, {
|
|
2511
|
+
...(answer.transcript ? wire(answer.transcript, await store.languages(id)) : { id, language, lines: [], languages: await store.languages(id) }),
|
|
2512
|
+
translating: answer.translating,
|
|
2513
|
+
});
|
|
2514
|
+
return;
|
|
2515
|
+
}
|
|
2516
|
+
const transcript = answer.transcript;
|
|
2517
|
+
if (format === "json") {
|
|
2518
|
+
json(response, 200, wire(transcript, await store.languages(id)));
|
|
2519
|
+
return;
|
|
2520
|
+
}
|
|
2521
|
+
const text = format === "srt" ? toSrt(transcript.lines) : format === "vtt" ? toVtt(transcript.lines) : toText(transcript.lines);
|
|
2522
|
+
const name = (transcript.title || id).replace(/[^\w.-]+/g, "_").slice(0, 60) || "transcript";
|
|
2523
|
+
response.writeHead(200, {
|
|
2524
|
+
...CORS,
|
|
2525
|
+
"content-type": `${format === "vtt" ? "text/vtt" : "text/plain"}; charset=utf-8`,
|
|
2526
|
+
"content-length": Buffer.byteLength(text),
|
|
2527
|
+
"cache-control": "no-store",
|
|
2528
|
+
"content-disposition": `inline; filename="${name}${transcript.language ? `.${transcript.language}` : ""}.${format}"`,
|
|
2529
|
+
});
|
|
2530
|
+
response.end(text);
|
|
2531
|
+
return;
|
|
2532
|
+
}
|
|
2533
|
+
if (action === "lines" && request.method === "POST") {
|
|
2534
|
+
let body = {};
|
|
2535
|
+
try {
|
|
2536
|
+
body = JSON.parse(await readBody(request, 4 * 1024 * 1024));
|
|
2537
|
+
}
|
|
2538
|
+
catch {
|
|
2539
|
+
json(response, 400, { error: "bad JSON, or more than 4 MB of it" });
|
|
2540
|
+
return;
|
|
2541
|
+
}
|
|
2542
|
+
const media = typeof body.media === "string" ? body.media.trim() : "";
|
|
2543
|
+
if (media === "" || media.length > 2048 || !/^(file|url|live):/.test(media)) {
|
|
2544
|
+
json(response, 400, { error: "media is the identity: file:v1:<hash>, url:<address> or live:<server>/<channel>@<started>" });
|
|
2545
|
+
return;
|
|
2546
|
+
}
|
|
2547
|
+
if (transcriptIdOf(media) !== id) {
|
|
2548
|
+
json(response, 400, { error: "that media is not this transcript's id" });
|
|
2549
|
+
return;
|
|
2550
|
+
}
|
|
2551
|
+
const language = languageCode(body.language);
|
|
2552
|
+
const translatedFrom = body.translatedFrom === undefined || body.translatedFrom === null ? "" : languageCode(body.translatedFrom);
|
|
2553
|
+
if (language === null || translatedFrom === null) {
|
|
2554
|
+
json(response, 400, { error: "language and translatedFrom are two-letter codes" });
|
|
2555
|
+
return;
|
|
2556
|
+
}
|
|
2557
|
+
const lines = linesFrom(body.lines);
|
|
2558
|
+
if (lines.length === 0 && body.complete !== true) {
|
|
2559
|
+
json(response, 400, { error: "lines: a list of {start, end, text}, seconds into the media" });
|
|
2560
|
+
return;
|
|
2561
|
+
}
|
|
2562
|
+
const saved = await store.save({
|
|
2563
|
+
media,
|
|
2564
|
+
language,
|
|
2565
|
+
translatedFrom: translatedFrom === "" ? null : translatedFrom,
|
|
2566
|
+
model: typeof body.model === "string" ? body.model.slice(0, 100) : "",
|
|
2567
|
+
title: typeof body.title === "string" ? body.title : "",
|
|
2568
|
+
complete: body.complete === true,
|
|
2569
|
+
by: who.id,
|
|
2570
|
+
lines,
|
|
2571
|
+
});
|
|
2572
|
+
json(response, 200, { id, language: saved?.language ?? language, saved: lines.length, seconds: saved?.seconds ?? 0, complete: saved?.complete ?? false });
|
|
2573
|
+
return;
|
|
2574
|
+
}
|
|
2575
|
+
if (action === undefined && request.method === "DELETE") {
|
|
2576
|
+
const gone = await store.forget(id, who.id);
|
|
2577
|
+
json(response, gone ? 200 : 404, gone ? { ok: true } : { error: "nothing of yours under that id" });
|
|
2578
|
+
return;
|
|
2579
|
+
}
|
|
2580
|
+
json(response, 405, { error: "GET a transcript, POST its lines, or DELETE it" });
|
|
2581
|
+
}
|
|
2582
|
+
catch (error) {
|
|
2583
|
+
if (error instanceof SpeechError)
|
|
2584
|
+
json(response, error.status, { error: error.message });
|
|
2585
|
+
else
|
|
2586
|
+
throw error;
|
|
2587
|
+
}
|
|
2588
|
+
return;
|
|
2589
|
+
}
|
|
2590
|
+
/*
|
|
2591
|
+
* MCP over HTTP: the tools `nixamp mcp` offers on a pipe, at
|
|
2592
|
+
* nixamp.com/mcp for an agent with no nixamp installed. A JSON-RPC
|
|
2593
|
+
* message, or a batch of them, in a POST; the answers as JSON. The
|
|
2594
|
+
* bearer token says whose tools they are, as on every other route, and
|
|
2595
|
+
* the tools call nixamp.com as that account. No stream is offered on
|
|
2596
|
+
* GET: nothing here has to be pushed, so a client is told so and gets
|
|
2597
|
+
* on with POST.
|
|
2598
|
+
*/
|
|
2599
|
+
if (path === "/mcp" && options.accounts && options.site) {
|
|
2600
|
+
if (request.method === "GET") {
|
|
2601
|
+
response.writeHead(405, { ...CORS, allow: "POST, DELETE" });
|
|
2602
|
+
response.end();
|
|
2603
|
+
return;
|
|
2604
|
+
}
|
|
2605
|
+
if (request.method === "DELETE") {
|
|
2606
|
+
response.writeHead(204, CORS);
|
|
2607
|
+
response.end();
|
|
2608
|
+
return;
|
|
2609
|
+
}
|
|
2610
|
+
if (request.method !== "POST") {
|
|
2611
|
+
json(response, 405, { error: "POST a JSON-RPC message" });
|
|
2612
|
+
return;
|
|
2613
|
+
}
|
|
2614
|
+
const token = tokenFrom(request.headers);
|
|
2615
|
+
const who = await options.accounts.whoIs(token);
|
|
2616
|
+
if (who === null) {
|
|
2617
|
+
const error = JSON.stringify({ error: "sign in: a nixamp token as the bearer (`nixamp token create`), or OAuth" });
|
|
2618
|
+
response.writeHead(401, {
|
|
2619
|
+
...CORS,
|
|
2620
|
+
"content-type": "application/json; charset=utf-8",
|
|
2621
|
+
"content-length": Buffer.byteLength(error),
|
|
2622
|
+
"www-authenticate": `Bearer resource_metadata="${options.site}/.well-known/oauth-protected-resource"`,
|
|
2623
|
+
});
|
|
2624
|
+
response.end(error);
|
|
2625
|
+
return;
|
|
2626
|
+
}
|
|
2627
|
+
let parsed;
|
|
2628
|
+
try {
|
|
2629
|
+
parsed = JSON.parse(await readBody(request, 1024 * 1024));
|
|
2630
|
+
}
|
|
2631
|
+
catch {
|
|
2632
|
+
json(response, 400, { jsonrpc: "2.0", id: null, error: { code: -32700, message: "parse error" } });
|
|
2633
|
+
return;
|
|
2634
|
+
}
|
|
2635
|
+
const messages = Array.isArray(parsed) ? parsed : [parsed];
|
|
2636
|
+
const session = { site: options.site, token };
|
|
2637
|
+
const answers = [];
|
|
2638
|
+
for (const message of messages) {
|
|
2639
|
+
if (!message || typeof message !== "object" || typeof message.method !== "string") {
|
|
2640
|
+
answers.push({ jsonrpc: "2.0", id: null, error: { code: -32600, message: "invalid request" } });
|
|
2641
|
+
continue;
|
|
2642
|
+
}
|
|
2643
|
+
const answer = await mcpMessage(message, { session, fetcher: fetch });
|
|
2644
|
+
if (answer)
|
|
2645
|
+
answers.push(answer);
|
|
2646
|
+
}
|
|
2647
|
+
if (answers.length === 0) {
|
|
2648
|
+
response.writeHead(202, CORS);
|
|
2649
|
+
response.end();
|
|
2650
|
+
return;
|
|
2651
|
+
}
|
|
2652
|
+
json(response, 200, Array.isArray(parsed) ? answers : answers[0]);
|
|
2653
|
+
return;
|
|
2654
|
+
}
|
|
2655
|
+
/* Where a client of /mcp finds the authorization server (RFC 9728). */
|
|
2656
|
+
if (path === "/.well-known/oauth-protected-resource" && options.site && options.authServer) {
|
|
2657
|
+
json(response, 200, {
|
|
2658
|
+
resource: options.site,
|
|
2659
|
+
authorization_servers: [options.site],
|
|
2660
|
+
bearer_methods_supported: ["header"],
|
|
2661
|
+
scopes_supported: SCOPE_NAMES,
|
|
2662
|
+
resource_name: "nixamp",
|
|
2663
|
+
});
|
|
2664
|
+
return;
|
|
2665
|
+
}
|
|
2374
2666
|
if (path === "/api/v1/me/handle" && options.handles && options.accounts) {
|
|
2375
2667
|
const handles = options.handles;
|
|
2376
2668
|
const who = await options.accounts.whoIs(tokenFrom(request.headers));
|
|
@@ -3537,11 +3829,17 @@ export function createHandler(engine, options) {
|
|
|
3537
3829
|
// reconnecting EventSource saw, which the browser sends by itself.
|
|
3538
3830
|
const lastSeen = request.headers["last-event-id"];
|
|
3539
3831
|
const after = Number(url.searchParams.get("after") ?? (Array.isArray(lastSeen) ? lastSeen[0] : lastSeen) ?? 0) || 0;
|
|
3832
|
+
// The lines in another language, translated as they are heard; "" is what was said.
|
|
3833
|
+
const wanted = languageCode(url.searchParams.get("language"));
|
|
3834
|
+
if (wanted === null) {
|
|
3835
|
+
json(response, 400, { error: "language is a two-letter code, such as de or sv" });
|
|
3836
|
+
return;
|
|
3837
|
+
}
|
|
3540
3838
|
if (action === "transcript") {
|
|
3541
3839
|
// Asking keeps the captioner up: it stops a minute after the last ask.
|
|
3542
|
-
captions.subscribe(id, () => undefined)?.();
|
|
3840
|
+
captions.subscribe(id, () => undefined, wanted)?.();
|
|
3543
3841
|
json(response, 200, {
|
|
3544
|
-
channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), ...captions.status(id), lines: captions.recent(id, after),
|
|
3842
|
+
channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted),
|
|
3545
3843
|
});
|
|
3546
3844
|
return;
|
|
3547
3845
|
}
|
|
@@ -3555,14 +3853,14 @@ export function createHandler(engine, options) {
|
|
|
3555
3853
|
const write = (event, data, eventId) => {
|
|
3556
3854
|
response.write(`${eventId === undefined ? "" : `id: ${eventId}\n`}event: ${event}\ndata: ${JSON.stringify(data)}\n\n`);
|
|
3557
3855
|
};
|
|
3558
|
-
const off = captions.subscribe(id, (line) => write("line", line, line.at));
|
|
3856
|
+
const off = captions.subscribe(id, (line) => write("line", line, line.at), wanted);
|
|
3559
3857
|
if (off === null) {
|
|
3560
3858
|
response.end();
|
|
3561
3859
|
return;
|
|
3562
3860
|
}
|
|
3563
3861
|
// How far behind the live edge a newcomer's playback starts, so the
|
|
3564
3862
|
// page can hold each line until its own sound gets there.
|
|
3565
|
-
write("hello", { channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), ...captions.status(id), lines: captions.recent(id, after) });
|
|
3863
|
+
write("hello", { channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), wanted, ...captions.status(id), lines: captions.recent(id, after, wanted) });
|
|
3566
3864
|
const beat = setInterval(() => response.write(": beat\n\n"), 20_000);
|
|
3567
3865
|
beat.unref?.();
|
|
3568
3866
|
const done = () => {
|
|
@@ -4864,6 +5162,22 @@ export async function serve(argv, version = "0.1.0") {
|
|
|
4864
5162
|
console.error(ready ? `nixamp: hearing with ${speech.model}` : `nixamp: not hearing: ${speech.lastFailure}`);
|
|
4865
5163
|
});
|
|
4866
5164
|
}
|
|
5165
|
+
// Translation lives beside the ear: the same optional library, a model
|
|
5166
|
+
// per pair of languages, loaded when first asked for. NIXAMP_MT_WARM
|
|
5167
|
+
// names pairs to load at boot (en-de,en-sv) so the first line does not
|
|
5168
|
+
// wait; NIXAMP_MT=off turns it off.
|
|
5169
|
+
const translator = accounts && process.env["NIXAMP_MT"] !== "off" ? new Translator() : undefined;
|
|
5170
|
+
if (translator) {
|
|
5171
|
+
const warm = (process.env["NIXAMP_MT_WARM"] ?? "").split(",").map((one) => one.trim()).filter(Boolean);
|
|
5172
|
+
if (warm.length > 0) {
|
|
5173
|
+
void translator.warm(warm).then((ready) => {
|
|
5174
|
+
console.error(ready ? `nixamp: translating (${warm.join(", ")} loaded)` : `nixamp: not translating everything: ${translator.lastFailure}`);
|
|
5175
|
+
});
|
|
5176
|
+
}
|
|
5177
|
+
}
|
|
5178
|
+
// What was heard, kept: where the accounts are, under the media's identity.
|
|
5179
|
+
const transcripts = pool ? new Transcripts(pool, (message) => console.log(message)) : undefined;
|
|
5180
|
+
const translations = transcripts ? new StoredTranslations(transcripts, translator, (message) => console.log(message)) : undefined;
|
|
4867
5181
|
// nixamp as an OAuth 2.1 authorization server, and the watch parties a
|
|
4868
5182
|
// client site bridges through it. Both need the same three things -- a
|
|
4869
5183
|
// database, accounts, and a site to be the issuer of -- so both appear or
|
|
@@ -5051,6 +5365,33 @@ export async function serve(argv, version = "0.1.0") {
|
|
|
5051
5365
|
listen: (id, listener) => channels.listen(id, listener),
|
|
5052
5366
|
session: () => readSession(),
|
|
5053
5367
|
onEvent: (message) => console.log(` ${message}`),
|
|
5368
|
+
// What a channel is playing, for the transcript store: a film by
|
|
5369
|
+
// its bytes and where it has got to, a link by its address, and
|
|
5370
|
+
// anything live as the broadcast it is.
|
|
5371
|
+
mediaOf: (id) => {
|
|
5372
|
+
const info = channels.info(id);
|
|
5373
|
+
if (!info)
|
|
5374
|
+
return null;
|
|
5375
|
+
const base = { title: info.name, startedAt: info.startedAt, backlog: info.codecs?.video ? BACKLOG_SECONDS : 3 };
|
|
5376
|
+
if (info.via === "pull" && !info.live && typeof info.position === "number" && info.source) {
|
|
5377
|
+
if (/^https?:\/\//.test(info.source)) {
|
|
5378
|
+
// The pasted address while it is still known: it outlives a media URL with a token in it.
|
|
5379
|
+
let pasted = info.source;
|
|
5380
|
+
for (const [link] of links)
|
|
5381
|
+
if (linkChannelId(link) === id)
|
|
5382
|
+
pasted = link;
|
|
5383
|
+
return { ...base, media: mediaOfUrl(pasted), position: info.position };
|
|
5384
|
+
}
|
|
5385
|
+
try {
|
|
5386
|
+
return { ...base, media: fileFingerprint(info.source), position: info.position };
|
|
5387
|
+
}
|
|
5388
|
+
catch {
|
|
5389
|
+
return null;
|
|
5390
|
+
}
|
|
5391
|
+
}
|
|
5392
|
+
const self = publishable_?.url || options.publicUrl || `${options.host}:${options.port}`;
|
|
5393
|
+
return { ...base, media: mediaOfLive(self, id, info.startedAt) };
|
|
5394
|
+
},
|
|
5054
5395
|
})
|
|
5055
5396
|
: undefined;
|
|
5056
5397
|
const owner = new Owner({
|
|
@@ -5254,6 +5595,10 @@ export async function serve(argv, version = "0.1.0") {
|
|
|
5254
5595
|
...(rooms ? { rooms } : {}),
|
|
5255
5596
|
...(trollbox ? { trollbox } : {}),
|
|
5256
5597
|
...(speech ? { speech } : {}),
|
|
5598
|
+
...(translator ? { translator } : {}),
|
|
5599
|
+
...(transcripts ? { transcripts } : {}),
|
|
5600
|
+
...(translations ? { translations } : {}),
|
|
5601
|
+
site: nixampSite,
|
|
5257
5602
|
// Other people's OpenProfiles, read for the voice a line is spoken in,
|
|
5258
5603
|
// and the voices to speak in: ElevenLabs when the Telnyx account holds
|
|
5259
5604
|
// the key for it (an integration secret named "elevenlabs"), Kokoro
|
package/dist/speech.d.ts
CHANGED
|
@@ -20,16 +20,31 @@ export declare class SpeechError extends Error {
|
|
|
20
20
|
readonly status: number;
|
|
21
21
|
constructor(message: string, status: number);
|
|
22
22
|
}
|
|
23
|
+
/** A stretch of the clip and what was said in it, seconds from the clip's start. */
|
|
24
|
+
export interface Segment {
|
|
25
|
+
start: number;
|
|
26
|
+
end: number;
|
|
27
|
+
text: string;
|
|
28
|
+
}
|
|
23
29
|
export interface Heard {
|
|
24
30
|
text: string;
|
|
25
31
|
/** How long the audio was, in seconds. */
|
|
26
32
|
seconds: number;
|
|
33
|
+
/** The language heard, as told or as guessed; absent when neither was possible. */
|
|
34
|
+
language?: string;
|
|
35
|
+
/** The clip in pieces with their timing, when asked for. */
|
|
36
|
+
segments?: Segment[];
|
|
27
37
|
}
|
|
28
|
-
|
|
29
|
-
export type Recognizer = (pcm: Float32Array, options: {
|
|
38
|
+
export interface RecognizeOptions {
|
|
30
39
|
language?: string;
|
|
31
|
-
|
|
40
|
+
/** Also say when each piece was said. Slower: the ear takes about twice as long. */
|
|
41
|
+
timestamps?: boolean;
|
|
42
|
+
}
|
|
43
|
+
/** Mono samples in [-1, 1] at RATE -> words. What the loaded model is, to this file. */
|
|
44
|
+
export type Recognizer = (pcm: Float32Array, options: RecognizeOptions) => Promise<{
|
|
32
45
|
text: string;
|
|
46
|
+
language?: string;
|
|
47
|
+
segments?: Segment[];
|
|
33
48
|
}>;
|
|
34
49
|
export interface SpeechOptions {
|
|
35
50
|
/** A Hugging Face model id; NIXAMP_STT_MODEL otherwise; whisper-base by default. */
|
|
@@ -66,6 +81,8 @@ export declare function encodeWav(samples: Float32Array, rate?: number): Uint8Ar
|
|
|
66
81
|
export declare function tidy(text: string): string;
|
|
67
82
|
/** A two-letter language code, or nothing: Whisper guesses when not told. */
|
|
68
83
|
export declare function languageOf(value: unknown): string | undefined;
|
|
84
|
+
/** Load the library, or say why not with a status. */
|
|
85
|
+
export declare function loadTransformers<T>(): Promise<T>;
|
|
69
86
|
export declare class Speech {
|
|
70
87
|
readonly model: string;
|
|
71
88
|
private readonly cacheDir;
|
|
@@ -93,6 +110,7 @@ export declare class Speech {
|
|
|
93
110
|
*/
|
|
94
111
|
transcribe(bytes: Uint8Array, options?: {
|
|
95
112
|
language?: string;
|
|
113
|
+
timestamps?: boolean;
|
|
96
114
|
by?: string;
|
|
97
115
|
}): Promise<Heard>;
|
|
98
116
|
}
|
package/dist/speech.js
CHANGED
|
@@ -196,13 +196,13 @@ export function languageOf(value) {
|
|
|
196
196
|
const code = value.trim().toLowerCase();
|
|
197
197
|
return /^[a-z]{2}$/.test(code) ? code : undefined;
|
|
198
198
|
}
|
|
199
|
-
|
|
199
|
+
/** Load the library, or say why not with a status. */
|
|
200
|
+
export async function loadTransformers() {
|
|
200
201
|
// By name held in a variable: an optional dependency that is not on disk
|
|
201
202
|
// must fail here, at the first ask, and not when the file is imported.
|
|
202
203
|
const name = "@huggingface/transformers";
|
|
203
|
-
let transformers;
|
|
204
204
|
try {
|
|
205
|
-
|
|
205
|
+
return (await import(__rewriteRelativeImportExtension(name)));
|
|
206
206
|
}
|
|
207
207
|
catch (error) {
|
|
208
208
|
// The reason travels with the refusal: "not installed" and "installed
|
|
@@ -211,16 +211,62 @@ async function loadWhisper(model, cacheDir) {
|
|
|
211
211
|
const why = String(error.message ?? error).split("\n")[0] ?? "";
|
|
212
212
|
throw new SpeechError(`this nixamp cannot hear: @huggingface/transformers did not load here (${why}). nixamp.com can.`, 503);
|
|
213
213
|
}
|
|
214
|
+
}
|
|
215
|
+
/**
|
|
216
|
+
* Which language a clip is in, the way whisper.cpp asks: the encoder once
|
|
217
|
+
* over the first thirty seconds, one decoder step from the start token,
|
|
218
|
+
* and the loudest of the language tokens wins. Whisper told nothing
|
|
219
|
+
* assumes English, and a Swedish sentence heard as English comes back as
|
|
220
|
+
* the same three English words repeated to the end of the clip; guessing
|
|
221
|
+
* first costs an encoder pass and is what makes a caption in Swedish say
|
|
222
|
+
* something. Undefined for a model that knows one language.
|
|
223
|
+
*/
|
|
224
|
+
async function detectLanguage(transformers, asr, pcm) {
|
|
225
|
+
const config = asr.model.generation_config;
|
|
226
|
+
const languages = config.lang_to_id;
|
|
227
|
+
if (!languages || config.is_multilingual === false)
|
|
228
|
+
return undefined;
|
|
229
|
+
const inputs = await asr.processor(pcm.subarray(0, 30 * RATE));
|
|
230
|
+
const start = new transformers.Tensor("int64", BigInt64Array.from([BigInt(config.decoder_start_token_id)]), [1, 1]);
|
|
231
|
+
const out = await asr.model({ input_features: inputs.input_features, decoder_input_ids: start });
|
|
232
|
+
const logits = out.logits.data;
|
|
233
|
+
let best;
|
|
234
|
+
let loudest = -Infinity;
|
|
235
|
+
for (const [token, id] of Object.entries(languages)) {
|
|
236
|
+
const score = logits[id];
|
|
237
|
+
if (typeof score === "number" && score > loudest) {
|
|
238
|
+
loudest = score;
|
|
239
|
+
best = token;
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
// The token is written <|sv|>.
|
|
243
|
+
const code = best?.replace(/^<\|/, "").replace(/\|>$/, "");
|
|
244
|
+
return code && /^[a-z]{2,3}$/.test(code) ? code : undefined;
|
|
245
|
+
}
|
|
246
|
+
async function loadWhisper(model, cacheDir) {
|
|
247
|
+
const transformers = await loadTransformers();
|
|
214
248
|
transformers.env.cacheDir = cacheDir;
|
|
215
249
|
const recognize = await transformers.pipeline("automatic-speech-recognition", model, { dtype: "q8" });
|
|
216
|
-
return async (pcm, { language }) => {
|
|
250
|
+
return async (pcm, { language, timestamps }) => {
|
|
251
|
+
const spoken = language ?? (await detectLanguage(transformers, recognize, pcm));
|
|
217
252
|
const heard = await recognize(pcm, {
|
|
218
253
|
// Whisper hears thirty seconds at a time; longer is heard in overlapping pieces.
|
|
219
254
|
chunk_length_s: 30,
|
|
220
255
|
stride_length_s: 5,
|
|
221
|
-
...(
|
|
256
|
+
...(spoken ? { language: spoken, task: "transcribe" } : {}),
|
|
257
|
+
...(timestamps ? { return_timestamps: true } : {}),
|
|
222
258
|
});
|
|
223
|
-
|
|
259
|
+
const pieces = Array.isArray(heard) ? heard : [heard];
|
|
260
|
+
const text = pieces.map((piece) => piece.text).join(" ");
|
|
261
|
+
const seconds = pcm.length / RATE;
|
|
262
|
+
const segments = timestamps
|
|
263
|
+
? pieces.flatMap((piece) => piece.chunks ?? []).map((chunk) => ({
|
|
264
|
+
start: Math.max(0, chunk.timestamp[0] ?? 0),
|
|
265
|
+
end: Math.min(seconds, chunk.timestamp[1] ?? seconds),
|
|
266
|
+
text: chunk.text,
|
|
267
|
+
}))
|
|
268
|
+
: undefined;
|
|
269
|
+
return { text, ...(spoken ? { language: spoken } : {}), ...(segments ? { segments } : {}) };
|
|
224
270
|
};
|
|
225
271
|
}
|
|
226
272
|
export class Speech {
|
|
@@ -303,13 +349,24 @@ export class Speech {
|
|
|
303
349
|
this.waiting += 1;
|
|
304
350
|
const turn = this.tail.then(async () => {
|
|
305
351
|
const recognize = await this.ear();
|
|
306
|
-
return recognize(pcm,
|
|
352
|
+
return recognize(pcm, {
|
|
353
|
+
...(options.language ? { language: options.language } : {}),
|
|
354
|
+
...(options.timestamps ? { timestamps: true } : {}),
|
|
355
|
+
});
|
|
307
356
|
});
|
|
308
357
|
// The queue moves on whether or not this one was heard.
|
|
309
358
|
this.tail = turn.catch(() => undefined);
|
|
310
359
|
try {
|
|
311
360
|
const heard = await turn;
|
|
312
|
-
|
|
361
|
+
const segments = heard.segments
|
|
362
|
+
?.map((segment) => ({ start: segment.start, end: segment.end, text: tidy(segment.text) }))
|
|
363
|
+
.filter((segment) => segment.text !== "");
|
|
364
|
+
return {
|
|
365
|
+
text: tidy(heard.text),
|
|
366
|
+
seconds,
|
|
367
|
+
...(heard.language ? { language: heard.language } : {}),
|
|
368
|
+
...(segments ? { segments } : {}),
|
|
369
|
+
};
|
|
313
370
|
}
|
|
314
371
|
finally {
|
|
315
372
|
this.waiting -= 1;
|