nixamp 0.23.7 → 0.24.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +79 -3
  2. package/dist/captions.d.ts +57 -10
  3. package/dist/captions.js +253 -29
  4. package/dist/main.js +63 -19
  5. package/dist/mcp.d.ts +7 -1
  6. package/dist/mcp.js +159 -24
  7. package/dist/server.d.ts +11 -0
  8. package/dist/server.js +353 -8
  9. package/dist/speech.d.ts +21 -3
  10. package/dist/speech.js +65 -8
  11. package/dist/transcribe.d.ts +70 -0
  12. package/dist/transcribe.js +337 -41
  13. package/dist/transcript-client.d.ts +81 -0
  14. package/dist/transcript-client.js +76 -0
  15. package/dist/transcript.d.ts +7 -2
  16. package/dist/transcript.js +73 -5
  17. package/dist/transcripts.d.ts +135 -0
  18. package/dist/transcripts.js +384 -0
  19. package/dist/translate-cli.d.ts +16 -0
  20. package/dist/translate-cli.js +97 -0
  21. package/dist/translate-jobs.d.ts +39 -0
  22. package/dist/translate-jobs.js +120 -0
  23. package/dist/translate.d.ts +78 -0
  24. package/dist/translate.js +241 -0
  25. package/dist/warm.d.ts +1 -0
  26. package/dist/warm.js +40 -0
  27. package/package.json +1 -1
  28. package/src/captions.ts +276 -29
  29. package/src/main.ts +63 -19
  30. package/src/mcp.ts +156 -21
  31. package/src/server.ts +352 -8
  32. package/src/speech.ts +103 -13
  33. package/src/transcribe.ts +374 -41
  34. package/src/transcript-client.ts +147 -0
  35. package/src/transcript.ts +76 -4
  36. package/src/transcripts.ts +462 -0
  37. package/src/translate-cli.ts +111 -0
  38. package/src/translate-jobs.ts +132 -0
  39. package/src/translate.ts +276 -0
  40. package/src/warm.ts +43 -0
  41. package/web/dist/assets/{hls-3VKVEQE3-CI1U7kbP.js → hls-3VKVEQE3-Dtl-3mpW.js} +1 -1
  42. package/web/dist/assets/index-B0h4Nexr.js +1 -0
  43. package/web/dist/assets/index-BTGV3Pi5.css +1 -0
  44. package/web/dist/assets/{mpegts-CPOYjgRP.js → mpegts-BJC48bFV.js} +1 -1
  45. package/web/dist/assets/{mpegts-LO6RVLD6-CE8YPjx1.js → mpegts-LO6RVLD6-CUIAB9k3.js} +1 -1
  46. package/web/dist/index.html +12 -8
  47. package/web/dist/sw.js +6 -6
  48. package/web/dist/assets/index-46pwGn5-.css +0 -1
  49. package/web/dist/assets/index-5H3sUGHu.js +0 -1
package/src/mcp.ts CHANGED
@@ -18,10 +18,13 @@
18
18
  * every diagnostic goes to stderr. That is the one rule of this file.
19
19
  */
20
20
  import { createInterface } from "node:readline";
21
+ import { basename, extname } from "node:path";
21
22
  import { clock, type PartyRow } from "./party.ts";
22
23
  import { readSession } from "./session.ts";
23
- import { askToHear, wavOf } from "./transcribe.ts";
24
+ import { askToHear, awaitTranscript, hearWhole, rendered, wavOf, type Window } from "./transcribe.ts";
24
25
  import { readTranscript } from "./transcript.ts";
26
+ import { fetchTranscript, listTranscripts, translateTexts, type StoredTranscript } from "./transcript-client.ts";
27
+ import { fileFingerprint, idFrom, languageCode, mediaOfUrl, transcriptIdOf } from "./transcripts.ts";
25
28
  import { personaLines, readPersona, readVoices, writePersona } from "./profile.ts";
26
29
 
27
30
  export const PROTOCOL_VERSION = "2025-06-18";
@@ -102,18 +105,55 @@ export const TOOLS: ToolDefinition[] = [
102
105
  {
103
106
  name: "transcribe_audio",
104
107
  description:
105
- "The words in a recording on this machine, heard by nixamp.com's own open-source ear (Whisper). Any format ffmpeg reads; up to a minute. Given a server, the words are also posted to that server's trollbox as this account.",
108
+ "The words in a recording, a film or a link on this machine, heard by nixamp.com's own open-source ear (Whisper) a minute at a time, with when each line is said, and kept on nixamp.com under the file's fingerprint so the same file is never heard twice by anybody. Any format ffmpeg reads. Given a server, the recording is a short clip and the words are posted to that server's trollbox as this account instead.",
106
109
  inputSchema: {
107
110
  type: "object",
108
111
  properties: {
109
- path: { ...STRING, description: "The recording's path on this machine." },
112
+ path: { ...STRING, description: "The recording's path on this machine, or a URL ffmpeg can read." },
110
113
  language: { ...STRING, description: "A two-letter language code, when Whisper should not guess." },
111
- server: { ...STRING, description: "Post the words to this nixamp's trollbox: its address, as in its share link." },
114
+ translate: { ...STRING, description: "Also in this language: a two-letter code such as de or sv (several: de,sv). Made once on nixamp.com and kept." },
115
+ format: { ...STRING, description: "How to answer: lines (default, with seconds), srt, vtt or txt." },
116
+ fresh: { type: "boolean", description: "Hear it again even though it is kept." },
117
+ server: { ...STRING, description: "Post the words to this nixamp's trollbox: its address, as in its share link. A clip of a minute at most." },
112
118
  channel: { ...STRING, description: "Which of that server's channels; its own stream (live) by default." },
113
119
  },
114
120
  required: ["path"],
115
121
  },
116
122
  },
123
+ {
124
+ name: "transcript_get",
125
+ description:
126
+ "A kept transcript from nixamp.com: what a file, a link or a past live said, by its id or its media identity (file:v1:<hash>, url:<address>, live:<server>/<channel>@<started>). Ask for a language and it is translated once, on nixamp.com, and kept; a long one is answered with progress and is ready on a later ask.",
127
+ inputSchema: {
128
+ type: "object",
129
+ properties: {
130
+ media: { ...STRING, description: "The transcript's id, or the media identity." },
131
+ language: { ...STRING, description: "A two-letter code for a translation; the original when left out." },
132
+ format: { ...STRING, description: "lines (default, with seconds), srt, vtt or txt." },
133
+ },
134
+ required: ["media"],
135
+ },
136
+ },
137
+ {
138
+ name: "transcripts_list",
139
+ description: "What this account has had written down on nixamp.com: each transcript's id, what it is, its language, how many lines, and when.",
140
+ inputSchema: { type: "object", properties: {} },
141
+ },
142
+ {
143
+ name: "translate_text",
144
+ description:
145
+ "Text in another language, by an open-source model on nixamp.com's own CPU (OPUS-MT). Two-letter codes; German and Swedish among them, and anything with a model from or into English.",
146
+ inputSchema: {
147
+ type: "object",
148
+ properties: {
149
+ text: { ...STRING, description: "The text. Or `texts`, a list." },
150
+ texts: { type: "array", items: STRING, description: "Several texts, answered in the same order." },
151
+ from: { ...STRING, description: "The language the text is in, e.g. en." },
152
+ to: { ...STRING, description: "The language wanted, e.g. sv." },
153
+ },
154
+ required: ["from", "to"],
155
+ },
156
+ },
117
157
  {
118
158
  name: "trollbox_say",
119
159
  description:
@@ -131,13 +171,14 @@ export const TOOLS: ToolDefinition[] = [
131
171
  {
132
172
  name: "transcript_read",
133
173
  description:
134
- "What a live channel is saying: the recent lines of its transcript, oldest first, each with when its sound was heard. The server carrying the channel captions it while somebody asks. Pass the server's address and share key, and the channel's id.",
174
+ "What a live channel is saying: the recent lines of its transcript, oldest first, each with when its sound was heard. The server carrying the channel captions it while somebody asks, and translates each line when a language is asked for. Pass the server's address and share key, and the channel's id.",
135
175
  inputSchema: {
136
176
  type: "object",
137
177
  properties: {
138
178
  url: { ...STRING, description: "The nixamp server's address, e.g. https://server1.chovy.nixamp.com:4321." },
139
179
  key: { ...STRING, description: "The share key from its link, when it has one." },
140
180
  channel: { ...STRING, description: "The channel's id on that server (default: main)." },
181
+ language: { ...STRING, description: "The lines in this language (a two-letter code); as heard when left out." },
141
182
  after: { type: "number", description: "Only lines heard after this moment (ms since the epoch)." },
142
183
  },
143
184
  required: ["url"],
@@ -198,9 +239,23 @@ export interface McpOptions {
198
239
  session?: { site: string; token: string } | null;
199
240
  /** How a recording becomes a WAV; the tests hand in a fake. */
200
241
  wavOf?: typeof wavOf;
242
+ /** The whole of a file as windows of sound, and a file's identity; the tests hand in fakes. */
243
+ windows?: (source: string) => AsyncIterable<Window>;
244
+ fingerprint?: (path: string) => string;
245
+ sleep?: (ms: number) => Promise<void>;
246
+ /** How many times a translation is asked about before answering with its progress. */
247
+ polls?: number;
201
248
  say?: (line: string) => void;
202
249
  }
203
250
 
251
+ /** A kept transcript, as a tool answers it. */
252
+ function transcriptText(transcript: StoredTranscript, format: string): string {
253
+ const shape = format === "srt" || format === "vtt" || format === "txt" ? format : "lines";
254
+ const head = `${transcript.title || transcript.media} (${transcript.id.slice(0, 12)}), ${transcript.language || "language unknown"}${transcript.translatedFrom ? ` from ${transcript.translatedFrom}` : ""}, ${transcript.lines.length} lines${transcript.complete ? "" : ", so far"}`;
255
+ const others = transcript.languages.filter((one) => one.language !== transcript.language).map((one) => one.language || "original");
256
+ return `${head}${others.length > 0 ? `; also in ${others.join(", ")}` : ""}\n\n${rendered(transcript.lines, shape)}`;
257
+ }
258
+
204
259
  /** A tool answer, in the shape MCP wants: content blocks, and a flag for failure. */
205
260
  export interface ToolResult {
206
261
  content: { type: "text"; text: string }[];
@@ -309,22 +364,101 @@ export async function callTool(name: string, args: Record<string, unknown>, opti
309
364
  if (name === "transcribe_audio") {
310
365
  const path = typeof args["path"] === "string" ? args["path"] : "";
311
366
  if (!path) return failed("Which recording? Pass its path.");
312
- let wav: Uint8Array;
367
+ const language = languageCode(args["language"]) || undefined;
368
+ if (server) {
369
+ let wav: Uint8Array;
370
+ try {
371
+ wav = (options.wavOf ?? wavOf)(path);
372
+ } catch (error) {
373
+ return failed((error as Error).message);
374
+ }
375
+ const answer = await askToHear(session, { wav, ...(language ? { language } : {}), server, channel }, send, site);
376
+ if (!answer.ok) return failed(answer.error);
377
+ if (answer.heard.text === "") return text("Heard nothing in that recording.");
378
+ return text(answer.heard.message
379
+ ? `${answer.heard.text}\n\nSaid in the room for ${channel} at ${server} as ${answer.heard.message.handle}.`
380
+ : answer.heard.text);
381
+ }
382
+ // The whole of it, kept under what it is.
383
+ let media: string;
313
384
  try {
314
- wav = (options.wavOf ?? wavOf)(path);
315
- } catch (error) {
316
- return failed((error as Error).message);
385
+ media = /^https?:\/\//.test(path) ? mediaOfUrl(path) : (options.fingerprint ?? fileFingerprint)(path);
386
+ } catch {
387
+ return failed(`cannot read ${path}`);
317
388
  }
318
- const answer = await askToHear(session, {
319
- wav,
320
- ...(typeof args["language"] === "string" ? { language: args["language"] } : {}),
321
- ...(server ? { server, channel } : {}),
322
- }, send, site);
323
- if (!answer.ok) return failed(answer.error);
324
- if (answer.heard.text === "") return text("Heard nothing in that recording.");
325
- return text(answer.heard.message
326
- ? `${answer.heard.text}\n\nSaid in the room for ${channel} at ${server} as ${answer.heard.message.handle}.`
327
- : answer.heard.text);
389
+ const id = transcriptIdOf(media);
390
+ const title = /^https?:\/\//.test(path) ? path : basename(path, extname(path));
391
+ const format = typeof args["format"] === "string" ? args["format"] : "lines";
392
+ const signed = { site, token: session.token };
393
+ let original: StoredTranscript | null = null;
394
+ if (args["fresh"] !== true) {
395
+ const kept = await fetchTranscript(signed, id, "", send);
396
+ if (kept.ok && kept.body.complete) original = kept.body;
397
+ else if (!kept.ok && kept.status !== 404) return failed(kept.error);
398
+ }
399
+ if (!original) {
400
+ const heard = await hearWhole(signed, path, media, { ...(language ? { language } : {}), title }, {
401
+ fetcher: send,
402
+ ...(options.windows ? { windows: options.windows } : {}),
403
+ ...(options.sleep ? { sleep: options.sleep } : {}),
404
+ ...(options.say ? { onProgress: options.say } : {}),
405
+ });
406
+ if (!heard.ok) return failed(heard.error);
407
+ if (heard.heard.lines.length === 0) return text("Heard nothing in that.");
408
+ const kept = await fetchTranscript(signed, id, "", send);
409
+ original = kept.ok ? kept.body : {
410
+ id, media, kind: "file", language: heard.heard.language, translatedFrom: null, model: heard.heard.model, complete: true, title,
411
+ seconds: heard.heard.seconds, updatedAt: "", lines: heard.heard.lines, languages: [],
412
+ };
413
+ }
414
+ const wanted = (typeof args["translate"] === "string" ? args["translate"] : "").split(",").map((one) => languageCode(one)).filter((one): one is string => typeof one === "string" && one !== "");
415
+ const parts = [transcriptText(original, format)];
416
+ for (const to of wanted) {
417
+ if (to === original.language) continue;
418
+ const got = await awaitTranscript(signed, id, to, { fetcher: send, ...(options.sleep ? { sleep: options.sleep } : {}), ...(options.polls !== undefined ? { polls: options.polls } : {}) });
419
+ if (!got.ok) return failed(`could not get it in ${to}: ${got.error}`);
420
+ parts.push(got.body.translating
421
+ ? `In ${to}: still being translated, ${got.body.translating.done} of ${got.body.translating.total} lines. Ask transcript_get for ${id} in ${to} in a moment.`
422
+ : transcriptText(got.body, format));
423
+ }
424
+ return text(parts.join("\n\n"));
425
+ }
426
+
427
+ if (name === "transcript_get") {
428
+ const named = typeof args["media"] === "string" ? args["media"].trim() : "";
429
+ if (!named) return failed("Which transcript? Pass its id or the media identity.");
430
+ const language = languageCode(args["language"]);
431
+ if (language === null) return failed("language is a two-letter code, such as de or sv.");
432
+ const got = await awaitTranscript({ site, token: session.token }, idFrom(named), language, {
433
+ fetcher: send, ...(options.sleep ? { sleep: options.sleep } : {}), polls: options.polls ?? 1,
434
+ });
435
+ if (!got.ok) return failed(got.error);
436
+ if (got.body.translating) {
437
+ return text(`Still being translated to ${language}: ${got.body.translating.done} of ${got.body.translating.total} lines. Ask again in a moment.`);
438
+ }
439
+ return text(transcriptText(got.body, typeof args["format"] === "string" ? args["format"] : "lines"));
440
+ }
441
+
442
+ if (name === "transcripts_list") {
443
+ const got = await listTranscripts({ site, token: session.token }, send);
444
+ if (!got.ok) return failed(got.error);
445
+ if (got.body.transcripts.length === 0) return text("Nothing has been written down for this account yet.");
446
+ return text(got.body.transcripts.map((one) =>
447
+ `${one.id} ${one.language || "?"}${one.translatedFrom ? `<${one.translatedFrom}` : ""} ${one.lines} lines${one.complete ? "" : " so far"} ${one.title || one.media} ${one.updatedAt}`,
448
+ ).join("\n"));
449
+ }
450
+
451
+ if (name === "translate_text") {
452
+ const texts = Array.isArray(args["texts"])
453
+ ? args["texts"].filter((one): one is string => typeof one === "string")
454
+ : typeof args["text"] === "string" ? [args["text"]] : [];
455
+ if (texts.length === 0) return failed("Translate what? Pass text, or texts.");
456
+ const from = languageCode(args["from"]);
457
+ const to = languageCode(args["to"]);
458
+ if (!from || !to) return failed("from and to are two-letter language codes, such as en and sv.");
459
+ const got = await translateTexts({ site, token: session.token }, texts, from, to, send);
460
+ if (!got.ok) return failed(got.error);
461
+ return text(got.body.texts.join("\n"));
328
462
  }
329
463
 
330
464
  if (name === "trollbox_say") {
@@ -349,6 +483,7 @@ export async function callTool(name: string, args: Record<string, unknown>, opti
349
483
  channel === "live" ? "main" : channel,
350
484
  typeof args["after"] === "number" ? args["after"] : 0,
351
485
  send,
486
+ languageCode(args["language"]) || "",
352
487
  );
353
488
  if (!got.ok) return failed(got.error);
354
489
  if (got.answer.recent.length === 0) {
@@ -409,9 +544,9 @@ export async function handleMessage(message: Request, options: McpOptions = {}):
409
544
  return reply({
410
545
  protocolVersion: PROTOCOL_VERSION,
411
546
  capabilities: { tools: { listChanged: false } },
412
- serverInfo: { name: "nixamp", title: "nixamp: watch parties and rooms", version: "1" },
547
+ serverInfo: { name: "nixamp", title: "nixamp: watch parties, rooms and transcripts", version: "2" },
413
548
  instructions:
414
- "Watch parties on nixamp. A party lives on the site hosting the film and is bridged here as a room every nixamp client can join. Codes are the ones that site shows; positions are seconds into the film.",
549
+ "Watch parties on nixamp: a party lives on the site hosting the film and is bridged here as a room every nixamp client can join; codes are the ones that site shows, positions are seconds into the film. Rooms: say and read trollbox lines, hear a recording. Transcripts: a file, a link or a live is written down once by nixamp.com's own ear and kept under what it is; ask for it in another language and it is translated once and kept too.",
415
550
  });
416
551
  }
417
552
  // Notifications carry no id and are answered with silence, which is what