nixamp 0.23.7 → 0.24.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +79 -3
  2. package/dist/captions.d.ts +57 -10
  3. package/dist/captions.js +253 -29
  4. package/dist/main.js +63 -19
  5. package/dist/mcp.d.ts +7 -1
  6. package/dist/mcp.js +159 -24
  7. package/dist/server.d.ts +11 -0
  8. package/dist/server.js +353 -8
  9. package/dist/speech.d.ts +21 -3
  10. package/dist/speech.js +65 -8
  11. package/dist/transcribe.d.ts +70 -0
  12. package/dist/transcribe.js +337 -41
  13. package/dist/transcript-client.d.ts +81 -0
  14. package/dist/transcript-client.js +76 -0
  15. package/dist/transcript.d.ts +7 -2
  16. package/dist/transcript.js +73 -5
  17. package/dist/transcripts.d.ts +135 -0
  18. package/dist/transcripts.js +384 -0
  19. package/dist/translate-cli.d.ts +16 -0
  20. package/dist/translate-cli.js +97 -0
  21. package/dist/translate-jobs.d.ts +39 -0
  22. package/dist/translate-jobs.js +120 -0
  23. package/dist/translate.d.ts +78 -0
  24. package/dist/translate.js +241 -0
  25. package/dist/warm.d.ts +1 -0
  26. package/dist/warm.js +40 -0
  27. package/package.json +1 -1
  28. package/src/captions.ts +276 -29
  29. package/src/main.ts +63 -19
  30. package/src/mcp.ts +156 -21
  31. package/src/server.ts +352 -8
  32. package/src/speech.ts +103 -13
  33. package/src/transcribe.ts +374 -41
  34. package/src/transcript-client.ts +147 -0
  35. package/src/transcript.ts +76 -4
  36. package/src/transcripts.ts +462 -0
  37. package/src/translate-cli.ts +111 -0
  38. package/src/translate-jobs.ts +132 -0
  39. package/src/translate.ts +276 -0
  40. package/src/warm.ts +43 -0
  41. package/web/dist/assets/{hls-3VKVEQE3-CI1U7kbP.js → hls-3VKVEQE3-Dtl-3mpW.js} +1 -1
  42. package/web/dist/assets/index-B0h4Nexr.js +1 -0
  43. package/web/dist/assets/index-BTGV3Pi5.css +1 -0
  44. package/web/dist/assets/{mpegts-CPOYjgRP.js → mpegts-BJC48bFV.js} +1 -1
  45. package/web/dist/assets/{mpegts-LO6RVLD6-CE8YPjx1.js → mpegts-LO6RVLD6-CUIAB9k3.js} +1 -1
  46. package/web/dist/index.html +12 -8
  47. package/web/dist/sw.js +6 -6
  48. package/web/dist/assets/index-46pwGn5-.css +0 -1
  49. package/web/dist/assets/index-5H3sUGHu.js +0 -1
package/src/captions.ts CHANGED
@@ -20,10 +20,21 @@
20
20
  * A quiet window -- the gap between songs, a picture with no talking -- is
21
21
  * never sent. Most of a music channel is that, and hearing it costs the
22
22
  * same as hearing speech.
23
+ *
24
+ * What is heard is kept (see transcripts.ts): every line goes to nixamp.com
25
+ * under the identity of what the channel is playing, as seconds into it.
26
+ * When a captioner starts it asks for what is already known, and a window
27
+ * whose moment the store has been through is read out of it instead of
28
+ * heard. A film captioned once is captioned by nobody again; a live is
29
+ * kept as the broadcast it was. And a viewer may ask for the lines in
30
+ * another language: each heard line is translated once, on nixamp.com,
31
+ * handed to whoever wanted that language, and kept beside the original.
23
32
  */
24
33
  import { spawn } from "node:child_process";
25
34
  import type { Listener } from "./channels.ts";
26
35
  import { RATE } from "./speech.ts";
36
+ import { fetchTranscript, keepLines, translateTexts } from "./transcript-client.ts";
37
+ import { covered, lineAt, transcriptIdOf, type TranscriptLine } from "./transcripts.ts";
27
38
 
28
39
  export interface CaptionLine {
29
40
  /** The channel's id. */
@@ -32,6 +43,10 @@ export interface CaptionLine {
32
43
  at: number;
33
44
  until: number;
34
45
  text: string;
46
+ /** The language of the words, as heard or as translated into; absent when nobody said. */
47
+ language?: string;
48
+ /** What was heard, when this line is a translation of it. */
49
+ original?: string;
35
50
  }
36
51
 
37
52
  /** What turns a channel's bytes into 16 kHz mono 16-bit PCM. ffmpeg, or a test's stand-in. */
@@ -40,6 +55,22 @@ export interface Decoder {
40
55
  end(): void;
41
56
  }
42
57
 
58
+ /**
59
+ * What a channel is playing, for the store: its identity, and how the
60
+ * sound a new listener gets maps onto seconds of it.
61
+ */
62
+ export interface ChannelMedia {
63
+ /** The media identity, as transcripts.ts spells one. */
64
+ media: string;
65
+ title: string;
66
+ /** Seconds into the media at this moment, for a film; absent for anything live. */
67
+ position?: number;
68
+ /** When the channel began, wall clock, ms. A live's seconds count from here. */
69
+ startedAt: number;
70
+ /** How many seconds behind the live edge a new listener's sound starts. */
71
+ backlog: number;
72
+ }
73
+
43
74
  export interface CaptionsOptions {
44
75
  /** A listener on a channel, or null when there is no such channel. */
45
76
  listen: (id: string, listener: Listener) => (() => void) | null;
@@ -49,10 +80,14 @@ export interface CaptionsOptions {
49
80
  fetcher?: typeof fetch;
50
81
  /** How bytes become PCM. The default spawns ffmpeg; the tests hand in something quieter. */
51
82
  decoder?: (onPcm: (pcm: Buffer) => void, onEnd: () => void) => Decoder;
83
+ /** What a channel is playing, for the store. Null, or absent, means the lines are not kept. */
84
+ mediaOf?: (id: string) => ChannelMedia | null;
52
85
  now?: () => number;
53
86
  onEvent?: (message: string) => void;
54
87
  windowMs?: number;
55
88
  idleMs?: number;
89
+ /** How often heard lines go to the store. */
90
+ flushMs?: number;
56
91
  }
57
92
 
58
93
  export const WINDOW_MS = 5000;
@@ -63,6 +98,10 @@ export const IDLE_MS = 60_000;
63
98
  export const QUIET = 0.004;
64
99
  /** Windows waiting on the ear at once. Past this the sound is dropped, not queued: late words are worse than none. */
65
100
  export const IN_FLIGHT = 2;
101
+ /** Heard lines wait this long, at most, before they are kept. */
102
+ export const FLUSH_MS = 20_000;
103
+ /** Or this many. */
104
+ export const FLUSH_LINES = 12;
66
105
 
67
106
  /**
68
107
  * The ffmpeg arguments: whatever arrives on stdin, as PCM on stdout, with
@@ -148,13 +187,52 @@ export function wavAround(pcm: Buffer, rate = RATE): Buffer {
148
187
  return Buffer.concat([header, pcm]);
149
188
  }
150
189
 
190
+ /**
191
+ * Seconds into the media that a window covers. A film is paced in real
192
+ * time from a known position, so a new listener's first byte is that
193
+ * position less the backlog it was handed, and every window after it is
194
+ * one window further on. Anything live counts from when the channel
195
+ * began, in wall-clock time.
196
+ */
197
+ export function mediaSpan(media: ChannelMedia, windowIndex: number, windowSeconds: number, at: number, until: number): { start: number; end: number } {
198
+ if (typeof media.position === "number") {
199
+ const join = Math.max(0, media.position - media.backlog);
200
+ return { start: round(join + windowIndex * windowSeconds), end: round(join + (windowIndex + 1) * windowSeconds) };
201
+ }
202
+ return { start: round(Math.max(0, (at - media.startedAt) / 1000)), end: round(Math.max(0, (until - media.startedAt) / 1000)) };
203
+ }
204
+
205
+ function round(seconds: number): number {
206
+ return Math.round(seconds * 1000) / 1000;
207
+ }
208
+
209
+ export { lineAt };
210
+
151
211
  type Subscriber = (line: CaptionLine) => void;
152
212
 
213
+ export interface CaptionStatus {
214
+ on: boolean;
215
+ lines: number;
216
+ error: string;
217
+ /** The language heard, when known. */
218
+ language: string;
219
+ /** How many lines the store already had for this media when the captioner began. */
220
+ known: number;
221
+ /** The languages lines are being given in besides the original. */
222
+ languages: string[];
223
+ }
224
+
153
225
  class Captioner {
226
+ /** The lines as heard, oldest first. */
154
227
  readonly lines: CaptionLine[] = [];
155
- readonly subscribers = new Set<Subscriber>();
228
+ /** The lines in each other language somebody asked for. */
229
+ readonly linesBy = new Map<string, CaptionLine[]>();
230
+ readonly subscribers = new Map<Subscriber, string>();
156
231
  /** The last thing that went wrong, for whoever asks why there are no lines. */
157
232
  error = "";
233
+ /** What the ear says the sound is in, or the store said it was; "" until one of them has. */
234
+ language = "";
235
+ private model = "";
158
236
  private decoder: Decoder | null = null;
159
237
  private detach: (() => void) | null = null;
160
238
  private idle: ReturnType<typeof setTimeout> | null = null;
@@ -163,17 +241,41 @@ class Captioner {
163
241
  private inFlight = 0;
164
242
  private stopped = false;
165
243
  private complainedAt = 0;
244
+ private windows = 0;
245
+ /** What the channel is playing, when the store is to be told. */
246
+ private readonly media: ChannelMedia | null;
247
+ private readonly transcriptId: string | null;
248
+ /** What the store had, by language ("" for the original), and which stored moments have been read out. */
249
+ private readonly known = new Map<string, TranscriptLine[]>();
250
+ private readonly readOut = new Set<number>();
251
+ private readonly asked = new Set<string>();
252
+ /** Lines heard or translated here and not yet kept, by language. */
253
+ private readonly unsaved = new Map<string, TranscriptLine[]>();
254
+ private flush: ReturnType<typeof setTimeout> | null = null;
255
+ /** Translations in order, per language: a slow one must not overtake the next. */
256
+ private readonly chains = new Map<string, Promise<void>>();
166
257
 
167
258
  constructor(
168
259
  readonly id: string,
169
260
  private readonly options: CaptionsOptions,
170
261
  private readonly onStop: () => void,
171
- ) {}
262
+ ) {
263
+ this.media = options.mediaOf?.(id) ?? null;
264
+ this.transcriptId = this.media ? transcriptIdOf(this.media.media) : null;
265
+ }
172
266
 
173
267
  private get windowBytes(): number {
174
268
  return Math.round(((this.options.windowMs ?? WINDOW_MS) / 1000) * RATE) * 2;
175
269
  }
176
270
 
271
+ private get windowSeconds(): number {
272
+ return (this.options.windowMs ?? WINDOW_MS) / 1000;
273
+ }
274
+
275
+ private now(): number {
276
+ return (this.options.now ?? Date.now)();
277
+ }
278
+
177
279
  start(): boolean {
178
280
  const make = this.options.decoder ?? ((onPcm, onEnd) => ffmpegDecoder(this.options.ffmpeg, onPcm, onEnd));
179
281
  this.decoder = make((pcm) => this.onPcm(pcm), () => this.stop());
@@ -185,9 +287,29 @@ class Captioner {
185
287
  this.stop();
186
288
  return false;
187
289
  }
290
+ void this.consult("");
188
291
  return true;
189
292
  }
190
293
 
294
+ /** Ask the store what it already knows of this media in a language, once. */
295
+ private async consult(language: string): Promise<void> {
296
+ if (!this.transcriptId || this.asked.has(language)) return;
297
+ this.asked.add(language);
298
+ const session = this.options.session();
299
+ if (session === null) return;
300
+ const got = await fetchTranscript(session, this.transcriptId, language, this.options.fetcher ?? fetch);
301
+ if (this.stopped) return;
302
+ if (!got.ok) {
303
+ if (got.status !== 404) this.complain(`the store did not answer: ${got.error}`);
304
+ return;
305
+ }
306
+ this.known.set(language, got.body.lines);
307
+ if (language === "" && this.language === "" && got.body.language) this.language = got.body.language;
308
+ if (got.body.lines.length > 0) {
309
+ this.options.onEvent?.(`captions for "${this.id}": the store knows ${got.body.lines.length} lines of this${language ? ` in ${language}` : ""}`);
310
+ }
311
+ }
312
+
191
313
  private onPcm(pcm: Buffer): void {
192
314
  if (this.stopped) return;
193
315
  this.pending.push(pcm);
@@ -199,13 +321,29 @@ class Captioner {
199
321
  const rest = all.subarray(size);
200
322
  this.pending = rest.length > 0 ? [Buffer.from(rest)] : [];
201
323
  this.pendingBytes = rest.length;
202
- const until = (this.options.now ?? Date.now)();
203
- void this.hear(Buffer.from(window), until - (this.options.windowMs ?? WINDOW_MS), until);
324
+ const until = this.now();
325
+ const index = this.windows;
326
+ this.windows += 1;
327
+ void this.hear(Buffer.from(window), until - (this.options.windowMs ?? WINDOW_MS), until, index);
204
328
  }
205
329
  }
206
330
 
207
- private async hear(pcm: Buffer, at: number, until: number): Promise<void> {
331
+ private async hear(pcm: Buffer, at: number, until: number, index: number): Promise<void> {
208
332
  if (isQuiet(pcm)) return;
333
+ const span = this.media ? mediaSpan(this.media, index, this.windowSeconds, at, until) : null;
334
+ if (span) {
335
+ const stored = covered(this.known.get("") ?? [], span.start, span.end);
336
+ if (stored.length > 0) {
337
+ // The store has been through this moment: read it out, and let the ear rest.
338
+ for (const line of stored) {
339
+ if (this.readOut.has(line.start)) continue;
340
+ this.readOut.add(line.start);
341
+ const lineAt = at + (line.start - span.start) * 1000;
342
+ this.emit({ channel: this.id, at: lineAt, until: lineAt + (line.end - line.start) * 1000, text: line.text, ...(this.language ? { language: this.language } : {}) }, line.start);
343
+ }
344
+ return;
345
+ }
346
+ }
209
347
  if (this.inFlight >= IN_FLIGHT) return;
210
348
  const session = this.options.session();
211
349
  if (session === null) {
@@ -215,12 +353,14 @@ class Captioner {
215
353
  this.inFlight += 1;
216
354
  try {
217
355
  const wav = wavAround(pcm);
218
- const response = await (this.options.fetcher ?? fetch)(`${session.site.replace(/\/+$/, "")}/api/v1/speech/transcribe`, {
356
+ const url = new URL(`${session.site.replace(/\/+$/, "")}/api/v1/speech/transcribe`);
357
+ if (this.language) url.searchParams.set("language", this.language);
358
+ const response = await (this.options.fetcher ?? fetch)(url.toString(), {
219
359
  method: "POST",
220
360
  headers: { authorization: `Bearer ${session.token}`, "content-type": "audio/wav" },
221
361
  body: new Blob([wav.buffer.slice(wav.byteOffset, wav.byteOffset + wav.byteLength) as ArrayBuffer]),
222
362
  });
223
- const body = (await response.json().catch(() => ({}))) as { text?: string; error?: string };
363
+ const body = (await response.json().catch(() => ({}))) as { text?: string; language?: string; model?: string; error?: string };
224
364
  if (!response.ok) {
225
365
  this.complain(body.error ?? `nixamp.com answered ${response.status}`);
226
366
  return;
@@ -228,16 +368,11 @@ class Captioner {
228
368
  const text = (body.text ?? "").trim();
229
369
  if (text === "" || this.stopped) return;
230
370
  this.error = "";
231
- const line: CaptionLine = { channel: this.id, at, until, text };
232
- this.lines.push(line);
233
- while (this.lines.length > KEEP) this.lines.shift();
234
- for (const subscriber of this.subscribers) {
235
- try {
236
- subscriber(line);
237
- } catch {
238
- // A listener that throws is not this channel's problem.
239
- }
240
- }
371
+ if (body.language && this.language === "") this.language = body.language;
372
+ if (body.model) this.model = body.model;
373
+ const line: CaptionLine = { channel: this.id, at, until, text, ...(this.language ? { language: this.language } : {}) };
374
+ this.emit(line, span?.start ?? null);
375
+ if (span) this.keep("", { start: span.start, end: span.end, text });
241
376
  } catch (error) {
242
377
  this.complain(`could not reach the ear: ${(error as Error).message}`);
243
378
  } finally {
@@ -245,17 +380,112 @@ class Captioner {
245
380
  }
246
381
  }
247
382
 
383
+ /** A line as heard, to whoever wants the original, and translated to whoever wants another language. */
384
+ private emit(line: CaptionLine, mediaStart: number | null): void {
385
+ this.lines.push(line);
386
+ while (this.lines.length > KEEP) this.lines.shift();
387
+ for (const [subscriber, language] of this.subscribers) {
388
+ if (language === "" || language === this.language) this.tell(subscriber, line);
389
+ }
390
+ for (const language of this.wanted()) this.translated(language, line, mediaStart);
391
+ }
392
+
393
+ private tell(subscriber: Subscriber, line: CaptionLine): void {
394
+ try {
395
+ subscriber(line);
396
+ } catch {
397
+ // A listener that throws is not this channel's problem.
398
+ }
399
+ }
400
+
401
+ /** The languages somebody wants besides the one being heard. */
402
+ private wanted(): Set<string> {
403
+ const languages = new Set<string>();
404
+ for (const language of this.subscribers.values()) if (language !== "" && language !== this.language) languages.add(language);
405
+ return languages;
406
+ }
407
+
408
+ /** The line in another language: from the store when it has been through this moment, from nixamp.com otherwise. */
409
+ private translated(language: string, line: CaptionLine, mediaStart: number | null): void {
410
+ const chain = (this.chains.get(language) ?? Promise.resolve()).then(async () => {
411
+ if (this.stopped) return;
412
+ let text = "";
413
+ const stored = mediaStart === null ? null : lineAt(this.known.get(language) ?? [], mediaStart);
414
+ if (stored) {
415
+ text = stored.text;
416
+ } else {
417
+ const session = this.options.session();
418
+ if (session === null) return;
419
+ const got = await translateTexts(session, [line.text], this.language, language, this.options.fetcher ?? fetch);
420
+ if (!got.ok) {
421
+ this.complain(`could not translate to ${language}: ${got.error}`);
422
+ return;
423
+ }
424
+ text = (got.body.texts[0] ?? "").trim();
425
+ if (text === "") return;
426
+ if (mediaStart !== null) this.keep(language, { start: mediaStart, end: round(mediaStart + (line.until - line.at) / 1000), text });
427
+ }
428
+ if (this.stopped) return;
429
+ const said: CaptionLine = { ...line, text, language, original: line.text };
430
+ const lines = this.linesBy.get(language) ?? [];
431
+ lines.push(said);
432
+ while (lines.length > KEEP) lines.shift();
433
+ this.linesBy.set(language, lines);
434
+ for (const [subscriber, wanted] of this.subscribers) if (wanted === language) this.tell(subscriber, said);
435
+ });
436
+ this.chains.set(language, chain.catch(() => undefined));
437
+ }
438
+
439
+ /** A line for the store, kept with the others of its language until the next flush. */
440
+ private keep(language: string, line: TranscriptLine): void {
441
+ if (!this.media) return;
442
+ const lines = this.unsaved.get(language) ?? [];
443
+ lines.push(line);
444
+ this.unsaved.set(language, lines);
445
+ if (lines.length >= FLUSH_LINES) {
446
+ void this.flushNow();
447
+ return;
448
+ }
449
+ if (this.flush === null) {
450
+ this.flush = setTimeout(() => void this.flushNow(), this.options.flushMs ?? FLUSH_MS);
451
+ this.flush.unref?.();
452
+ }
453
+ }
454
+
455
+ /** Everything not yet kept, to the store. Never throws; a store that is away costs nothing but the keeping. */
456
+ private async flushNow(): Promise<void> {
457
+ if (this.flush) clearTimeout(this.flush);
458
+ this.flush = null;
459
+ if (!this.media || !this.transcriptId) return;
460
+ const session = this.options.session();
461
+ if (session === null) return;
462
+ const batches = [...this.unsaved.entries()].filter(([, lines]) => lines.length > 0);
463
+ this.unsaved.clear();
464
+ for (const [language, lines] of batches) {
465
+ const got = await keepLines(session, this.transcriptId, {
466
+ media: this.media.media,
467
+ title: this.media.title,
468
+ language: language === "" ? this.language : language,
469
+ ...(language === "" ? {} : { translatedFrom: this.language }),
470
+ ...(this.model ? { model: this.model } : {}),
471
+ lines,
472
+ }, this.options.fetcher ?? fetch);
473
+ if (!got.ok) this.complain(`the store did not keep ${lines.length} lines: ${got.error}`);
474
+ }
475
+ }
476
+
248
477
  /** Said once a minute at most: a broken ear would otherwise say so twelve times a minute. */
249
478
  private complain(message: string): void {
250
479
  this.error = message;
251
- const now = (this.options.now ?? Date.now)();
480
+ const now = this.now();
252
481
  if (now - this.complainedAt < 60_000) return;
253
482
  this.complainedAt = now;
254
483
  this.options.onEvent?.(`captions for "${this.id}": ${message}`);
255
484
  }
256
485
 
257
- subscribe(subscriber: Subscriber): () => void {
258
- this.subscribers.add(subscriber);
486
+ subscribe(subscriber: Subscriber, language = ""): () => void {
487
+ this.subscribers.set(subscriber, language);
488
+ if (language !== "") void this.consult(language);
259
489
  if (this.idle) clearTimeout(this.idle);
260
490
  this.idle = null;
261
491
  return () => {
@@ -269,6 +499,22 @@ class Captioner {
269
499
  };
270
500
  }
271
501
 
502
+ recent(after: number, language = ""): CaptionLine[] {
503
+ const lines = language === "" || language === this.language ? this.lines : (this.linesBy.get(language) ?? []);
504
+ return after > 0 ? lines.filter((line) => line.at > after) : [...lines];
505
+ }
506
+
507
+ status(): CaptionStatus {
508
+ return {
509
+ on: true,
510
+ lines: this.lines.length,
511
+ error: this.error,
512
+ language: this.language,
513
+ known: this.known.get("")?.length ?? 0,
514
+ languages: [...this.linesBy.keys()],
515
+ };
516
+ }
517
+
272
518
  stop(): void {
273
519
  if (this.stopped) return;
274
520
  this.stopped = true;
@@ -281,6 +527,7 @@ class Captioner {
281
527
  this.pending = [];
282
528
  this.pendingBytes = 0;
283
529
  this.subscribers.clear();
530
+ void this.flushNow();
284
531
  this.onStop();
285
532
  }
286
533
  }
@@ -299,9 +546,10 @@ export class Captions {
299
546
  * Lines for a channel as they are heard, starting the captioner if it is
300
547
  * not running. Null when there is no such channel. The returned function
301
548
  * is how to stop listening; the captioner itself stops a minute after the
302
- * last listener does.
549
+ * last listener does. A language asks for the lines translated into it;
550
+ * "" is the original.
303
551
  */
304
- subscribe(id: string, subscriber: Subscriber): (() => void) | null {
552
+ subscribe(id: string, subscriber: Subscriber, language = ""): (() => void) | null {
305
553
  let captioner = this.running.get(id);
306
554
  if (!captioner) {
307
555
  const made = new Captioner(id, this.options, () => {
@@ -312,20 +560,19 @@ export class Captions {
312
560
  this.options.onEvent?.(`captions for "${id}": started`);
313
561
  captioner = made;
314
562
  }
315
- return captioner.subscribe(subscriber);
563
+ return captioner.subscribe(subscriber, language);
316
564
  }
317
565
 
318
- /** The recent lines of a channel, oldest first, after a moment when given. Empty when nobody has asked for them. */
319
- recent(id: string, after = 0): CaptionLine[] {
566
+ /** The recent lines of a channel, oldest first, after a moment when given, in a language when asked. Empty when nobody has asked for them. */
567
+ recent(id: string, after = 0, language = ""): CaptionLine[] {
320
568
  const captioner = this.running.get(id);
321
- if (!captioner) return [];
322
- return after > 0 ? captioner.lines.filter((line) => line.at > after) : [...captioner.lines];
569
+ return captioner ? captioner.recent(after, language) : [];
323
570
  }
324
571
 
325
572
  /** Whether a channel is being captioned, and what last went wrong if the lines are not coming. */
326
- status(id: string): { on: boolean; lines: number; error: string } {
573
+ status(id: string): CaptionStatus {
327
574
  const captioner = this.running.get(id);
328
- return captioner ? { on: true, lines: captioner.lines.length, error: captioner.error } : { on: false, lines: 0, error: "" };
575
+ return captioner ? captioner.status() : { on: false, lines: 0, error: "", language: "", known: 0, languages: [] };
329
576
  }
330
577
 
331
578
  stopAll(): void {
package/src/main.ts CHANGED
@@ -84,8 +84,9 @@ const HELP = `nixamp — it really whips the terminal's ass.
84
84
  nixamp server list|add|remove the machines you run, kept against your account
85
85
  nixamp party list|join|host watch parties, here and on the sites nixamp is connected to
86
86
  nixamp mcp speak Model Context Protocol on stdin, for an agent
87
- nixamp transcribe FILE [--say SERVER] the words in a recording, and into a trollbox
88
- nixamp transcript --channel ID [--follow] what a channel is saying, as it says it
87
+ nixamp transcribe FILE [--translate sv] a recording or a whole film written down, kept, and in other languages
88
+ nixamp transcript --channel ID [--follow] what a channel is saying, as it says it; --kept for what nixamp.com keeps
89
+ nixamp translate --to sv TEXT say it in another language
89
90
  nixamp profile [--handle H] [--voice V] [--profile URL] who the rooms know you as
90
91
  nixamp voices the voices a line is read in on the phone
91
92
  nixamp opendir list|add|remove folders found on the web, published for everyone
@@ -267,31 +268,60 @@ takes it away again.
267
268
  nixamp mcp speak Model Context Protocol on stdin and stdout
268
269
 
269
270
  It offers the watch party tools: list them, read one, put one on the air,
270
- say where playback is, end it. And the room tools: transcribe a recording
271
- (transcribe_audio, which can post the words straight into a trollbox), say a
272
- line in a room (trollbox_say), read a room (trollbox_read), read what a
273
- channel is saying (transcript_read), and who you are in the rooms and how
274
- you sound on the phone (profile_get, profile_set, voices_list). It acts as
275
- whoever this machine is signed in as, so \`nixamp login\` (or NIXAMP_TOKEN)
276
- comes first.
277
-
278
- Point an MCP client at it as a stdio server running \`nixamp mcp\`.
271
+ say where playback is, end it. The room tools: transcribe a recording or a
272
+ whole film (transcribe_audio, kept on nixamp.com, translated on request, or
273
+ posted straight into a trollbox), say a line in a room (trollbox_say), read
274
+ a room (trollbox_read), read what a channel is saying (transcript_read, in
275
+ any language). The transcript tools: a kept transcript by its id or its
276
+ media (transcript_get), what this account has had written down
277
+ (transcripts_list), and text in another language (translate_text). And who
278
+ you are in the rooms and how you sound on the phone (profile_get,
279
+ profile_set, voices_list). It acts as whoever this machine is signed in as,
280
+ so \`nixamp login\` (or NIXAMP_TOKEN) comes first.
281
+
282
+ Point an MCP client at it as a stdio server running \`nixamp mcp\`. The same
283
+ tools are at https://nixamp.com/mcp over HTTP, with a nixamp token
284
+ (\`nixamp token create\`) as the bearer, for an agent with no nixamp installed.
279
285
  `,
280
286
  transcribe: `nixamp transcribe — say it, and have it written down.
281
287
 
282
- nixamp transcribe FILE the words in a recording
283
- nixamp transcribe FILE --say SERVER and post them to that server's trollbox
284
- nixamp transcribe FILE --say SERVER --channel ID to one channel's room (default: live)
285
- nixamp transcribe FILE --language de when Whisper should not guess
288
+ nixamp transcribe FILE the words in a recording, or a whole film, kept on nixamp.com
289
+ nixamp transcribe URL the same for a link ffmpeg can read
290
+ nixamp transcribe FILE --language sv when Whisper should not guess
291
+ nixamp transcribe FILE --translate de and in German too (de,sv for both)
292
+ nixamp transcribe FILE --srt | --vtt as subtitles, on stdout
293
+ nixamp transcribe FILE --out DIR subtitle files in DIR, one per language
294
+ nixamp transcribe FILE --fresh hear it again even though it is kept
286
295
  nixamp transcribe FILE --json the answer as JSON
296
+ nixamp transcribe CLIP --say SERVER a short clip, posted to that server's trollbox
297
+ nixamp transcribe CLIP --say SERVER --channel ID to one channel's room (default: live)
287
298
 
288
- FILE is any recording ffmpeg can read; a WAV needs no ffmpeg at all. The
289
- hearing is done by nixamp.com with an open-source model (Whisper, through
290
- Transformers.js) on its own CPU: nothing goes to a speech vendor. It needs a
291
- sign-in (\`nixamp login\`) and nothing else. Up to a minute at a time.
299
+ FILE is any recording ffmpeg can read; a WAV under a minute needs no ffmpeg
300
+ at all. The hearing is done by nixamp.com with an open-source model (Whisper,
301
+ through Transformers.js) on its own CPU: nothing goes to a speech vendor. It
302
+ needs a sign-in (\`nixamp login\`) and nothing else.
303
+
304
+ A film is heard a minute at a time, with when each line is said, and kept on
305
+ nixamp.com under the file's fingerprint: the next \`nixamp transcribe\` of the
306
+ same file, on any machine, and the next server to put it on the air, read
307
+ the lines instead of hearing them. A translation is made once, on nixamp.com
308
+ with an open-source model (OPUS-MT), and kept beside the original.
292
309
 
293
310
  The same ear is behind the microphone button in every nixamp.com trollbox,
294
311
  and behind the transcribe_audio tool of \`nixamp mcp\`.
312
+ `,
313
+ translate: `nixamp translate — say it in another language.
314
+
315
+ nixamp translate --to sv "Hello there" Swedish, from English
316
+ nixamp translate --from de --to en "Guten Tag"
317
+ cat lines.txt | nixamp translate --to de each line, in order
318
+ nixamp translate --languages what nixamp.com can translate between
319
+
320
+ The models are open-source (OPUS-MT, through Transformers.js) and run on
321
+ nixamp.com's own CPU; a pair with no model of its own goes through English.
322
+ Needs a sign-in (\`nixamp login\`). The same models turn a live's captions
323
+ into another language as they are said, and a kept transcript into one on
324
+ request.
295
325
  `,
296
326
  profile: `nixamp profile — who the rooms know you as.
297
327
 
@@ -312,13 +342,22 @@ one picked for the account and kept. Two people in a room are two voices.
312
342
  nixamp transcript --channel ID the recent lines from this machine's daemon
313
343
  nixamp transcript --url URL --key K --channel ID from another server, with its share link
314
344
  nixamp transcript ... --follow and keep printing as it speaks
345
+ nixamp transcript ... --language sv the lines in Swedish, translated as they are said
315
346
  nixamp transcript ... --json the lines as JSON
347
+ nixamp transcript --kept MEDIA_OR_ID [--language de] [--srt|--vtt|--txt]
348
+ a transcript nixamp.com keeps: a file's, a link's, a past live's
349
+ nixamp transcript --list what this account has had written down
316
350
 
317
351
  A server captions a channel while somebody is asking for its transcript: its
318
352
  own ffmpeg turns the sound into five-second windows, nixamp.com's ear turns
319
353
  those into lines, each stamped with when its sound was heard. The page shows
320
354
  them as subtitles, held until its own sound gets there; this prints them.
321
355
  The server needs an ffmpeg and a sign-in (\`nixamp login\`).
356
+
357
+ What it hears is kept on nixamp.com under what the channel is playing: a
358
+ film by its bytes, a link by its address, a live as the one broadcast it
359
+ was. A channel playing something already kept reads the lines instead of
360
+ hearing them, and a language asked for is translated once and kept too.
322
361
  `,
323
362
  attach: `nixamp attach — the player, in front of the running daemon.
324
363
 
@@ -544,6 +583,11 @@ export async function main(): Promise<void> {
544
583
  process.exitCode = await mcp();
545
584
  return;
546
585
  }
586
+ if (first === "translate") {
587
+ const { translate } = await import("./translate-cli.ts");
588
+ process.exitCode = await translate(rest);
589
+ return;
590
+ }
547
591
  if (first === "token" || first === "tokens") {
548
592
  const { tokens } = await import("./session.ts");
549
593
  process.exitCode = await tokens(rest);