nixamp 0.23.7 → 0.24.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +79 -3
  2. package/dist/captions.d.ts +57 -10
  3. package/dist/captions.js +253 -29
  4. package/dist/main.js +63 -19
  5. package/dist/mcp.d.ts +7 -1
  6. package/dist/mcp.js +159 -24
  7. package/dist/server.d.ts +11 -0
  8. package/dist/server.js +353 -8
  9. package/dist/speech.d.ts +21 -3
  10. package/dist/speech.js +65 -8
  11. package/dist/transcribe.d.ts +70 -0
  12. package/dist/transcribe.js +337 -41
  13. package/dist/transcript-client.d.ts +81 -0
  14. package/dist/transcript-client.js +76 -0
  15. package/dist/transcript.d.ts +7 -2
  16. package/dist/transcript.js +73 -5
  17. package/dist/transcripts.d.ts +135 -0
  18. package/dist/transcripts.js +384 -0
  19. package/dist/translate-cli.d.ts +16 -0
  20. package/dist/translate-cli.js +97 -0
  21. package/dist/translate-jobs.d.ts +39 -0
  22. package/dist/translate-jobs.js +120 -0
  23. package/dist/translate.d.ts +78 -0
  24. package/dist/translate.js +241 -0
  25. package/dist/warm.d.ts +1 -0
  26. package/dist/warm.js +40 -0
  27. package/package.json +1 -1
  28. package/src/captions.ts +276 -29
  29. package/src/main.ts +63 -19
  30. package/src/mcp.ts +156 -21
  31. package/src/server.ts +352 -8
  32. package/src/speech.ts +103 -13
  33. package/src/transcribe.ts +374 -41
  34. package/src/transcript-client.ts +147 -0
  35. package/src/transcript.ts +76 -4
  36. package/src/transcripts.ts +462 -0
  37. package/src/translate-cli.ts +111 -0
  38. package/src/translate-jobs.ts +132 -0
  39. package/src/translate.ts +276 -0
  40. package/src/warm.ts +43 -0
  41. package/web/dist/assets/{hls-3VKVEQE3-CI1U7kbP.js → hls-3VKVEQE3-Dtl-3mpW.js} +1 -1
  42. package/web/dist/assets/index-B0h4Nexr.js +1 -0
  43. package/web/dist/assets/index-BTGV3Pi5.css +1 -0
  44. package/web/dist/assets/{mpegts-CPOYjgRP.js → mpegts-BJC48bFV.js} +1 -1
  45. package/web/dist/assets/{mpegts-LO6RVLD6-CE8YPjx1.js → mpegts-LO6RVLD6-CUIAB9k3.js} +1 -1
  46. package/web/dist/index.html +12 -8
  47. package/web/dist/sw.js +6 -6
  48. package/web/dist/assets/index-46pwGn5-.css +0 -1
  49. package/web/dist/assets/index-5H3sUGHu.js +0 -1
@@ -1,9 +1,14 @@
1
1
  import { type Tools } from "./audio.ts";
2
2
  import { type Session } from "./session.ts";
3
+ import { type Segment } from "./speech.ts";
4
+ import { type Got, type StoredTranscript } from "./transcript-client.ts";
5
+ import { stamp, type TranscriptLine } from "./transcripts.ts";
3
6
  /** Where the sound may be sent, as a request. */
4
7
  export interface Ask {
5
8
  wav: Uint8Array;
6
9
  language?: string;
10
+ /** The pieces with their timing too. */
11
+ timestamps?: boolean;
7
12
  /** A room to post the words to: the server's address, and its channel. */
8
13
  server?: string;
9
14
  channel?: string;
@@ -12,6 +17,8 @@ export interface Heard {
12
17
  text: string;
13
18
  seconds: number;
14
19
  model?: string;
20
+ language?: string;
21
+ segments?: Segment[];
15
22
  /** The trollbox line, when a room was named. */
16
23
  message?: {
17
24
  id: string;
@@ -28,16 +35,79 @@ export type Answer = {
28
35
  status: number;
29
36
  error: string;
30
37
  };
38
+ /** A minute of sound at a time: the most the ear takes in one ask. */
39
+ export declare const WINDOW_SECONDS = 60;
40
+ /** Heard lines go to the store this often, so a run that dies keeps most of what it heard. */
41
+ export declare const KEEP_EVERY = 20;
31
42
  /**
32
43
  * The file as a WAV: as it is when it already is one, through ffmpeg to
33
44
  * 16 kHz mono otherwise. Throws a sentence when neither is possible.
34
45
  */
35
46
  export declare function wavOf(path: string, tools?: () => Pick<Tools, "ffmpeg" | "carries">): Uint8Array;
47
+ /** A window of sound: where it begins in the media, and 16-bit mono PCM at RATE. */
48
+ export interface Window {
49
+ offset: number;
50
+ pcm: Buffer;
51
+ }
52
+ /**
53
+ * The whole of a file or a link as windows of PCM, through ffmpeg, as it
54
+ * decodes: a film is hours of sound and is never held at once.
55
+ */
56
+ export declare function windowsOf(source: string, tools?: () => Pick<Tools, "ffmpeg" | "carries">, seconds?: number): AsyncGenerator<Window>;
36
57
  /** The ask, made: one POST to the site the session belongs to. */
37
58
  export declare function askToHear(session: Pick<Session, "site" | "token">, ask: Ask, fetcher?: typeof fetch, site?: string): Promise<Answer>;
59
+ export interface WholeHeard {
60
+ lines: TranscriptLine[];
61
+ language: string;
62
+ model: string;
63
+ /** How far the sound went, seconds. */
64
+ seconds: number;
65
+ }
66
+ export interface WholeDeps {
67
+ fetcher?: typeof fetch;
68
+ windows?: (source: string) => AsyncIterable<Window>;
69
+ sleep?: (ms: number) => Promise<void>;
70
+ onProgress?: (line: string) => void;
71
+ }
72
+ /** m:ss, or h:mm:ss, for a progress line. */
73
+ export declare function clock(seconds: number): string;
74
+ /**
75
+ * The whole of some media, heard a window at a time and kept as it goes.
76
+ * The ear's throttle is so many seconds of sound a minute; a 429 is a
77
+ * wait, not a failure. Quiet windows are skipped without an ask.
78
+ */
79
+ export declare function hearWhole(session: Pick<Session, "site" | "token">, source: string, media: string, options?: {
80
+ language?: string;
81
+ title?: string;
82
+ }, deps?: WholeDeps): Promise<{
83
+ ok: true;
84
+ heard: WholeHeard;
85
+ } | {
86
+ ok: false;
87
+ error: string;
88
+ }>;
89
+ /** A kept transcript in a language, waiting while nixamp.com translates it. */
90
+ export declare function awaitTranscript(session: Pick<Session, "site" | "token">, id: string, language: string, deps?: {
91
+ fetcher?: typeof fetch;
92
+ sleep?: (ms: number) => Promise<void>;
93
+ onProgress?: (line: string) => void;
94
+ polls?: number;
95
+ }): Promise<Got<StoredTranscript>>;
96
+ /** A line as the terminal prints it: seconds into the media, then the words. */
97
+ export declare function printedLine(line: TranscriptLine): string;
38
98
  export interface TranscribeDeps {
39
99
  fetcher?: typeof fetch;
40
100
  wavOf?: typeof wavOf;
41
101
  session?: Pick<Session, "site" | "token"> | null;
102
+ /** The whole of a file as windows; the tests hand in a fake. */
103
+ windows?: (source: string) => AsyncIterable<Window>;
104
+ /** A file's identity; the tests hand in a fake. */
105
+ fingerprint?: (path: string) => string;
106
+ sleep?: (ms: number) => Promise<void>;
107
+ /** How many times a translation is asked about before giving up; forever, except in the tests. */
108
+ polls?: number;
42
109
  }
110
+ /** A transcript's lines as one of the formats, for stdout or a file. */
111
+ export declare function rendered(lines: TranscriptLine[], format: "srt" | "vtt" | "txt" | "lines"): string;
43
112
  export declare function transcribe(argv: string[], deps?: TranscribeDeps): Promise<number>;
113
+ export { stamp };
@@ -1,38 +1,60 @@
1
1
  /**
2
- * `nixamp transcribe` -- a recording in, the words out, and into a room.
2
+ * `nixamp transcribe` -- a recording in, the words out, kept, and into a room.
3
3
  *
4
- * The ear is nixamp.com's (see speech.ts): this machine sends a WAV and
5
- * gets text back, signed in as whoever it is signed in as. Anything that
6
- * is not already a WAV goes through the ffmpeg nixamp plays with, to 16 kHz
7
- * mono, which is what the ear listens at and the smallest thing to send.
4
+ * The ear is nixamp.com's (see speech.ts): this machine sends WAVs and gets
5
+ * text back, signed in as whoever it is signed in as. Anything that is not
6
+ * already a WAV goes through the ffmpeg nixamp plays with, to 16 kHz mono,
7
+ * which is what the ear listens at and the smallest thing to send.
8
+ *
9
+ * A short clip is one ask. A whole film is the same ask a hundred times, a
10
+ * minute of sound each, with the pieces' timing asked for, and what comes
11
+ * back is kept on nixamp.com under the file's fingerprint (see
12
+ * transcripts.ts): the next `nixamp transcribe` of the same file, on any
13
+ * machine, and the next server to put it on the air, read the lines
14
+ * instead of hearing them. Ask for another language and nixamp.com
15
+ * translates the kept lines once, and keeps that too.
8
16
  *
9
17
  * With `--say SERVER`, the words are posted to that server's trollbox as
10
18
  * this account, by the same rules as typing them: one line a second, and
11
19
  * signed with the public handle. The MCP tool of the same name is this
12
20
  * function with a different front.
13
21
  */
14
- import { spawnSync } from "node:child_process";
15
- import { readFileSync } from "node:fs";
22
+ import { spawn, spawnSync } from "node:child_process";
23
+ import { mkdirSync, readFileSync, writeFileSync } from "node:fs";
24
+ import { basename, extname, join } from "node:path";
16
25
  import { detectTools } from "./audio.js";
26
+ import { isQuiet } from "./captions.js";
17
27
  import { readSession } from "./session.js";
18
- import { RATE, isWav } from "./speech.js";
28
+ import { MAX_SECONDS, RATE, isWav } from "./speech.js";
29
+ import { fetchTranscript, keepLines } from "./transcript-client.js";
30
+ import { fileFingerprint, languageCode, mediaOfUrl, stamp, toSrt, toText, toVtt, transcriptIdOf } from "./transcripts.js";
19
31
  const HELP = `nixamp transcribe — say it, and have it written down.
20
32
 
21
- nixamp transcribe FILE the words in a recording
22
- nixamp transcribe FILE --say SERVER and post them to that server's trollbox
23
- nixamp transcribe FILE --say SERVER --channel ID to one channel's room (default: live)
24
- nixamp transcribe FILE --language de when Whisper should not guess
33
+ nixamp transcribe FILE the words in a recording, or a whole film, kept on nixamp.com
34
+ nixamp transcribe URL the same for a link ffmpeg can read
35
+ nixamp transcribe FILE --language sv when Whisper should not guess
36
+ nixamp transcribe FILE --translate de and in German too (de,sv for both)
37
+ nixamp transcribe FILE --srt | --vtt as subtitles, on stdout
38
+ nixamp transcribe FILE --out DIR subtitle files in DIR, one per language
39
+ nixamp transcribe FILE --fresh hear it again even though it is kept
25
40
  nixamp transcribe FILE --json the answer as JSON
41
+ nixamp transcribe CLIP --say SERVER a short clip, posted to that server's trollbox
42
+ nixamp transcribe CLIP --say SERVER --channel ID to one channel's room (default: live)
26
43
 
27
- FILE is any recording ffmpeg can read; a WAV needs no ffmpeg at all. The
28
- hearing is done by nixamp.com with an open-source model on its own CPU, so
29
- this needs a sign-in (\`nixamp login\`) and nothing else. Up to a minute at
30
- a time.
44
+ FILE is any recording ffmpeg can read; a WAV under a minute needs no ffmpeg
45
+ at all. The hearing is done by nixamp.com with an open-source model on its
46
+ own CPU, so this needs a sign-in (\`nixamp login\`) and nothing else. A film
47
+ is heard a minute at a time and kept as it goes; the next ask for the same
48
+ file, from anywhere, reads what was kept.
31
49
 
32
50
  SERVER is the address of the nixamp whose room it is, as in its share link:
33
51
  https://server1.chovy.nixamp.com:4321. The room is that server's own stream
34
52
  unless --channel names one of its channels.
35
53
  `;
54
+ /** A minute of sound at a time: the most the ear takes in one ask. */
55
+ export const WINDOW_SECONDS = MAX_SECONDS;
56
+ /** Heard lines go to the store this often, so a run that dies keeps most of what it heard. */
57
+ export const KEEP_EVERY = 20;
36
58
  /**
37
59
  * The file as a WAV: as it is when it already is one, through ffmpeg to
38
60
  * 16 kHz mono otherwise. Throws a sentence when neither is possible.
@@ -60,11 +82,58 @@ export function wavOf(path, tools = detectTools) {
60
82
  }
61
83
  return new Uint8Array(run.stdout);
62
84
  }
85
+ /**
86
+ * The whole of a file or a link as windows of PCM, through ffmpeg, as it
87
+ * decodes: a film is hours of sound and is never held at once.
88
+ */
89
+ export async function* windowsOf(source, tools = detectTools, seconds = WINDOW_SECONDS) {
90
+ const found = tools();
91
+ if (found.carries === false)
92
+ throw new Error(`there is no ffmpeg here to read ${source}. Install ffmpeg.`);
93
+ const [command, ...prefix] = found.ffmpeg;
94
+ const child = spawn(command, [...prefix, "-v", "error", "-nostats", "-i", source, "-vn", "-ac", "1", "-ar", String(RATE), "-f", "s16le", "pipe:1"], {
95
+ stdio: ["ignore", "pipe", "pipe"],
96
+ });
97
+ let complaint = "";
98
+ child.stderr.on("data", (chunk) => {
99
+ complaint = `${complaint}${chunk.toString("utf8")}`.slice(-2000);
100
+ });
101
+ const size = seconds * RATE * 2;
102
+ let pending = [];
103
+ let pendingBytes = 0;
104
+ let offset = 0;
105
+ for await (const chunk of child.stdout) {
106
+ pending.push(chunk);
107
+ pendingBytes += chunk.length;
108
+ while (pendingBytes >= size) {
109
+ const all = Buffer.concat(pending);
110
+ yield { offset, pcm: Buffer.from(all.subarray(0, size)) };
111
+ offset += seconds;
112
+ const rest = all.subarray(size);
113
+ pending = rest.length > 0 ? [Buffer.from(rest)] : [];
114
+ pendingBytes = rest.length;
115
+ }
116
+ }
117
+ // The tail: anything longer than half a second is worth hearing.
118
+ if (pendingBytes >= RATE)
119
+ yield { offset, pcm: Buffer.concat(pending) };
120
+ const status = await new Promise((done) => {
121
+ if (child.exitCode !== null)
122
+ done(child.exitCode);
123
+ else
124
+ child.on("close", (code) => done(code));
125
+ });
126
+ if (status !== 0 && offset === 0 && pendingBytes < RATE) {
127
+ throw new Error(`ffmpeg could not read ${source}${complaint ? `: ${complaint.trim().split("\n").pop()}` : ""}`);
128
+ }
129
+ }
63
130
  /** The ask, made: one POST to the site the session belongs to. */
64
131
  export async function askToHear(session, ask, fetcher = fetch, site = session.site) {
65
132
  const url = new URL(`${site.replace(/\/+$/, "")}/api/v1/speech/transcribe`);
66
133
  if (ask.language)
67
134
  url.searchParams.set("language", ask.language);
135
+ if (ask.timestamps)
136
+ url.searchParams.set("timestamps", "1");
68
137
  if (ask.server) {
69
138
  url.searchParams.set("server", ask.server);
70
139
  url.searchParams.set("channel", ask.channel || "live");
@@ -89,14 +158,148 @@ export async function askToHear(session, ask, fetcher = fetch, site = session.si
89
158
  text: body.text ?? "",
90
159
  seconds: body.seconds ?? 0,
91
160
  ...(body.model ? { model: body.model } : {}),
161
+ ...(body.language ? { language: body.language } : {}),
162
+ ...(body.segments ? { segments: body.segments } : {}),
92
163
  ...(body.message ? { message: body.message } : {}),
93
164
  },
94
165
  };
95
166
  }
167
+ /** A WAV around 16-bit mono PCM at RATE, for one window. */
168
+ function wavOfPcm(pcm) {
169
+ const header = Buffer.alloc(44);
170
+ header.write("RIFF", 0, "ascii");
171
+ header.writeUInt32LE(36 + pcm.length, 4);
172
+ header.write("WAVE", 8, "ascii");
173
+ header.write("fmt ", 12, "ascii");
174
+ header.writeUInt32LE(16, 16);
175
+ header.writeUInt16LE(1, 20);
176
+ header.writeUInt16LE(1, 22);
177
+ header.writeUInt32LE(RATE, 24);
178
+ header.writeUInt32LE(RATE * 2, 28);
179
+ header.writeUInt16LE(2, 32);
180
+ header.writeUInt16LE(16, 34);
181
+ header.write("data", 36, "ascii");
182
+ header.writeUInt32LE(pcm.length, 40);
183
+ return new Uint8Array(Buffer.concat([header, pcm]));
184
+ }
185
+ /** m:ss, or h:mm:ss, for a progress line. */
186
+ export function clock(seconds) {
187
+ const whole = Math.max(0, Math.floor(seconds));
188
+ const h = Math.floor(whole / 3600);
189
+ const m = Math.floor((whole % 3600) / 60);
190
+ const s = whole % 60;
191
+ return h > 0 ? `${h}:${String(m).padStart(2, "0")}:${String(s).padStart(2, "0")}` : `${m}:${String(s).padStart(2, "0")}`;
192
+ }
193
+ /**
194
+ * The whole of some media, heard a window at a time and kept as it goes.
195
+ * The ear's throttle is so many seconds of sound a minute; a 429 is a
196
+ * wait, not a failure. Quiet windows are skipped without an ask.
197
+ */
198
+ export async function hearWhole(session, source, media, options = {}, deps = {}) {
199
+ const fetcher = deps.fetcher ?? fetch;
200
+ const sleep = deps.sleep ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms)));
201
+ const windows = deps.windows ?? ((path) => windowsOf(path));
202
+ const id = transcriptIdOf(media);
203
+ const lines = [];
204
+ let unsaved = [];
205
+ let language = options.language ?? "";
206
+ let model = "";
207
+ let seconds = 0;
208
+ let sinceKept = 0;
209
+ const keep = async (complete) => {
210
+ const batch = complete ? lines : unsaved;
211
+ if (batch.length === 0 && !complete)
212
+ return null;
213
+ const got = await keepLines(session, id, {
214
+ media, language, ...(model ? { model } : {}), ...(options.title ? { title: options.title } : {}), lines: batch, ...(complete ? { complete: true } : {}),
215
+ }, fetcher);
216
+ unsaved = [];
217
+ sinceKept = 0;
218
+ return got.ok ? null : got.error;
219
+ };
220
+ try {
221
+ for await (const window of windows(source)) {
222
+ seconds = window.offset + window.pcm.length / (RATE * 2);
223
+ if (isQuiet(window.pcm))
224
+ continue;
225
+ const wav = wavOfPcm(window.pcm);
226
+ let answer = { ok: false, status: 0, error: "not asked" };
227
+ for (let attempt = 0; attempt < 40; attempt++) {
228
+ answer = await askToHear(session, { wav, timestamps: true, ...(language ? { language } : {}) }, fetcher);
229
+ if (answer.ok || answer.status !== 429)
230
+ break;
231
+ deps.onProgress?.(` ${clock(window.offset)}: the ear is busy; waiting`);
232
+ await sleep(10_000);
233
+ }
234
+ if (!answer.ok)
235
+ return { ok: false, error: `at ${clock(window.offset)}: ${answer.error}` };
236
+ if (answer.heard.language && language === "")
237
+ language = answer.heard.language;
238
+ if (answer.heard.model)
239
+ model = answer.heard.model;
240
+ const pieces = answer.heard.segments && answer.heard.segments.length > 0
241
+ ? answer.heard.segments
242
+ : answer.heard.text ? [{ start: 0, end: window.pcm.length / (RATE * 2), text: answer.heard.text }] : [];
243
+ for (const piece of pieces) {
244
+ const line = { start: round(window.offset + piece.start), end: round(window.offset + piece.end), text: piece.text };
245
+ lines.push(line);
246
+ unsaved.push(line);
247
+ }
248
+ deps.onProgress?.(` ${clock(seconds)} heard${language ? ` (${language})` : ""}: ${lines.length} lines`);
249
+ sinceKept += 1;
250
+ if (sinceKept >= KEEP_EVERY) {
251
+ const failed = await keep(false);
252
+ if (failed)
253
+ deps.onProgress?.(` the store did not keep the lines so far: ${failed}`);
254
+ }
255
+ }
256
+ }
257
+ catch (error) {
258
+ return { ok: false, error: error.message };
259
+ }
260
+ const failed = await keep(true);
261
+ if (failed)
262
+ return { ok: false, error: `heard it all, but nixamp.com would not keep it: ${failed}` };
263
+ return { ok: true, heard: { lines, language, model, seconds } };
264
+ }
265
+ function round(seconds) {
266
+ return Math.round(seconds * 1000) / 1000;
267
+ }
268
+ /** A kept transcript in a language, waiting while nixamp.com translates it. */
269
+ export async function awaitTranscript(session, id, language, deps = {}) {
270
+ const sleep = deps.sleep ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms)));
271
+ let last = -1;
272
+ for (let poll = 0;; poll++) {
273
+ const got = await fetchTranscript(session, id, language, deps.fetcher ?? fetch);
274
+ if (!got.ok || !got.body.translating)
275
+ return got;
276
+ const { done, total } = got.body.translating;
277
+ if (done !== last)
278
+ deps.onProgress?.(` translating to ${language}: ${done} of ${total} lines`);
279
+ last = done;
280
+ if (deps.polls !== undefined && poll + 1 >= deps.polls)
281
+ return got;
282
+ await sleep(3000);
283
+ }
284
+ }
285
+ /** A line as the terminal prints it: seconds into the media, then the words. */
286
+ export function printedLine(line) {
287
+ return `${clock(line.start).padStart(7)} ${line.text}`;
288
+ }
96
289
  function flag(argv, name) {
97
290
  const at = argv.indexOf(name);
98
291
  return at === -1 ? undefined : argv[at + 1];
99
292
  }
293
+ /** A transcript's lines as one of the formats, for stdout or a file. */
294
+ export function rendered(lines, format) {
295
+ if (format === "srt")
296
+ return toSrt(lines);
297
+ if (format === "vtt")
298
+ return toVtt(lines);
299
+ if (format === "txt")
300
+ return toText(lines);
301
+ return lines.map(printedLine).join("\n");
302
+ }
100
303
  export async function transcribe(argv, deps = {}) {
101
304
  if (argv.length === 0 || argv.includes("--help") || argv.includes("-h") || argv[0] === "help") {
102
305
  console.log(HELP);
@@ -107,7 +310,7 @@ export async function transcribe(argv, deps = {}) {
107
310
  console.error("nixamp: not signed in. Try `nixamp login`.");
108
311
  return 1;
109
312
  }
110
- const withValue = new Set(["--say", "--channel", "--language", "--site"]);
313
+ const withValue = new Set(["--say", "--channel", "--language", "--site", "--translate", "--out"]);
111
314
  const file = argv.find((one, at) => !one.startsWith("-") && !(at > 0 && withValue.has(argv[at - 1])));
112
315
  if (!file) {
113
316
  console.error("nixamp: which recording? `nixamp transcribe clip.m4a`.");
@@ -118,38 +321,131 @@ export async function transcribe(argv, deps = {}) {
118
321
  console.error("nixamp: --say needs the server's address, as in its share link.");
119
322
  return 64;
120
323
  }
121
- let wav;
324
+ const language = languageCode(flag(argv, "--language")) || undefined;
325
+ if (argv.includes("--language") && !language) {
326
+ console.error("nixamp: --language is a two-letter code, such as sv.");
327
+ return 64;
328
+ }
329
+ const site = flag(argv, "--site") ?? session.site;
330
+ const signed = { site, token: session.token };
331
+ const fetcher = deps.fetcher ?? fetch;
332
+ const asJson = argv.includes("--json");
333
+ const format = argv.includes("--srt") ? "srt" : argv.includes("--vtt") ? "vtt" : argv.includes("--txt") ? "txt" : "lines";
334
+ // A clip for a room: one ask, the words into the trollbox, as before.
335
+ if (server) {
336
+ let wav;
337
+ try {
338
+ wav = (deps.wavOf ?? wavOf)(file);
339
+ }
340
+ catch (error) {
341
+ console.error(`nixamp: ${error.message}`);
342
+ return 1;
343
+ }
344
+ const answer = await askToHear(signed, { wav, ...(language ? { language } : {}), server, channel: flag(argv, "--channel") ?? "live" }, fetcher);
345
+ if (!answer.ok) {
346
+ console.error(`nixamp: ${answer.error}`);
347
+ return 1;
348
+ }
349
+ if (asJson) {
350
+ console.log(JSON.stringify(answer.heard, null, 2));
351
+ return 0;
352
+ }
353
+ if (answer.heard.text === "") {
354
+ console.error("nixamp: heard nothing in that.");
355
+ return 1;
356
+ }
357
+ console.log(answer.heard.text);
358
+ if (answer.heard.message)
359
+ console.error(` Said in the room for ${flag(argv, "--channel") ?? "live"} at ${server} as ${answer.heard.message.handle}.`);
360
+ else
361
+ console.error(" Nothing was posted: there were no words to post.");
362
+ return 0;
363
+ }
364
+ // The whole thing, kept under what it is.
365
+ let media;
122
366
  try {
123
- wav = (deps.wavOf ?? wavOf)(file);
367
+ media = /^https?:\/\//.test(file) ? mediaOfUrl(file) : (deps.fingerprint ?? fileFingerprint)(file);
124
368
  }
125
- catch (error) {
126
- console.error(`nixamp: ${error.message}`);
369
+ catch {
370
+ console.error(`nixamp: cannot read ${file}`);
127
371
  return 1;
128
372
  }
129
- const ask = {
130
- wav,
131
- ...(flag(argv, "--language") ? { language: flag(argv, "--language") } : {}),
132
- ...(server ? { server, channel: flag(argv, "--channel") ?? "live" } : {}),
133
- };
134
- const answer = await askToHear(session, ask, deps.fetcher ?? fetch, flag(argv, "--site") ?? session.site);
135
- if (!answer.ok) {
136
- console.error(`nixamp: ${answer.error}`);
137
- return 1;
373
+ const id = transcriptIdOf(media);
374
+ const title = /^https?:\/\//.test(file) ? file : basename(file, extname(file));
375
+ const wanted = (flag(argv, "--translate") ?? "").split(",").map((one) => languageCode(one)).filter((one) => typeof one === "string" && one !== "");
376
+ if (argv.includes("--translate") && wanted.length === 0) {
377
+ console.error("nixamp: --translate is one or more two-letter codes, such as de,sv.");
378
+ return 64;
138
379
  }
139
- if (argv.includes("--json")) {
140
- console.log(JSON.stringify(answer.heard, null, 2));
141
- return 0;
380
+ const progress = (line) => console.error(line);
381
+ let original = null;
382
+ if (!argv.includes("--fresh")) {
383
+ const kept = await fetchTranscript(signed, id, "", fetcher);
384
+ if (kept.ok && kept.body.complete) {
385
+ original = kept.body;
386
+ progress(` Already written down (${original.lines.length} lines${original.language ? `, ${original.language}` : ""}); --fresh hears it again.`);
387
+ }
388
+ else if (!kept.ok && kept.status !== 404) {
389
+ console.error(`nixamp: ${kept.error}`);
390
+ return 1;
391
+ }
142
392
  }
143
- if (answer.heard.text === "") {
144
- console.error("nixamp: heard nothing in that.");
145
- return 1;
393
+ if (!original) {
394
+ progress(` Hearing ${title}, a minute at a time...`);
395
+ const heard = await hearWhole(signed, file, media, { ...(language ? { language } : {}), title }, {
396
+ fetcher, onProgress: progress, ...(deps.windows ? { windows: deps.windows } : {}), ...(deps.sleep ? { sleep: deps.sleep } : {}),
397
+ });
398
+ if (!heard.ok) {
399
+ console.error(`nixamp: ${heard.error}`);
400
+ return 1;
401
+ }
402
+ if (heard.heard.lines.length === 0) {
403
+ console.error("nixamp: heard nothing in that.");
404
+ return 1;
405
+ }
406
+ const kept = await fetchTranscript(signed, id, "", fetcher);
407
+ original = kept.ok ? kept.body : {
408
+ id, media, kind: "file", language: heard.heard.language, translatedFrom: null, model: heard.heard.model, complete: true, title,
409
+ seconds: heard.heard.seconds, updatedAt: "", lines: heard.heard.lines, languages: [],
410
+ };
411
+ progress(` Kept on nixamp.com as ${id.slice(0, 12)}: ${original.lines.length} lines${original.language ? ` of ${original.language}` : ""}.`);
146
412
  }
147
- console.log(answer.heard.text);
148
- if (answer.heard.message) {
149
- console.error(` Said in the room for ${ask.channel} at ${server} as ${answer.heard.message.handle}.`);
413
+ const versions = [original];
414
+ for (const to of wanted) {
415
+ if (to === original.language)
416
+ continue;
417
+ const got = await awaitTranscript(signed, id, to, { fetcher, onProgress: progress, ...(deps.sleep ? { sleep: deps.sleep } : {}), ...(deps.polls !== undefined ? { polls: deps.polls } : {}) });
418
+ if (!got.ok) {
419
+ console.error(`nixamp: could not get it in ${to}: ${got.error}`);
420
+ return 1;
421
+ }
422
+ if (got.body.translating) {
423
+ console.error(`nixamp: ${to} is still being translated (${got.body.translating.done} of ${got.body.translating.total}); ask again in a moment.`);
424
+ return 1;
425
+ }
426
+ versions.push(got.body);
150
427
  }
151
- else if (server) {
152
- console.error(" Nothing was posted: there were no words to post.");
428
+ const out = flag(argv, "--out");
429
+ if (out) {
430
+ mkdirSync(out, { recursive: true });
431
+ const stem = title.replace(/[^\w.-]+/g, "_").slice(0, 80) || "transcript";
432
+ const extension = format === "lines" ? "txt" : format;
433
+ for (const version of versions) {
434
+ const path = join(out, `${stem}${version.language ? `.${version.language}` : ""}.${asJson ? "json" : extension}`);
435
+ writeFileSync(path, asJson ? JSON.stringify(version, null, 2) : rendered(version.lines, format === "lines" ? "txt" : format));
436
+ console.log(path);
437
+ }
438
+ return 0;
153
439
  }
440
+ if (asJson) {
441
+ console.log(JSON.stringify(versions.length === 1 ? versions[0] : versions, null, 2));
442
+ return 0;
443
+ }
444
+ versions.forEach((version, i) => {
445
+ if (versions.length > 1)
446
+ console.error(i === 0 ? `-- ${version.language || "original"} --` : `\n-- ${version.language} --`);
447
+ console.log(rendered(version.lines, format));
448
+ });
154
449
  return 0;
155
450
  }
451
+ export { stamp };
@@ -0,0 +1,81 @@
1
+ /**
2
+ * The transcript store and the translator, as a client sees them.
3
+ *
4
+ * Both live on nixamp.com (see transcripts.ts and translate.ts); a server
5
+ * captioning a channel, the CLI writing a film down, and the MCP tools all
6
+ * talk to them through these few calls, signed in as whoever they are.
7
+ * Every answer is a plain object or a sentence, never a throw, because a
8
+ * store that is briefly away must cost the memory and not the captions.
9
+ */
10
+ import type { TranscriptLine } from "./transcripts.ts";
11
+ export interface Signed {
12
+ site: string;
13
+ token: string;
14
+ }
15
+ export interface StoredTranscript {
16
+ id: string;
17
+ media: string;
18
+ kind: string;
19
+ language: string;
20
+ translatedFrom: string | null;
21
+ model: string;
22
+ complete: boolean;
23
+ title: string;
24
+ seconds: number;
25
+ updatedAt: string;
26
+ lines: TranscriptLine[];
27
+ languages: {
28
+ language: string;
29
+ translatedFrom: string | null;
30
+ lines: number;
31
+ complete: boolean;
32
+ }[];
33
+ /** A translation that is still being made: how far it has got. */
34
+ translating?: {
35
+ done: number;
36
+ total: number;
37
+ };
38
+ }
39
+ export type Got<T> = {
40
+ ok: true;
41
+ body: T;
42
+ } | {
43
+ ok: false;
44
+ status: number;
45
+ error: string;
46
+ };
47
+ /** The transcript of some media in a language ("" for the original), if the store has one. */
48
+ export declare function fetchTranscript(signed: Signed, id: string, language?: string, fetcher?: typeof fetch): Promise<Got<StoredTranscript>>;
49
+ export interface LinesToKeep {
50
+ media: string;
51
+ language: string;
52
+ translatedFrom?: string | null;
53
+ model?: string;
54
+ title?: string;
55
+ complete?: boolean;
56
+ lines: TranscriptLine[];
57
+ }
58
+ /** Keep lines: appended to what the store has, or the whole thing when `complete`. */
59
+ export declare function keepLines(signed: Signed, id: string, ask: LinesToKeep, fetcher?: typeof fetch): Promise<Got<{
60
+ saved: number;
61
+ seconds: number;
62
+ complete: boolean;
63
+ }>>;
64
+ export interface Translated {
65
+ texts: string[];
66
+ from: string;
67
+ to: string;
68
+ model: string;
69
+ }
70
+ /** Texts in another language. `from` may be "" when nixamp.com should tell. */
71
+ export declare function translateTexts(signed: Signed, texts: string[], from: string, to: string, fetcher?: typeof fetch): Promise<Got<Translated>>;
72
+ /** Forget some media's transcripts. Only whoever stored them may. */
73
+ export declare function forgetTranscript(signed: Signed, id: string, fetcher?: typeof fetch): Promise<Got<{
74
+ ok: true;
75
+ }>>;
76
+ /** What this account has had written down. */
77
+ export declare function listTranscripts(signed: Signed, fetcher?: typeof fetch): Promise<Got<{
78
+ transcripts: (Omit<StoredTranscript, "lines" | "languages"> & {
79
+ lines: number;
80
+ })[];
81
+ }>>;