nixamp 0.23.7 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +79 -3
- package/dist/captions.d.ts +57 -10
- package/dist/captions.js +253 -29
- package/dist/main.js +63 -19
- package/dist/mcp.d.ts +7 -1
- package/dist/mcp.js +159 -24
- package/dist/server.d.ts +11 -0
- package/dist/server.js +353 -8
- package/dist/speech.d.ts +21 -3
- package/dist/speech.js +65 -8
- package/dist/transcribe.d.ts +70 -0
- package/dist/transcribe.js +337 -41
- package/dist/transcript-client.d.ts +81 -0
- package/dist/transcript-client.js +76 -0
- package/dist/transcript.d.ts +7 -2
- package/dist/transcript.js +73 -5
- package/dist/transcripts.d.ts +135 -0
- package/dist/transcripts.js +384 -0
- package/dist/translate-cli.d.ts +16 -0
- package/dist/translate-cli.js +97 -0
- package/dist/translate-jobs.d.ts +39 -0
- package/dist/translate-jobs.js +120 -0
- package/dist/translate.d.ts +78 -0
- package/dist/translate.js +241 -0
- package/dist/warm.d.ts +1 -0
- package/dist/warm.js +40 -0
- package/package.json +1 -1
- package/src/captions.ts +276 -29
- package/src/main.ts +63 -19
- package/src/mcp.ts +156 -21
- package/src/server.ts +352 -8
- package/src/speech.ts +103 -13
- package/src/transcribe.ts +374 -41
- package/src/transcript-client.ts +147 -0
- package/src/transcript.ts +76 -4
- package/src/transcripts.ts +462 -0
- package/src/translate-cli.ts +111 -0
- package/src/translate-jobs.ts +132 -0
- package/src/translate.ts +276 -0
- package/src/warm.ts +43 -0
- package/web/dist/assets/{hls-3VKVEQE3-CI1U7kbP.js → hls-3VKVEQE3-Dtl-3mpW.js} +1 -1
- package/web/dist/assets/index-B0h4Nexr.js +1 -0
- package/web/dist/assets/index-BTGV3Pi5.css +1 -0
- package/web/dist/assets/{mpegts-CPOYjgRP.js → mpegts-BJC48bFV.js} +1 -1
- package/web/dist/assets/{mpegts-LO6RVLD6-CE8YPjx1.js → mpegts-LO6RVLD6-CUIAB9k3.js} +1 -1
- package/web/dist/index.html +12 -8
- package/web/dist/sw.js +6 -6
- package/web/dist/assets/index-46pwGn5-.css +0 -1
- package/web/dist/assets/index-5H3sUGHu.js +0 -1
package/dist/main.js
CHANGED
|
@@ -60,8 +60,9 @@ const HELP = `nixamp — it really whips the terminal's ass.
|
|
|
60
60
|
nixamp server list|add|remove the machines you run, kept against your account
|
|
61
61
|
nixamp party list|join|host watch parties, here and on the sites nixamp is connected to
|
|
62
62
|
nixamp mcp speak Model Context Protocol on stdin, for an agent
|
|
63
|
-
nixamp transcribe FILE [--
|
|
64
|
-
nixamp transcript --channel ID [--follow] what a channel is saying, as it says it
|
|
63
|
+
nixamp transcribe FILE [--translate sv] a recording or a whole film written down, kept, and in other languages
|
|
64
|
+
nixamp transcript --channel ID [--follow] what a channel is saying, as it says it; --kept for what nixamp.com keeps
|
|
65
|
+
nixamp translate --to sv TEXT say it in another language
|
|
65
66
|
nixamp profile [--handle H] [--voice V] [--profile URL] who the rooms know you as
|
|
66
67
|
nixamp voices the voices a line is read in on the phone
|
|
67
68
|
nixamp opendir list|add|remove folders found on the web, published for everyone
|
|
@@ -241,31 +242,60 @@ takes it away again.
|
|
|
241
242
|
nixamp mcp speak Model Context Protocol on stdin and stdout
|
|
242
243
|
|
|
243
244
|
It offers the watch party tools: list them, read one, put one on the air,
|
|
244
|
-
say where playback is, end it.
|
|
245
|
-
(transcribe_audio,
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
245
|
+
say where playback is, end it. The room tools: transcribe a recording or a
|
|
246
|
+
whole film (transcribe_audio, kept on nixamp.com, translated on request, or
|
|
247
|
+
posted straight into a trollbox), say a line in a room (trollbox_say), read
|
|
248
|
+
a room (trollbox_read), read what a channel is saying (transcript_read, in
|
|
249
|
+
any language). The transcript tools: a kept transcript by its id or its
|
|
250
|
+
media (transcript_get), what this account has had written down
|
|
251
|
+
(transcripts_list), and text in another language (translate_text). And who
|
|
252
|
+
you are in the rooms and how you sound on the phone (profile_get,
|
|
253
|
+
profile_set, voices_list). It acts as whoever this machine is signed in as,
|
|
254
|
+
so \`nixamp login\` (or NIXAMP_TOKEN) comes first.
|
|
255
|
+
|
|
256
|
+
Point an MCP client at it as a stdio server running \`nixamp mcp\`. The same
|
|
257
|
+
tools are at https://nixamp.com/mcp over HTTP, with a nixamp token
|
|
258
|
+
(\`nixamp token create\`) as the bearer, for an agent with no nixamp installed.
|
|
253
259
|
`,
|
|
254
260
|
transcribe: `nixamp transcribe — say it, and have it written down.
|
|
255
261
|
|
|
256
|
-
nixamp transcribe FILE the words in a recording
|
|
257
|
-
nixamp transcribe
|
|
258
|
-
nixamp transcribe FILE --
|
|
259
|
-
nixamp transcribe FILE --
|
|
262
|
+
nixamp transcribe FILE the words in a recording, or a whole film, kept on nixamp.com
|
|
263
|
+
nixamp transcribe URL the same for a link ffmpeg can read
|
|
264
|
+
nixamp transcribe FILE --language sv when Whisper should not guess
|
|
265
|
+
nixamp transcribe FILE --translate de and in German too (de,sv for both)
|
|
266
|
+
nixamp transcribe FILE --srt | --vtt as subtitles, on stdout
|
|
267
|
+
nixamp transcribe FILE --out DIR subtitle files in DIR, one per language
|
|
268
|
+
nixamp transcribe FILE --fresh hear it again even though it is kept
|
|
260
269
|
nixamp transcribe FILE --json the answer as JSON
|
|
270
|
+
nixamp transcribe CLIP --say SERVER a short clip, posted to that server's trollbox
|
|
271
|
+
nixamp transcribe CLIP --say SERVER --channel ID to one channel's room (default: live)
|
|
261
272
|
|
|
262
|
-
FILE is any recording ffmpeg can read; a WAV needs no ffmpeg
|
|
263
|
-
hearing is done by nixamp.com with an open-source model (Whisper,
|
|
264
|
-
Transformers.js) on its own CPU: nothing goes to a speech vendor. It
|
|
265
|
-
sign-in (\`nixamp login\`) and nothing else.
|
|
273
|
+
FILE is any recording ffmpeg can read; a WAV under a minute needs no ffmpeg
|
|
274
|
+
at all. The hearing is done by nixamp.com with an open-source model (Whisper,
|
|
275
|
+
through Transformers.js) on its own CPU: nothing goes to a speech vendor. It
|
|
276
|
+
needs a sign-in (\`nixamp login\`) and nothing else.
|
|
277
|
+
|
|
278
|
+
A film is heard a minute at a time, with when each line is said, and kept on
|
|
279
|
+
nixamp.com under the file's fingerprint: the next \`nixamp transcribe\` of the
|
|
280
|
+
same file, on any machine, and the next server to put it on the air, read
|
|
281
|
+
the lines instead of hearing them. A translation is made once, on nixamp.com
|
|
282
|
+
with an open-source model (OPUS-MT), and kept beside the original.
|
|
266
283
|
|
|
267
284
|
The same ear is behind the microphone button in every nixamp.com trollbox,
|
|
268
285
|
and behind the transcribe_audio tool of \`nixamp mcp\`.
|
|
286
|
+
`,
|
|
287
|
+
translate: `nixamp translate — say it in another language.
|
|
288
|
+
|
|
289
|
+
nixamp translate --to sv "Hello there" Swedish, from English
|
|
290
|
+
nixamp translate --from de --to en "Guten Tag"
|
|
291
|
+
cat lines.txt | nixamp translate --to de each line, in order
|
|
292
|
+
nixamp translate --languages what nixamp.com can translate between
|
|
293
|
+
|
|
294
|
+
The models are open-source (OPUS-MT, through Transformers.js) and run on
|
|
295
|
+
nixamp.com's own CPU; a pair with no model of its own goes through English.
|
|
296
|
+
Needs a sign-in (\`nixamp login\`). The same models turn a live's captions
|
|
297
|
+
into another language as they are said, and a kept transcript into one on
|
|
298
|
+
request.
|
|
269
299
|
`,
|
|
270
300
|
profile: `nixamp profile — who the rooms know you as.
|
|
271
301
|
|
|
@@ -286,13 +316,22 @@ one picked for the account and kept. Two people in a room are two voices.
|
|
|
286
316
|
nixamp transcript --channel ID the recent lines from this machine's daemon
|
|
287
317
|
nixamp transcript --url URL --key K --channel ID from another server, with its share link
|
|
288
318
|
nixamp transcript ... --follow and keep printing as it speaks
|
|
319
|
+
nixamp transcript ... --language sv the lines in Swedish, translated as they are said
|
|
289
320
|
nixamp transcript ... --json the lines as JSON
|
|
321
|
+
nixamp transcript --kept MEDIA_OR_ID [--language de] [--srt|--vtt|--txt]
|
|
322
|
+
a transcript nixamp.com keeps: a file's, a link's, a past live's
|
|
323
|
+
nixamp transcript --list what this account has had written down
|
|
290
324
|
|
|
291
325
|
A server captions a channel while somebody is asking for its transcript: its
|
|
292
326
|
own ffmpeg turns the sound into five-second windows, nixamp.com's ear turns
|
|
293
327
|
those into lines, each stamped with when its sound was heard. The page shows
|
|
294
328
|
them as subtitles, held until its own sound gets there; this prints them.
|
|
295
329
|
The server needs an ffmpeg and a sign-in (\`nixamp login\`).
|
|
330
|
+
|
|
331
|
+
What it hears is kept on nixamp.com under what the channel is playing: a
|
|
332
|
+
film by its bytes, a link by its address, a live as the one broadcast it
|
|
333
|
+
was. A channel playing something already kept reads the lines instead of
|
|
334
|
+
hearing them, and a language asked for is translated once and kept too.
|
|
296
335
|
`,
|
|
297
336
|
attach: `nixamp attach — the player, in front of the running daemon.
|
|
298
337
|
|
|
@@ -515,6 +554,11 @@ export async function main() {
|
|
|
515
554
|
process.exitCode = await mcp();
|
|
516
555
|
return;
|
|
517
556
|
}
|
|
557
|
+
if (first === "translate") {
|
|
558
|
+
const { translate } = await import("./translate-cli.js");
|
|
559
|
+
process.exitCode = await translate(rest);
|
|
560
|
+
return;
|
|
561
|
+
}
|
|
518
562
|
if (first === "token" || first === "tokens") {
|
|
519
563
|
const { tokens } = await import("./session.js");
|
|
520
564
|
process.exitCode = await tokens(rest);
|
package/dist/mcp.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { wavOf } from "./transcribe.ts";
|
|
1
|
+
import { wavOf, type Window } from "./transcribe.ts";
|
|
2
2
|
export declare const PROTOCOL_VERSION = "2025-06-18";
|
|
3
3
|
interface Request {
|
|
4
4
|
jsonrpc: "2.0";
|
|
@@ -21,6 +21,12 @@ export interface McpOptions {
|
|
|
21
21
|
} | null;
|
|
22
22
|
/** How a recording becomes a WAV; the tests hand in a fake. */
|
|
23
23
|
wavOf?: typeof wavOf;
|
|
24
|
+
/** The whole of a file as windows of sound, and a file's identity; the tests hand in fakes. */
|
|
25
|
+
windows?: (source: string) => AsyncIterable<Window>;
|
|
26
|
+
fingerprint?: (path: string) => string;
|
|
27
|
+
sleep?: (ms: number) => Promise<void>;
|
|
28
|
+
/** How many times a translation is asked about before answering with its progress. */
|
|
29
|
+
polls?: number;
|
|
24
30
|
say?: (line: string) => void;
|
|
25
31
|
}
|
|
26
32
|
/** A tool answer, in the shape MCP wants: content blocks, and a flag for failure. */
|
package/dist/mcp.js
CHANGED
|
@@ -18,10 +18,13 @@
|
|
|
18
18
|
* every diagnostic goes to stderr. That is the one rule of this file.
|
|
19
19
|
*/
|
|
20
20
|
import { createInterface } from "node:readline";
|
|
21
|
+
import { basename, extname } from "node:path";
|
|
21
22
|
import { clock } from "./party.js";
|
|
22
23
|
import { readSession } from "./session.js";
|
|
23
|
-
import { askToHear, wavOf } from "./transcribe.js";
|
|
24
|
+
import { askToHear, awaitTranscript, hearWhole, rendered, wavOf } from "./transcribe.js";
|
|
24
25
|
import { readTranscript } from "./transcript.js";
|
|
26
|
+
import { fetchTranscript, listTranscripts, translateTexts } from "./transcript-client.js";
|
|
27
|
+
import { fileFingerprint, idFrom, languageCode, mediaOfUrl, transcriptIdOf } from "./transcripts.js";
|
|
25
28
|
import { personaLines, readPersona, readVoices, writePersona } from "./profile.js";
|
|
26
29
|
export const PROTOCOL_VERSION = "2025-06-18";
|
|
27
30
|
const STRING = { type: "string" };
|
|
@@ -81,18 +84,53 @@ export const TOOLS = [
|
|
|
81
84
|
},
|
|
82
85
|
{
|
|
83
86
|
name: "transcribe_audio",
|
|
84
|
-
description: "The words in a recording on this machine, heard by nixamp.com's own open-source ear (Whisper).
|
|
87
|
+
description: "The words in a recording, a film or a link on this machine, heard by nixamp.com's own open-source ear (Whisper) a minute at a time, with when each line is said, and kept on nixamp.com under the file's fingerprint so the same file is never heard twice by anybody. Any format ffmpeg reads. Given a server, the recording is a short clip and the words are posted to that server's trollbox as this account instead.",
|
|
85
88
|
inputSchema: {
|
|
86
89
|
type: "object",
|
|
87
90
|
properties: {
|
|
88
|
-
path: { ...STRING, description: "The recording's path on this machine." },
|
|
91
|
+
path: { ...STRING, description: "The recording's path on this machine, or a URL ffmpeg can read." },
|
|
89
92
|
language: { ...STRING, description: "A two-letter language code, when Whisper should not guess." },
|
|
90
|
-
|
|
93
|
+
translate: { ...STRING, description: "Also in this language: a two-letter code such as de or sv (several: de,sv). Made once on nixamp.com and kept." },
|
|
94
|
+
format: { ...STRING, description: "How to answer: lines (default, with seconds), srt, vtt or txt." },
|
|
95
|
+
fresh: { type: "boolean", description: "Hear it again even though it is kept." },
|
|
96
|
+
server: { ...STRING, description: "Post the words to this nixamp's trollbox: its address, as in its share link. A clip of a minute at most." },
|
|
91
97
|
channel: { ...STRING, description: "Which of that server's channels; its own stream (live) by default." },
|
|
92
98
|
},
|
|
93
99
|
required: ["path"],
|
|
94
100
|
},
|
|
95
101
|
},
|
|
102
|
+
{
|
|
103
|
+
name: "transcript_get",
|
|
104
|
+
description: "A kept transcript from nixamp.com: what a file, a link or a past live said, by its id or its media identity (file:v1:<hash>, url:<address>, live:<server>/<channel>@<started>). Ask for a language and it is translated once, on nixamp.com, and kept; a long one is answered with progress and is ready on a later ask.",
|
|
105
|
+
inputSchema: {
|
|
106
|
+
type: "object",
|
|
107
|
+
properties: {
|
|
108
|
+
media: { ...STRING, description: "The transcript's id, or the media identity." },
|
|
109
|
+
language: { ...STRING, description: "A two-letter code for a translation; the original when left out." },
|
|
110
|
+
format: { ...STRING, description: "lines (default, with seconds), srt, vtt or txt." },
|
|
111
|
+
},
|
|
112
|
+
required: ["media"],
|
|
113
|
+
},
|
|
114
|
+
},
|
|
115
|
+
{
|
|
116
|
+
name: "transcripts_list",
|
|
117
|
+
description: "What this account has had written down on nixamp.com: each transcript's id, what it is, its language, how many lines, and when.",
|
|
118
|
+
inputSchema: { type: "object", properties: {} },
|
|
119
|
+
},
|
|
120
|
+
{
|
|
121
|
+
name: "translate_text",
|
|
122
|
+
description: "Text in another language, by an open-source model on nixamp.com's own CPU (OPUS-MT). Two-letter codes; German and Swedish among them, and anything with a model from or into English.",
|
|
123
|
+
inputSchema: {
|
|
124
|
+
type: "object",
|
|
125
|
+
properties: {
|
|
126
|
+
text: { ...STRING, description: "The text. Or `texts`, a list." },
|
|
127
|
+
texts: { type: "array", items: STRING, description: "Several texts, answered in the same order." },
|
|
128
|
+
from: { ...STRING, description: "The language the text is in, e.g. en." },
|
|
129
|
+
to: { ...STRING, description: "The language wanted, e.g. sv." },
|
|
130
|
+
},
|
|
131
|
+
required: ["from", "to"],
|
|
132
|
+
},
|
|
133
|
+
},
|
|
96
134
|
{
|
|
97
135
|
name: "trollbox_say",
|
|
98
136
|
description: "Say a line in a live room's trollbox, as this account and under its public handle. A room is a nixamp server's address and one of its channels (or `live`, the server's own stream).",
|
|
@@ -108,13 +146,14 @@ export const TOOLS = [
|
|
|
108
146
|
},
|
|
109
147
|
{
|
|
110
148
|
name: "transcript_read",
|
|
111
|
-
description: "What a live channel is saying: the recent lines of its transcript, oldest first, each with when its sound was heard. The server carrying the channel captions it while somebody asks. Pass the server's address and share key, and the channel's id.",
|
|
149
|
+
description: "What a live channel is saying: the recent lines of its transcript, oldest first, each with when its sound was heard. The server carrying the channel captions it while somebody asks, and translates each line when a language is asked for. Pass the server's address and share key, and the channel's id.",
|
|
112
150
|
inputSchema: {
|
|
113
151
|
type: "object",
|
|
114
152
|
properties: {
|
|
115
153
|
url: { ...STRING, description: "The nixamp server's address, e.g. https://server1.chovy.nixamp.com:4321." },
|
|
116
154
|
key: { ...STRING, description: "The share key from its link, when it has one." },
|
|
117
155
|
channel: { ...STRING, description: "The channel's id on that server (default: main)." },
|
|
156
|
+
language: { ...STRING, description: "The lines in this language (a two-letter code); as heard when left out." },
|
|
118
157
|
after: { type: "number", description: "Only lines heard after this moment (ms since the epoch)." },
|
|
119
158
|
},
|
|
120
159
|
required: ["url"],
|
|
@@ -159,6 +198,13 @@ export const TOOLS = [
|
|
|
159
198
|
function said(line) {
|
|
160
199
|
return `${line.createdAt} ${line.handle}: ${line.body}`;
|
|
161
200
|
}
|
|
201
|
+
/** A kept transcript, as a tool answers it. */
|
|
202
|
+
function transcriptText(transcript, format) {
|
|
203
|
+
const shape = format === "srt" || format === "vtt" || format === "txt" ? format : "lines";
|
|
204
|
+
const head = `${transcript.title || transcript.media} (${transcript.id.slice(0, 12)}), ${transcript.language || "language unknown"}${transcript.translatedFrom ? ` from ${transcript.translatedFrom}` : ""}, ${transcript.lines.length} lines${transcript.complete ? "" : ", so far"}`;
|
|
205
|
+
const others = transcript.languages.filter((one) => one.language !== transcript.language).map((one) => one.language || "original");
|
|
206
|
+
return `${head}${others.length > 0 ? `; also in ${others.join(", ")}` : ""}\n\n${rendered(transcript.lines, shape)}`;
|
|
207
|
+
}
|
|
162
208
|
function text(value) {
|
|
163
209
|
return { content: [{ type: "text", text: value }] };
|
|
164
210
|
}
|
|
@@ -262,25 +308,114 @@ export async function callTool(name, args, options = {}) {
|
|
|
262
308
|
const path = typeof args["path"] === "string" ? args["path"] : "";
|
|
263
309
|
if (!path)
|
|
264
310
|
return failed("Which recording? Pass its path.");
|
|
265
|
-
|
|
311
|
+
const language = languageCode(args["language"]) || undefined;
|
|
312
|
+
if (server) {
|
|
313
|
+
let wav;
|
|
314
|
+
try {
|
|
315
|
+
wav = (options.wavOf ?? wavOf)(path);
|
|
316
|
+
}
|
|
317
|
+
catch (error) {
|
|
318
|
+
return failed(error.message);
|
|
319
|
+
}
|
|
320
|
+
const answer = await askToHear(session, { wav, ...(language ? { language } : {}), server, channel }, send, site);
|
|
321
|
+
if (!answer.ok)
|
|
322
|
+
return failed(answer.error);
|
|
323
|
+
if (answer.heard.text === "")
|
|
324
|
+
return text("Heard nothing in that recording.");
|
|
325
|
+
return text(answer.heard.message
|
|
326
|
+
? `${answer.heard.text}\n\nSaid in the room for ${channel} at ${server} as ${answer.heard.message.handle}.`
|
|
327
|
+
: answer.heard.text);
|
|
328
|
+
}
|
|
329
|
+
// The whole of it, kept under what it is.
|
|
330
|
+
let media;
|
|
266
331
|
try {
|
|
267
|
-
|
|
332
|
+
media = /^https?:\/\//.test(path) ? mediaOfUrl(path) : (options.fingerprint ?? fileFingerprint)(path);
|
|
268
333
|
}
|
|
269
|
-
catch
|
|
270
|
-
return failed(
|
|
334
|
+
catch {
|
|
335
|
+
return failed(`cannot read ${path}`);
|
|
271
336
|
}
|
|
272
|
-
const
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
if (
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
337
|
+
const id = transcriptIdOf(media);
|
|
338
|
+
const title = /^https?:\/\//.test(path) ? path : basename(path, extname(path));
|
|
339
|
+
const format = typeof args["format"] === "string" ? args["format"] : "lines";
|
|
340
|
+
const signed = { site, token: session.token };
|
|
341
|
+
let original = null;
|
|
342
|
+
if (args["fresh"] !== true) {
|
|
343
|
+
const kept = await fetchTranscript(signed, id, "", send);
|
|
344
|
+
if (kept.ok && kept.body.complete)
|
|
345
|
+
original = kept.body;
|
|
346
|
+
else if (!kept.ok && kept.status !== 404)
|
|
347
|
+
return failed(kept.error);
|
|
348
|
+
}
|
|
349
|
+
if (!original) {
|
|
350
|
+
const heard = await hearWhole(signed, path, media, { ...(language ? { language } : {}), title }, {
|
|
351
|
+
fetcher: send,
|
|
352
|
+
...(options.windows ? { windows: options.windows } : {}),
|
|
353
|
+
...(options.sleep ? { sleep: options.sleep } : {}),
|
|
354
|
+
...(options.say ? { onProgress: options.say } : {}),
|
|
355
|
+
});
|
|
356
|
+
if (!heard.ok)
|
|
357
|
+
return failed(heard.error);
|
|
358
|
+
if (heard.heard.lines.length === 0)
|
|
359
|
+
return text("Heard nothing in that.");
|
|
360
|
+
const kept = await fetchTranscript(signed, id, "", send);
|
|
361
|
+
original = kept.ok ? kept.body : {
|
|
362
|
+
id, media, kind: "file", language: heard.heard.language, translatedFrom: null, model: heard.heard.model, complete: true, title,
|
|
363
|
+
seconds: heard.heard.seconds, updatedAt: "", lines: heard.heard.lines, languages: [],
|
|
364
|
+
};
|
|
365
|
+
}
|
|
366
|
+
const wanted = (typeof args["translate"] === "string" ? args["translate"] : "").split(",").map((one) => languageCode(one)).filter((one) => typeof one === "string" && one !== "");
|
|
367
|
+
const parts = [transcriptText(original, format)];
|
|
368
|
+
for (const to of wanted) {
|
|
369
|
+
if (to === original.language)
|
|
370
|
+
continue;
|
|
371
|
+
const got = await awaitTranscript(signed, id, to, { fetcher: send, ...(options.sleep ? { sleep: options.sleep } : {}), ...(options.polls !== undefined ? { polls: options.polls } : {}) });
|
|
372
|
+
if (!got.ok)
|
|
373
|
+
return failed(`could not get it in ${to}: ${got.error}`);
|
|
374
|
+
parts.push(got.body.translating
|
|
375
|
+
? `In ${to}: still being translated, ${got.body.translating.done} of ${got.body.translating.total} lines. Ask transcript_get for ${id} in ${to} in a moment.`
|
|
376
|
+
: transcriptText(got.body, format));
|
|
377
|
+
}
|
|
378
|
+
return text(parts.join("\n\n"));
|
|
379
|
+
}
|
|
380
|
+
if (name === "transcript_get") {
|
|
381
|
+
const named = typeof args["media"] === "string" ? args["media"].trim() : "";
|
|
382
|
+
if (!named)
|
|
383
|
+
return failed("Which transcript? Pass its id or the media identity.");
|
|
384
|
+
const language = languageCode(args["language"]);
|
|
385
|
+
if (language === null)
|
|
386
|
+
return failed("language is a two-letter code, such as de or sv.");
|
|
387
|
+
const got = await awaitTranscript({ site, token: session.token }, idFrom(named), language, {
|
|
388
|
+
fetcher: send, ...(options.sleep ? { sleep: options.sleep } : {}), polls: options.polls ?? 1,
|
|
389
|
+
});
|
|
390
|
+
if (!got.ok)
|
|
391
|
+
return failed(got.error);
|
|
392
|
+
if (got.body.translating) {
|
|
393
|
+
return text(`Still being translated to ${language}: ${got.body.translating.done} of ${got.body.translating.total} lines. Ask again in a moment.`);
|
|
394
|
+
}
|
|
395
|
+
return text(transcriptText(got.body, typeof args["format"] === "string" ? args["format"] : "lines"));
|
|
396
|
+
}
|
|
397
|
+
if (name === "transcripts_list") {
|
|
398
|
+
const got = await listTranscripts({ site, token: session.token }, send);
|
|
399
|
+
if (!got.ok)
|
|
400
|
+
return failed(got.error);
|
|
401
|
+
if (got.body.transcripts.length === 0)
|
|
402
|
+
return text("Nothing has been written down for this account yet.");
|
|
403
|
+
return text(got.body.transcripts.map((one) => `${one.id} ${one.language || "?"}${one.translatedFrom ? `<${one.translatedFrom}` : ""} ${one.lines} lines${one.complete ? "" : " so far"} ${one.title || one.media} ${one.updatedAt}`).join("\n"));
|
|
404
|
+
}
|
|
405
|
+
if (name === "translate_text") {
|
|
406
|
+
const texts = Array.isArray(args["texts"])
|
|
407
|
+
? args["texts"].filter((one) => typeof one === "string")
|
|
408
|
+
: typeof args["text"] === "string" ? [args["text"]] : [];
|
|
409
|
+
if (texts.length === 0)
|
|
410
|
+
return failed("Translate what? Pass text, or texts.");
|
|
411
|
+
const from = languageCode(args["from"]);
|
|
412
|
+
const to = languageCode(args["to"]);
|
|
413
|
+
if (!from || !to)
|
|
414
|
+
return failed("from and to are two-letter language codes, such as en and sv.");
|
|
415
|
+
const got = await translateTexts({ site, token: session.token }, texts, from, to, send);
|
|
416
|
+
if (!got.ok)
|
|
417
|
+
return failed(got.error);
|
|
418
|
+
return text(got.body.texts.join("\n"));
|
|
284
419
|
}
|
|
285
420
|
if (name === "trollbox_say") {
|
|
286
421
|
if (!server)
|
|
@@ -302,7 +437,7 @@ export async function callTool(name, args, options = {}) {
|
|
|
302
437
|
const url = typeof args["url"] === "string" ? args["url"].trim() : "";
|
|
303
438
|
if (!url)
|
|
304
439
|
return failed("Which server? Pass its address.");
|
|
305
|
-
const got = await readTranscript({ url, key: typeof args["key"] === "string" && args["key"] ? args["key"] : null }, channel === "live" ? "main" : channel, typeof args["after"] === "number" ? args["after"] : 0, send);
|
|
440
|
+
const got = await readTranscript({ url, key: typeof args["key"] === "string" && args["key"] ? args["key"] : null }, channel === "live" ? "main" : channel, typeof args["after"] === "number" ? args["after"] : 0, send, languageCode(args["language"]) || "");
|
|
306
441
|
if (!got.ok)
|
|
307
442
|
return failed(got.error);
|
|
308
443
|
if (got.answer.recent.length === 0) {
|
|
@@ -368,8 +503,8 @@ export async function handleMessage(message, options = {}) {
|
|
|
368
503
|
return reply({
|
|
369
504
|
protocolVersion: PROTOCOL_VERSION,
|
|
370
505
|
capabilities: { tools: { listChanged: false } },
|
|
371
|
-
serverInfo: { name: "nixamp", title: "nixamp: watch parties and
|
|
372
|
-
instructions: "Watch parties on nixamp
|
|
506
|
+
serverInfo: { name: "nixamp", title: "nixamp: watch parties, rooms and transcripts", version: "2" },
|
|
507
|
+
instructions: "Watch parties on nixamp: a party lives on the site hosting the film and is bridged here as a room every nixamp client can join; codes are the ones that site shows, positions are seconds into the film. Rooms: say and read trollbox lines, hear a recording. Transcripts: a file, a link or a live is written down once by nixamp.com's own ear and kept under what it is; ask for it in another language and it is translated once and kept too.",
|
|
373
508
|
});
|
|
374
509
|
}
|
|
375
510
|
// Notifications carry no id and are answered with silence, which is what
|
package/dist/server.d.ts
CHANGED
|
@@ -32,6 +32,9 @@ import { Rooms } from "./rooms.ts";
|
|
|
32
32
|
import { Trollbox } from "./trollbox.ts";
|
|
33
33
|
import { Speech } from "./speech.ts";
|
|
34
34
|
import { Captions } from "./captions.ts";
|
|
35
|
+
import { Translator } from "./translate.ts";
|
|
36
|
+
import { StoredTranslations } from "./translate-jobs.ts";
|
|
37
|
+
import { Transcripts } from "./transcripts.ts";
|
|
35
38
|
import { Profiles, Voices } from "./voices.ts";
|
|
36
39
|
import { type Codecs } from "./audio.ts";
|
|
37
40
|
import { type Tools, type Track } from "./audio.ts";
|
|
@@ -699,6 +702,14 @@ export interface HandlerOptions {
|
|
|
699
702
|
trollbox?: Trollbox;
|
|
700
703
|
/** Speech to text: a line said out loud, heard here. Needs the optional model. */
|
|
701
704
|
speech?: Speech;
|
|
705
|
+
/** Translation: texts in another language, by a model here. The same optional library. */
|
|
706
|
+
translator?: Translator;
|
|
707
|
+
/** The transcript store: what was heard, kept under the media's identity. Where the accounts are. */
|
|
708
|
+
transcripts?: Transcripts;
|
|
709
|
+
/** Stored transcripts translated, as jobs. */
|
|
710
|
+
translations?: StoredTranslations;
|
|
711
|
+
/** nixamp.com's address, for the tools reached over /mcp to call. */
|
|
712
|
+
site?: string;
|
|
702
713
|
/** Other people's OpenProfiles, for the voice their lines are read in. */
|
|
703
714
|
profiles?: Profiles;
|
|
704
715
|
/** The voices lines are read in: ElevenLabs when Telnyx holds the key, Kokoro otherwise. */
|