nixamp 0.21.4 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +75 -0
- package/dist/captions.d.ts +75 -0
- package/dist/captions.js +308 -0
- package/dist/main.js +47 -2
- package/dist/mcp.d.ts +3 -0
- package/dist/mcp.js +134 -2
- package/dist/server.d.ts +6 -0
- package/dist/server.js +151 -1
- package/dist/speech.d.ts +96 -0
- package/dist/speech.js +311 -0
- package/dist/transcribe.d.ts +43 -0
- package/dist/transcribe.js +155 -0
- package/dist/transcript.d.ts +40 -0
- package/dist/transcript.js +120 -0
- package/package.json +5 -2
- package/src/captions.ts +334 -0
- package/src/main.ts +47 -2
- package/src/mcp.ts +145 -2
- package/src/server.ts +153 -1
- package/src/speech.ts +324 -0
- package/src/transcribe.ts +177 -0
- package/src/transcript.ts +150 -0
- package/web/dist/.well-known/openaccess.json +29 -0
- package/web/dist/assets/{hls-3VKVEQE3-RSItUqmr.js → hls-3VKVEQE3-DOUPX_xZ.js} +1 -1
- package/web/dist/assets/index-BIL7sVli.js +1 -0
- package/web/dist/assets/index-D3Is8ykj.css +1 -0
- package/web/dist/assets/{mpegts-LO6RVLD6-B5VNNeMl.js → mpegts-LO6RVLD6-C9gfFFrM.js} +1 -1
- package/web/dist/assets/{mpegts-CGsP3zeI.js → mpegts-XL4bjo7s.js} +1 -1
- package/web/dist/index.html +21 -2
- package/web/dist/sw.js +7 -6
- package/web/dist/assets/index-DMVUa6vY.js +0 -1
- package/web/dist/assets/index-jOXfym7D.css +0 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "nixamp",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.22.0",
|
|
4
4
|
"description": "It really whips the terminal's ass. A Winamp-shaped audio player for your terminal.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|
|
@@ -45,13 +45,16 @@
|
|
|
45
45
|
},
|
|
46
46
|
"dependencies": {
|
|
47
47
|
"@profullstack/auth-system": "^0.6.0",
|
|
48
|
-
"@profullstack/hqtui": "^0.
|
|
48
|
+
"@profullstack/hqtui": "^0.5.0",
|
|
49
49
|
"@profullstack/throttle": "0.2.2",
|
|
50
50
|
"@profullstack/x402-gateway": "0.6.0",
|
|
51
51
|
"acme-client": "5.4.0",
|
|
52
52
|
"pg": "^8.23.0",
|
|
53
53
|
"web-push": "^3.6.7"
|
|
54
54
|
},
|
|
55
|
+
"optionalDependencies": {
|
|
56
|
+
"@huggingface/transformers": "^4.2.0"
|
|
57
|
+
},
|
|
55
58
|
"devDependencies": {
|
|
56
59
|
"@types/node": "^26",
|
|
57
60
|
"@types/pg": "^8.23.1",
|
package/src/captions.ts
ADDED
|
@@ -0,0 +1,334 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Captions: what a live channel is saying, written down as it says it.
|
|
3
|
+
*
|
|
4
|
+
* One captioner per channel, started when the first person asks for the
|
|
5
|
+
* transcript and stopped a minute after the last one leaves, because it
|
|
6
|
+
* costs a CPU somewhere for as long as it runs. It listens to the channel
|
|
7
|
+
* exactly as a browser does -- the same bytes, from the same backlog --
|
|
8
|
+
* hands them to an ffmpeg that turns them into 16 kHz mono PCM, cuts that
|
|
9
|
+
* into five-second windows, and sends each window to nixamp.com's ear
|
|
10
|
+
* (see speech.ts) signed in as this server. The words come back as a
|
|
11
|
+
* line, stamped with the wall-clock moment the sound was heard, so a page
|
|
12
|
+
* can hold each line until its own playback gets there and the subtitles
|
|
13
|
+
* land close to the voice.
|
|
14
|
+
*
|
|
15
|
+
* Five seconds is the trade. Shorter windows hear less context and cost
|
|
16
|
+
* more asks; longer ones make the words later than the sound. A line is a
|
|
17
|
+
* window: no word-level timing, because the ear takes twice as long when
|
|
18
|
+
* asked for it and the page is only ever close to the voice, not on it.
|
|
19
|
+
*
|
|
20
|
+
* A quiet window -- the gap between songs, a picture with no talking -- is
|
|
21
|
+
* never sent. Most of a music channel is that, and hearing it costs the
|
|
22
|
+
* same as hearing speech.
|
|
23
|
+
*/
|
|
24
|
+
import { spawn } from "node:child_process";
|
|
25
|
+
import type { Listener } from "./channels.ts";
|
|
26
|
+
import { RATE } from "./speech.ts";
|
|
27
|
+
|
|
28
|
+
export interface CaptionLine {
|
|
29
|
+
/** The channel's id. */
|
|
30
|
+
channel: string;
|
|
31
|
+
/** When the sound this line is from began and ended, wall clock, ms. */
|
|
32
|
+
at: number;
|
|
33
|
+
until: number;
|
|
34
|
+
text: string;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** What turns a channel's bytes into 16 kHz mono 16-bit PCM. ffmpeg, or a test's stand-in. */
|
|
38
|
+
export interface Decoder {
|
|
39
|
+
write(chunk: Buffer): boolean;
|
|
40
|
+
end(): void;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export interface CaptionsOptions {
|
|
44
|
+
/** A listener on a channel, or null when there is no such channel. */
|
|
45
|
+
listen: (id: string, listener: Listener) => (() => void) | null;
|
|
46
|
+
ffmpeg: string[];
|
|
47
|
+
/** Whose ear to use: this server's own sign-in, read when a captioner starts. Null means no captions. */
|
|
48
|
+
session: () => { site: string; token: string } | null;
|
|
49
|
+
fetcher?: typeof fetch;
|
|
50
|
+
/** How bytes become PCM. The default spawns ffmpeg; the tests hand in something quieter. */
|
|
51
|
+
decoder?: (onPcm: (pcm: Buffer) => void, onEnd: () => void) => Decoder;
|
|
52
|
+
now?: () => number;
|
|
53
|
+
onEvent?: (message: string) => void;
|
|
54
|
+
windowMs?: number;
|
|
55
|
+
idleMs?: number;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export const WINDOW_MS = 5000;
|
|
59
|
+
/** Lines kept per channel for whoever arrives late. */
|
|
60
|
+
export const KEEP = 200;
|
|
61
|
+
export const IDLE_MS = 60_000;
|
|
62
|
+
/** Below this RMS (about -48 dBFS) a window is silence, and never sent. */
|
|
63
|
+
export const QUIET = 0.004;
|
|
64
|
+
/** Windows waiting on the ear at once. Past this the sound is dropped, not queued: late words are worse than none. */
|
|
65
|
+
export const IN_FLIGHT = 2;
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* The ffmpeg arguments: whatever arrives on stdin, as PCM on stdout, with
|
|
69
|
+
* a short probe so the first line is not long in coming. Never `-fflags
|
|
70
|
+
* nobuffer`: it drops the packets it probed, which took the first 1.7
|
|
71
|
+
* seconds out of every captioner and began every transcript mid-sentence.
|
|
72
|
+
*/
|
|
73
|
+
export function decoderArgs(): string[] {
|
|
74
|
+
return [
|
|
75
|
+
"-v", "error", "-nostats",
|
|
76
|
+
"-flags", "low_delay",
|
|
77
|
+
"-analyzeduration", "500000", "-probesize", "262144",
|
|
78
|
+
"-i", "pipe:0",
|
|
79
|
+
"-vn", "-ac", "1", "-ar", String(RATE), "-f", "s16le", "pipe:1",
|
|
80
|
+
];
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
function ffmpegDecoder(ffmpeg: string[], onPcm: (pcm: Buffer) => void, onEnd: () => void): Decoder {
|
|
84
|
+
const [command, ...prefix] = ffmpeg;
|
|
85
|
+
const child = spawn(command as string, [...prefix, ...decoderArgs()], { stdio: ["pipe", "pipe", "ignore"] });
|
|
86
|
+
let ended = false;
|
|
87
|
+
const end = (): void => {
|
|
88
|
+
if (ended) return;
|
|
89
|
+
ended = true;
|
|
90
|
+
onEnd();
|
|
91
|
+
};
|
|
92
|
+
child.stdin.on("error", () => undefined);
|
|
93
|
+
child.stdout.on("data", (chunk: Buffer) => onPcm(chunk));
|
|
94
|
+
child.on("error", end);
|
|
95
|
+
child.on("close", end);
|
|
96
|
+
return {
|
|
97
|
+
write: (chunk) => {
|
|
98
|
+
if (ended || child.stdin.destroyed) return false;
|
|
99
|
+
try {
|
|
100
|
+
child.stdin.write(chunk);
|
|
101
|
+
} catch {
|
|
102
|
+
return false;
|
|
103
|
+
}
|
|
104
|
+
return true;
|
|
105
|
+
},
|
|
106
|
+
end: () => {
|
|
107
|
+
try {
|
|
108
|
+
child.stdin.end();
|
|
109
|
+
} catch {
|
|
110
|
+
// Already gone.
|
|
111
|
+
}
|
|
112
|
+
if (!ended) {
|
|
113
|
+
const kill = setTimeout(() => child.kill("SIGKILL"), 2000);
|
|
114
|
+
kill.unref?.();
|
|
115
|
+
}
|
|
116
|
+
},
|
|
117
|
+
};
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/** Whether a window of 16-bit PCM has anything in it worth hearing. */
|
|
121
|
+
export function isQuiet(pcm: Buffer, threshold = QUIET): boolean {
|
|
122
|
+
const samples = Math.floor(pcm.length / 2);
|
|
123
|
+
if (samples === 0) return true;
|
|
124
|
+
let sum = 0;
|
|
125
|
+
for (let i = 0; i < samples; i++) {
|
|
126
|
+
const value = pcm.readInt16LE(i * 2) / 32768;
|
|
127
|
+
sum += value * value;
|
|
128
|
+
}
|
|
129
|
+
return Math.sqrt(sum / samples) < threshold;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/** A WAV around 16-bit mono PCM, without copying it through floats. */
|
|
133
|
+
export function wavAround(pcm: Buffer, rate = RATE): Buffer {
|
|
134
|
+
const header = Buffer.alloc(44);
|
|
135
|
+
header.write("RIFF", 0, "ascii");
|
|
136
|
+
header.writeUInt32LE(36 + pcm.length, 4);
|
|
137
|
+
header.write("WAVE", 8, "ascii");
|
|
138
|
+
header.write("fmt ", 12, "ascii");
|
|
139
|
+
header.writeUInt32LE(16, 16);
|
|
140
|
+
header.writeUInt16LE(1, 20);
|
|
141
|
+
header.writeUInt16LE(1, 22);
|
|
142
|
+
header.writeUInt32LE(rate, 24);
|
|
143
|
+
header.writeUInt32LE(rate * 2, 28);
|
|
144
|
+
header.writeUInt16LE(2, 32);
|
|
145
|
+
header.writeUInt16LE(16, 34);
|
|
146
|
+
header.write("data", 36, "ascii");
|
|
147
|
+
header.writeUInt32LE(pcm.length, 40);
|
|
148
|
+
return Buffer.concat([header, pcm]);
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
type Subscriber = (line: CaptionLine) => void;
|
|
152
|
+
|
|
153
|
+
class Captioner {
|
|
154
|
+
readonly lines: CaptionLine[] = [];
|
|
155
|
+
readonly subscribers = new Set<Subscriber>();
|
|
156
|
+
/** The last thing that went wrong, for whoever asks why there are no lines. */
|
|
157
|
+
error = "";
|
|
158
|
+
private decoder: Decoder | null = null;
|
|
159
|
+
private detach: (() => void) | null = null;
|
|
160
|
+
private idle: ReturnType<typeof setTimeout> | null = null;
|
|
161
|
+
private pending: Buffer[] = [];
|
|
162
|
+
private pendingBytes = 0;
|
|
163
|
+
private inFlight = 0;
|
|
164
|
+
private stopped = false;
|
|
165
|
+
private complainedAt = 0;
|
|
166
|
+
|
|
167
|
+
constructor(
|
|
168
|
+
readonly id: string,
|
|
169
|
+
private readonly options: CaptionsOptions,
|
|
170
|
+
private readonly onStop: () => void,
|
|
171
|
+
) {}
|
|
172
|
+
|
|
173
|
+
private get windowBytes(): number {
|
|
174
|
+
return Math.round(((this.options.windowMs ?? WINDOW_MS) / 1000) * RATE) * 2;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
start(): boolean {
|
|
178
|
+
const make = this.options.decoder ?? ((onPcm, onEnd) => ffmpegDecoder(this.options.ffmpeg, onPcm, onEnd));
|
|
179
|
+
this.decoder = make((pcm) => this.onPcm(pcm), () => this.stop());
|
|
180
|
+
this.detach = this.options.listen(this.id, {
|
|
181
|
+
write: (chunk) => this.decoder?.write(chunk) ?? false,
|
|
182
|
+
end: () => this.stop(),
|
|
183
|
+
});
|
|
184
|
+
if (this.detach === null) {
|
|
185
|
+
this.stop();
|
|
186
|
+
return false;
|
|
187
|
+
}
|
|
188
|
+
return true;
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
private onPcm(pcm: Buffer): void {
|
|
192
|
+
if (this.stopped) return;
|
|
193
|
+
this.pending.push(pcm);
|
|
194
|
+
this.pendingBytes += pcm.length;
|
|
195
|
+
const size = this.windowBytes;
|
|
196
|
+
while (this.pendingBytes >= size) {
|
|
197
|
+
const all = Buffer.concat(this.pending);
|
|
198
|
+
const window = all.subarray(0, size);
|
|
199
|
+
const rest = all.subarray(size);
|
|
200
|
+
this.pending = rest.length > 0 ? [Buffer.from(rest)] : [];
|
|
201
|
+
this.pendingBytes = rest.length;
|
|
202
|
+
const until = (this.options.now ?? Date.now)();
|
|
203
|
+
void this.hear(Buffer.from(window), until - (this.options.windowMs ?? WINDOW_MS), until);
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
private async hear(pcm: Buffer, at: number, until: number): Promise<void> {
|
|
208
|
+
if (isQuiet(pcm)) return;
|
|
209
|
+
if (this.inFlight >= IN_FLIGHT) return;
|
|
210
|
+
const session = this.options.session();
|
|
211
|
+
if (session === null) {
|
|
212
|
+
this.complain("this server is not signed in, so it cannot caption; `nixamp login` on it");
|
|
213
|
+
return;
|
|
214
|
+
}
|
|
215
|
+
this.inFlight += 1;
|
|
216
|
+
try {
|
|
217
|
+
const wav = wavAround(pcm);
|
|
218
|
+
const response = await (this.options.fetcher ?? fetch)(`${session.site.replace(/\/+$/, "")}/api/v1/speech/transcribe`, {
|
|
219
|
+
method: "POST",
|
|
220
|
+
headers: { authorization: `Bearer ${session.token}`, "content-type": "audio/wav" },
|
|
221
|
+
body: new Blob([wav.buffer.slice(wav.byteOffset, wav.byteOffset + wav.byteLength) as ArrayBuffer]),
|
|
222
|
+
});
|
|
223
|
+
const body = (await response.json().catch(() => ({}))) as { text?: string; error?: string };
|
|
224
|
+
if (!response.ok) {
|
|
225
|
+
this.complain(body.error ?? `nixamp.com answered ${response.status}`);
|
|
226
|
+
return;
|
|
227
|
+
}
|
|
228
|
+
const text = (body.text ?? "").trim();
|
|
229
|
+
if (text === "" || this.stopped) return;
|
|
230
|
+
this.error = "";
|
|
231
|
+
const line: CaptionLine = { channel: this.id, at, until, text };
|
|
232
|
+
this.lines.push(line);
|
|
233
|
+
while (this.lines.length > KEEP) this.lines.shift();
|
|
234
|
+
for (const subscriber of this.subscribers) {
|
|
235
|
+
try {
|
|
236
|
+
subscriber(line);
|
|
237
|
+
} catch {
|
|
238
|
+
// A listener that throws is not this channel's problem.
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
} catch (error) {
|
|
242
|
+
this.complain(`could not reach the ear: ${(error as Error).message}`);
|
|
243
|
+
} finally {
|
|
244
|
+
this.inFlight -= 1;
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/** Said once a minute at most: a broken ear would otherwise say so twelve times a minute. */
|
|
249
|
+
private complain(message: string): void {
|
|
250
|
+
this.error = message;
|
|
251
|
+
const now = (this.options.now ?? Date.now)();
|
|
252
|
+
if (now - this.complainedAt < 60_000) return;
|
|
253
|
+
this.complainedAt = now;
|
|
254
|
+
this.options.onEvent?.(`captions for "${this.id}": ${message}`);
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
subscribe(subscriber: Subscriber): () => void {
|
|
258
|
+
this.subscribers.add(subscriber);
|
|
259
|
+
if (this.idle) clearTimeout(this.idle);
|
|
260
|
+
this.idle = null;
|
|
261
|
+
return () => {
|
|
262
|
+
this.subscribers.delete(subscriber);
|
|
263
|
+
if (this.subscribers.size === 0 && !this.stopped) {
|
|
264
|
+
this.idle = setTimeout(() => {
|
|
265
|
+
if (this.subscribers.size === 0) this.stop();
|
|
266
|
+
}, this.options.idleMs ?? IDLE_MS);
|
|
267
|
+
this.idle.unref?.();
|
|
268
|
+
}
|
|
269
|
+
};
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
stop(): void {
|
|
273
|
+
if (this.stopped) return;
|
|
274
|
+
this.stopped = true;
|
|
275
|
+
if (this.idle) clearTimeout(this.idle);
|
|
276
|
+
this.idle = null;
|
|
277
|
+
this.detach?.();
|
|
278
|
+
this.detach = null;
|
|
279
|
+
this.decoder?.end();
|
|
280
|
+
this.decoder = null;
|
|
281
|
+
this.pending = [];
|
|
282
|
+
this.pendingBytes = 0;
|
|
283
|
+
this.subscribers.clear();
|
|
284
|
+
this.onStop();
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
export class Captions {
|
|
289
|
+
private readonly running = new Map<string, Captioner>();
|
|
290
|
+
|
|
291
|
+
constructor(private readonly options: CaptionsOptions) {}
|
|
292
|
+
|
|
293
|
+
/** Whether this server can caption at all: it has to be signed in for the ear to answer it. */
|
|
294
|
+
available(): boolean {
|
|
295
|
+
return this.options.session() !== null;
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
/**
|
|
299
|
+
* Lines for a channel as they are heard, starting the captioner if it is
|
|
300
|
+
* not running. Null when there is no such channel. The returned function
|
|
301
|
+
* is how to stop listening; the captioner itself stops a minute after the
|
|
302
|
+
* last listener does.
|
|
303
|
+
*/
|
|
304
|
+
subscribe(id: string, subscriber: Subscriber): (() => void) | null {
|
|
305
|
+
let captioner = this.running.get(id);
|
|
306
|
+
if (!captioner) {
|
|
307
|
+
const made = new Captioner(id, this.options, () => {
|
|
308
|
+
if (this.running.get(id) === made) this.running.delete(id);
|
|
309
|
+
});
|
|
310
|
+
this.running.set(id, made);
|
|
311
|
+
if (!made.start()) return null;
|
|
312
|
+
this.options.onEvent?.(`captions for "${id}": started`);
|
|
313
|
+
captioner = made;
|
|
314
|
+
}
|
|
315
|
+
return captioner.subscribe(subscriber);
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
/** The recent lines of a channel, oldest first, after a moment when given. Empty when nobody has asked for them. */
|
|
319
|
+
recent(id: string, after = 0): CaptionLine[] {
|
|
320
|
+
const captioner = this.running.get(id);
|
|
321
|
+
if (!captioner) return [];
|
|
322
|
+
return after > 0 ? captioner.lines.filter((line) => line.at > after) : [...captioner.lines];
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
/** Whether a channel is being captioned, and what last went wrong if the lines are not coming. */
|
|
326
|
+
status(id: string): { on: boolean; lines: number; error: string } {
|
|
327
|
+
const captioner = this.running.get(id);
|
|
328
|
+
return captioner ? { on: true, lines: captioner.lines.length, error: captioner.error } : { on: false, lines: 0, error: "" };
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
stopAll(): void {
|
|
332
|
+
for (const captioner of [...this.running.values()]) captioner.stop();
|
|
333
|
+
}
|
|
334
|
+
}
|
package/src/main.ts
CHANGED
|
@@ -83,6 +83,8 @@ const HELP = `nixamp — it really whips the terminal's ass.
|
|
|
83
83
|
nixamp server list|add|remove the machines you run, kept against your account
|
|
84
84
|
nixamp party list|join|host watch parties, here and on the sites nixamp is connected to
|
|
85
85
|
nixamp mcp speak Model Context Protocol on stdin, for an agent
|
|
86
|
+
nixamp transcribe FILE [--say SERVER] the words in a recording, and into a trollbox
|
|
87
|
+
nixamp transcript --channel ID [--follow] what a channel is saying, as it says it
|
|
86
88
|
nixamp opendir list|add|remove folders found on the web, published for everyone
|
|
87
89
|
nixamp update [version] re-run the installer, keeping your choices
|
|
88
90
|
nixamp uninstall [--yes] remove everything the installer created
|
|
@@ -262,10 +264,43 @@ takes it away again.
|
|
|
262
264
|
nixamp mcp speak Model Context Protocol on stdin and stdout
|
|
263
265
|
|
|
264
266
|
It offers the watch party tools: list them, read one, put one on the air,
|
|
265
|
-
say where playback is, end it.
|
|
266
|
-
|
|
267
|
+
say where playback is, end it. And the room tools: transcribe a recording
|
|
268
|
+
(transcribe_audio, which can post the words straight into a trollbox), say a
|
|
269
|
+
line in a room (trollbox_say), read a room (trollbox_read), read what a
|
|
270
|
+
channel is saying (transcript_read). It acts as
|
|
271
|
+
whoever this machine is signed in as, so \`nixamp login\` (or NIXAMP_TOKEN)
|
|
272
|
+
comes first.
|
|
267
273
|
|
|
268
274
|
Point an MCP client at it as a stdio server running \`nixamp mcp\`.
|
|
275
|
+
`,
|
|
276
|
+
transcribe: `nixamp transcribe — say it, and have it written down.
|
|
277
|
+
|
|
278
|
+
nixamp transcribe FILE the words in a recording
|
|
279
|
+
nixamp transcribe FILE --say SERVER and post them to that server's trollbox
|
|
280
|
+
nixamp transcribe FILE --say SERVER --channel ID to one channel's room (default: live)
|
|
281
|
+
nixamp transcribe FILE --language de when Whisper should not guess
|
|
282
|
+
nixamp transcribe FILE --json the answer as JSON
|
|
283
|
+
|
|
284
|
+
FILE is any recording ffmpeg can read; a WAV needs no ffmpeg at all. The
|
|
285
|
+
hearing is done by nixamp.com with an open-source model (Whisper, through
|
|
286
|
+
Transformers.js) on its own CPU: nothing goes to a speech vendor. It needs a
|
|
287
|
+
sign-in (\`nixamp login\`) and nothing else. Up to a minute at a time.
|
|
288
|
+
|
|
289
|
+
The same ear is behind the microphone button in every nixamp.com trollbox,
|
|
290
|
+
and behind the transcribe_audio tool of \`nixamp mcp\`.
|
|
291
|
+
`,
|
|
292
|
+
transcript: `nixamp transcript — what a channel is saying, written down.
|
|
293
|
+
|
|
294
|
+
nixamp transcript --channel ID the recent lines from this machine's daemon
|
|
295
|
+
nixamp transcript --url URL --key K --channel ID from another server, with its share link
|
|
296
|
+
nixamp transcript ... --follow and keep printing as it speaks
|
|
297
|
+
nixamp transcript ... --json the lines as JSON
|
|
298
|
+
|
|
299
|
+
A server captions a channel while somebody is asking for its transcript: its
|
|
300
|
+
own ffmpeg turns the sound into five-second windows, nixamp.com's ear turns
|
|
301
|
+
those into lines, each stamped with when its sound was heard. The page shows
|
|
302
|
+
them as subtitles, held until its own sound gets there; this prints them.
|
|
303
|
+
The server needs an ffmpeg and a sign-in (\`nixamp login\`).
|
|
269
304
|
`,
|
|
270
305
|
attach: `nixamp attach — the player, in front of the running daemon.
|
|
271
306
|
|
|
@@ -466,6 +501,16 @@ export async function main(): Promise<void> {
|
|
|
466
501
|
process.exitCode = await party(rest);
|
|
467
502
|
return;
|
|
468
503
|
}
|
|
504
|
+
if (first === "transcript" || first === "captions") {
|
|
505
|
+
const { transcript } = await import("./transcript.ts");
|
|
506
|
+
process.exitCode = await transcript(rest);
|
|
507
|
+
return;
|
|
508
|
+
}
|
|
509
|
+
if (first === "transcribe" || first === "dictate") {
|
|
510
|
+
const { transcribe } = await import("./transcribe.ts");
|
|
511
|
+
process.exitCode = await transcribe(rest);
|
|
512
|
+
return;
|
|
513
|
+
}
|
|
469
514
|
if (first === "mcp") {
|
|
470
515
|
const { mcp } = await import("./mcp.ts");
|
|
471
516
|
process.exitCode = await mcp();
|
package/src/mcp.ts
CHANGED
|
@@ -8,7 +8,9 @@
|
|
|
8
8
|
*
|
|
9
9
|
* What it offers is the watch party, because that is the part of nixamp an
|
|
10
10
|
* agent can usefully do something with: find the party, say where it is, put
|
|
11
|
-
* one on the air, move everybody to the same second.
|
|
11
|
+
* one on the air, move everybody to the same second. And the room: hear a
|
|
12
|
+
* recording (nixamp.com's own ear, see speech.ts), say a line in a trollbox,
|
|
13
|
+
* read one back. It signs in as whoever
|
|
12
14
|
* this machine is signed in as -- the session on disk, or NIXAMP_TOKEN --
|
|
13
15
|
* because an agent holding its own credential is a credential nobody revokes.
|
|
14
16
|
*
|
|
@@ -18,6 +20,8 @@
|
|
|
18
20
|
import { createInterface } from "node:readline";
|
|
19
21
|
import { clock, type PartyRow } from "./party.ts";
|
|
20
22
|
import { readSession } from "./session.ts";
|
|
23
|
+
import { askToHear, wavOf } from "./transcribe.ts";
|
|
24
|
+
import { readTranscript } from "./transcript.ts";
|
|
21
25
|
|
|
22
26
|
export const PROTOCOL_VERSION = "2025-06-18";
|
|
23
27
|
|
|
@@ -94,12 +98,82 @@ export const TOOLS: ToolDefinition[] = [
|
|
|
94
98
|
description: "End a watch party. Only its host may.",
|
|
95
99
|
inputSchema: { type: "object", properties: { code: STRING }, required: ["code"] },
|
|
96
100
|
},
|
|
101
|
+
{
|
|
102
|
+
name: "transcribe_audio",
|
|
103
|
+
description:
|
|
104
|
+
"The words in a recording on this machine, heard by nixamp.com's own open-source ear (Whisper). Any format ffmpeg reads; up to a minute. Given a server, the words are also posted to that server's trollbox as this account.",
|
|
105
|
+
inputSchema: {
|
|
106
|
+
type: "object",
|
|
107
|
+
properties: {
|
|
108
|
+
path: { ...STRING, description: "The recording's path on this machine." },
|
|
109
|
+
language: { ...STRING, description: "A two-letter language code, when Whisper should not guess." },
|
|
110
|
+
server: { ...STRING, description: "Post the words to this nixamp's trollbox: its address, as in its share link." },
|
|
111
|
+
channel: { ...STRING, description: "Which of that server's channels; its own stream (live) by default." },
|
|
112
|
+
},
|
|
113
|
+
required: ["path"],
|
|
114
|
+
},
|
|
115
|
+
},
|
|
116
|
+
{
|
|
117
|
+
name: "trollbox_say",
|
|
118
|
+
description:
|
|
119
|
+
"Say a line in a live room's trollbox, as this account and under its public handle. A room is a nixamp server's address and one of its channels (or `live`, the server's own stream).",
|
|
120
|
+
inputSchema: {
|
|
121
|
+
type: "object",
|
|
122
|
+
properties: {
|
|
123
|
+
server: { ...STRING, description: "The nixamp server's address, as in its share link." },
|
|
124
|
+
channel: { ...STRING, description: "The channel's id, or live (default)." },
|
|
125
|
+
text: { ...STRING, description: "The line. At most 500 characters." },
|
|
126
|
+
},
|
|
127
|
+
required: ["server", "text"],
|
|
128
|
+
},
|
|
129
|
+
},
|
|
130
|
+
{
|
|
131
|
+
name: "transcript_read",
|
|
132
|
+
description:
|
|
133
|
+
"What a live channel is saying: the recent lines of its transcript, oldest first, each with when its sound was heard. The server carrying the channel captions it while somebody asks. Pass the server's address and share key, and the channel's id.",
|
|
134
|
+
inputSchema: {
|
|
135
|
+
type: "object",
|
|
136
|
+
properties: {
|
|
137
|
+
url: { ...STRING, description: "The nixamp server's address, e.g. https://server1.chovy.nixamp.com:4321." },
|
|
138
|
+
key: { ...STRING, description: "The share key from its link, when it has one." },
|
|
139
|
+
channel: { ...STRING, description: "The channel's id on that server (default: main)." },
|
|
140
|
+
after: { type: "number", description: "Only lines heard after this moment (ms since the epoch)." },
|
|
141
|
+
},
|
|
142
|
+
required: ["url"],
|
|
143
|
+
},
|
|
144
|
+
},
|
|
145
|
+
{
|
|
146
|
+
name: "trollbox_read",
|
|
147
|
+
description: "The recent lines in a live room's trollbox, oldest first: who said what, and when.",
|
|
148
|
+
inputSchema: {
|
|
149
|
+
type: "object",
|
|
150
|
+
properties: {
|
|
151
|
+
server: { ...STRING, description: "The nixamp server's address, as in its share link." },
|
|
152
|
+
channel: { ...STRING, description: "The channel's id, or live (default)." },
|
|
153
|
+
after: { ...STRING, description: "Only lines after this moment (an ISO timestamp)." },
|
|
154
|
+
},
|
|
155
|
+
required: ["server"],
|
|
156
|
+
},
|
|
157
|
+
},
|
|
97
158
|
];
|
|
98
159
|
|
|
160
|
+
interface TrollboxLine {
|
|
161
|
+
id: string;
|
|
162
|
+
handle: string;
|
|
163
|
+
body: string;
|
|
164
|
+
createdAt: string;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
function said(line: TrollboxLine): string {
|
|
168
|
+
return `${line.createdAt} ${line.handle}: ${line.body}`;
|
|
169
|
+
}
|
|
170
|
+
|
|
99
171
|
export interface McpOptions {
|
|
100
172
|
fetcher?: typeof fetch;
|
|
101
173
|
/** Injected by the tests; the session on disk otherwise. */
|
|
102
174
|
session?: { site: string; token: string } | null;
|
|
175
|
+
/** How a recording becomes a WAV; the tests hand in a fake. */
|
|
176
|
+
wavOf?: typeof wavOf;
|
|
103
177
|
say?: (line: string) => void;
|
|
104
178
|
}
|
|
105
179
|
|
|
@@ -204,6 +278,75 @@ export async function callTool(name: string, args: Record<string, unknown>, opti
|
|
|
204
278
|
if (!response.ok) return failed(await answerOf(response));
|
|
205
279
|
return text(`Ended ${code}.`);
|
|
206
280
|
}
|
|
281
|
+
|
|
282
|
+
const server = typeof args["server"] === "string" ? args["server"].trim() : "";
|
|
283
|
+
const channel = typeof args["channel"] === "string" && args["channel"].trim() ? args["channel"].trim() : "live";
|
|
284
|
+
|
|
285
|
+
if (name === "transcribe_audio") {
|
|
286
|
+
const path = typeof args["path"] === "string" ? args["path"] : "";
|
|
287
|
+
if (!path) return failed("Which recording? Pass its path.");
|
|
288
|
+
let wav: Uint8Array;
|
|
289
|
+
try {
|
|
290
|
+
wav = (options.wavOf ?? wavOf)(path);
|
|
291
|
+
} catch (error) {
|
|
292
|
+
return failed((error as Error).message);
|
|
293
|
+
}
|
|
294
|
+
const answer = await askToHear(session, {
|
|
295
|
+
wav,
|
|
296
|
+
...(typeof args["language"] === "string" ? { language: args["language"] } : {}),
|
|
297
|
+
...(server ? { server, channel } : {}),
|
|
298
|
+
}, send, site);
|
|
299
|
+
if (!answer.ok) return failed(answer.error);
|
|
300
|
+
if (answer.heard.text === "") return text("Heard nothing in that recording.");
|
|
301
|
+
return text(answer.heard.message
|
|
302
|
+
? `${answer.heard.text}\n\nSaid in the room for ${channel} at ${server} as ${answer.heard.message.handle}.`
|
|
303
|
+
: answer.heard.text);
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
if (name === "trollbox_say") {
|
|
307
|
+
if (!server) return failed("Which room? Pass the server's address.");
|
|
308
|
+
const line = typeof args["text"] === "string" ? args["text"] : "";
|
|
309
|
+
if (!line.trim()) return failed("Say what? Pass the text.");
|
|
310
|
+
const response = await send(`${site}/api/v1/trollbox`, {
|
|
311
|
+
method: "POST",
|
|
312
|
+
headers,
|
|
313
|
+
body: JSON.stringify({ server, channel, body: line }),
|
|
314
|
+
});
|
|
315
|
+
if (!response.ok) return failed(await answerOf(response));
|
|
316
|
+
const body = (await response.json()) as { message?: TrollboxLine };
|
|
317
|
+
return text(body.message ? `Said, as ${body.message.handle}: ${body.message.body}` : "Said.");
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
if (name === "transcript_read") {
|
|
321
|
+
const url = typeof args["url"] === "string" ? args["url"].trim() : "";
|
|
322
|
+
if (!url) return failed("Which server? Pass its address.");
|
|
323
|
+
const got = await readTranscript(
|
|
324
|
+
{ url, key: typeof args["key"] === "string" && args["key"] ? args["key"] : null },
|
|
325
|
+
channel === "live" ? "main" : channel,
|
|
326
|
+
typeof args["after"] === "number" ? args["after"] : 0,
|
|
327
|
+
send,
|
|
328
|
+
);
|
|
329
|
+
if (!got.ok) return failed(got.error);
|
|
330
|
+
if (got.answer.recent.length === 0) {
|
|
331
|
+
return text(got.answer.error
|
|
332
|
+
? `Nothing yet: ${got.answer.error}`
|
|
333
|
+
: "Nothing said yet. The server has just started listening; ask again in a few seconds.");
|
|
334
|
+
}
|
|
335
|
+
return text(got.answer.recent.map((line) => `${new Date(line.at).toISOString()} ${line.text}`).join("\n"));
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
if (name === "trollbox_read") {
|
|
339
|
+
if (!server) return failed("Which room? Pass the server's address.");
|
|
340
|
+
const url = new URL(`${site}/api/v1/trollbox`);
|
|
341
|
+
url.searchParams.set("server", server);
|
|
342
|
+
url.searchParams.set("channel", channel);
|
|
343
|
+
if (typeof args["after"] === "string" && args["after"]) url.searchParams.set("after", args["after"]);
|
|
344
|
+
const response = await send(url.toString(), { headers });
|
|
345
|
+
if (!response.ok) return failed(await answerOf(response));
|
|
346
|
+
const body = (await response.json()) as { messages?: TrollboxLine[] };
|
|
347
|
+
const lines = body.messages ?? [];
|
|
348
|
+
return text(lines.length === 0 ? `Nobody has said anything in the room for ${channel} at ${server}.` : lines.map(said).join("\n"));
|
|
349
|
+
}
|
|
207
350
|
} catch (error) {
|
|
208
351
|
return failed(`Could not reach ${site}: ${(error as Error).message}`);
|
|
209
352
|
}
|
|
@@ -219,7 +362,7 @@ export async function handleMessage(message: Request, options: McpOptions = {}):
|
|
|
219
362
|
return reply({
|
|
220
363
|
protocolVersion: PROTOCOL_VERSION,
|
|
221
364
|
capabilities: { tools: { listChanged: false } },
|
|
222
|
-
serverInfo: { name: "nixamp", title: "nixamp watch parties", version: "1" },
|
|
365
|
+
serverInfo: { name: "nixamp", title: "nixamp: watch parties and rooms", version: "1" },
|
|
223
366
|
instructions:
|
|
224
367
|
"Watch parties on nixamp. A party lives on the site hosting the film and is bridged here as a room every nixamp client can join. Codes are the ones that site shows; positions are seconds into the film.",
|
|
225
368
|
});
|