nixamp 0.21.4 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +75 -0
- package/dist/captions.d.ts +75 -0
- package/dist/captions.js +308 -0
- package/dist/main.js +47 -2
- package/dist/mcp.d.ts +3 -0
- package/dist/mcp.js +134 -2
- package/dist/server.d.ts +6 -0
- package/dist/server.js +151 -1
- package/dist/speech.d.ts +96 -0
- package/dist/speech.js +311 -0
- package/dist/transcribe.d.ts +43 -0
- package/dist/transcribe.js +155 -0
- package/dist/transcript.d.ts +40 -0
- package/dist/transcript.js +120 -0
- package/package.json +5 -2
- package/src/captions.ts +334 -0
- package/src/main.ts +47 -2
- package/src/mcp.ts +145 -2
- package/src/server.ts +153 -1
- package/src/speech.ts +324 -0
- package/src/transcribe.ts +177 -0
- package/src/transcript.ts +150 -0
- package/web/dist/.well-known/openaccess.json +29 -0
- package/web/dist/assets/{hls-3VKVEQE3-RSItUqmr.js → hls-3VKVEQE3-DOUPX_xZ.js} +1 -1
- package/web/dist/assets/index-BIL7sVli.js +1 -0
- package/web/dist/assets/index-D3Is8ykj.css +1 -0
- package/web/dist/assets/{mpegts-LO6RVLD6-B5VNNeMl.js → mpegts-LO6RVLD6-C9gfFFrM.js} +1 -1
- package/web/dist/assets/{mpegts-CGsP3zeI.js → mpegts-XL4bjo7s.js} +1 -1
- package/web/dist/index.html +21 -2
- package/web/dist/sw.js +7 -6
- package/web/dist/assets/index-DMVUa6vY.js +0 -1
- package/web/dist/assets/index-jOXfym7D.css +0 -1
package/dist/speech.js
ADDED
|
@@ -0,0 +1,311 @@
|
|
|
1
|
+
var __rewriteRelativeImportExtension = (this && this.__rewriteRelativeImportExtension) || function (path, preserveJsx) {
|
|
2
|
+
if (typeof path === "string" && /^\.\.?\//.test(path)) {
|
|
3
|
+
return path.replace(/\.(tsx)$|((?:\.d)?)((?:\.[^./]+?)?)\.([cm]?)ts$/i, function (m, tsx, d, ext, cm) {
|
|
4
|
+
return tsx ? preserveJsx ? ".jsx" : ".js" : d && (!ext || !cm) ? m : (d + ext + "." + cm.toLowerCase() + "js");
|
|
5
|
+
});
|
|
6
|
+
}
|
|
7
|
+
return path;
|
|
8
|
+
};
|
|
9
|
+
/**
|
|
10
|
+
* Speech to text: a line said into a microphone, heard by nixamp.com.
|
|
11
|
+
*
|
|
12
|
+
* The ear is Whisper, run through Transformers.js: an Apache-2.0 library
|
|
13
|
+
* carrying MIT-licensed models, on this machine's own CPU. Nothing leaves
|
|
14
|
+
* for a speech vendor, and nothing is billed. The web page, the CLI and the
|
|
15
|
+
* MCP tools all send the same thing -- a short WAV -- to the same route,
|
|
16
|
+
* and get words back; the route can also drop those words straight into a
|
|
17
|
+
* trollbox, which is what dictating a line to a room means.
|
|
18
|
+
*
|
|
19
|
+
* The library is an optional dependency on purpose. It is hundreds of
|
|
20
|
+
* megabytes with the ONNX runtime under it, native per platform, and the
|
|
21
|
+
* CLI tarball's promise is "pure JavaScript, runs anywhere a Node does".
|
|
22
|
+
* So a `nixamp serve` on a laptop answers 503 to this, and every client
|
|
23
|
+
* asks nixamp.com instead -- which is where a trollbox line has to be
|
|
24
|
+
* signed in anyway.
|
|
25
|
+
*
|
|
26
|
+
* A WAV is decoded here rather than by ffmpeg because nixamp.com has no
|
|
27
|
+
* ffmpeg (see the Dockerfile). Anything else is converted before it is
|
|
28
|
+
* sent: the browser resamples what it recorded, the CLI runs ffmpeg.
|
|
29
|
+
*/
|
|
30
|
+
import { join } from "node:path";
|
|
31
|
+
import { stateDir } from "./daemon.js";
|
|
32
|
+
/** What Whisper listens at. Everything is brought to this before it is heard. */
|
|
33
|
+
export const RATE = 16_000;
|
|
34
|
+
/** A trollbox line, said out loud, is seconds long; a minute is the ceiling. */
|
|
35
|
+
export const MAX_SECONDS = 60;
|
|
36
|
+
/** A minute of 16-bit mono at 16 kHz is under 2 MB; 48 kHz stereo is under 12. */
|
|
37
|
+
export const MAX_BYTES = 12 * 1024 * 1024;
|
|
38
|
+
/**
|
|
39
|
+
* How much one account may have heard in a minute, in seconds of sound.
|
|
40
|
+
* Counted in sound rather than asks because a live channel being captioned
|
|
41
|
+
* asks twelve times a minute for five seconds each, and a person dictating
|
|
42
|
+
* asks twice for thirty: the cost is the sound, not the call. Five minutes
|
|
43
|
+
* of sound a minute is four channels captioned, or a conversation, and
|
|
44
|
+
* not a firehose.
|
|
45
|
+
*/
|
|
46
|
+
export const SECONDS_PER_MINUTE = 300;
|
|
47
|
+
/** How many may wait for the one CPU. Past this, the honest answer is "later". */
|
|
48
|
+
export const QUEUE_LIMIT = 8;
|
|
49
|
+
export const DEFAULT_MODEL = "onnx-community/whisper-base";
|
|
50
|
+
export class SpeechError extends Error {
|
|
51
|
+
status;
|
|
52
|
+
constructor(message, status) {
|
|
53
|
+
super(message);
|
|
54
|
+
this.status = status;
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
/** Whether these bytes are a RIFF/WAVE file, whatever the request called them. */
|
|
58
|
+
export function isWav(bytes) {
|
|
59
|
+
if (bytes.length < 12)
|
|
60
|
+
return false;
|
|
61
|
+
const ascii = (at, length) => String.fromCharCode(...bytes.subarray(at, at + length));
|
|
62
|
+
return ascii(0, 4) === "RIFF" && ascii(8, 4) === "WAVE";
|
|
63
|
+
}
|
|
64
|
+
/**
|
|
65
|
+
* A WAV file's samples, mixed to mono. PCM of 8, 16, 24 or 32 bits, or
|
|
66
|
+
* 32-bit float; the WAVE_FORMAT_EXTENSIBLE wrapper around either. That is
|
|
67
|
+
* what every encoder that matters writes, including the one in the page.
|
|
68
|
+
*/
|
|
69
|
+
export function decodeWav(bytes) {
|
|
70
|
+
if (!isWav(bytes))
|
|
71
|
+
throw new SpeechError("send a WAV file", 415);
|
|
72
|
+
const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
73
|
+
let at = 12;
|
|
74
|
+
let format = 0;
|
|
75
|
+
let channels = 0;
|
|
76
|
+
let rate = 0;
|
|
77
|
+
let bits = 0;
|
|
78
|
+
let dataAt = -1;
|
|
79
|
+
let dataLength = 0;
|
|
80
|
+
while (at + 8 <= bytes.length) {
|
|
81
|
+
const id = String.fromCharCode(...bytes.subarray(at, at + 4));
|
|
82
|
+
const length = view.getUint32(at + 4, true);
|
|
83
|
+
const body = at + 8;
|
|
84
|
+
if (id === "fmt " && length >= 16) {
|
|
85
|
+
format = view.getUint16(body, true);
|
|
86
|
+
channels = view.getUint16(body + 2, true);
|
|
87
|
+
rate = view.getUint32(body + 4, true);
|
|
88
|
+
bits = view.getUint16(body + 14, true);
|
|
89
|
+
// Extensible: the real format is the first two bytes of the sub-format GUID.
|
|
90
|
+
if (format === 0xfffe && length >= 26)
|
|
91
|
+
format = view.getUint16(body + 24, true);
|
|
92
|
+
}
|
|
93
|
+
else if (id === "data") {
|
|
94
|
+
dataAt = body;
|
|
95
|
+
// A streamed WAV says 0 or 0xFFFFFFFF for a length it did not know yet.
|
|
96
|
+
dataLength = length === 0 || body + length > bytes.length ? bytes.length - body : length;
|
|
97
|
+
break;
|
|
98
|
+
}
|
|
99
|
+
at = body + length + (length & 1);
|
|
100
|
+
}
|
|
101
|
+
if (dataAt < 0 || channels < 1 || rate < 8000 || rate > 192_000)
|
|
102
|
+
throw new SpeechError("that WAV has no sound in it", 415);
|
|
103
|
+
const pcm = format === 1 && (bits === 8 || bits === 16 || bits === 24 || bits === 32);
|
|
104
|
+
const float = format === 3 && bits === 32;
|
|
105
|
+
if (!pcm && !float)
|
|
106
|
+
throw new SpeechError(`WAV format ${format} at ${bits} bits is not one this reads: send 16-bit PCM`, 415);
|
|
107
|
+
const width = bits / 8;
|
|
108
|
+
const frames = Math.floor(dataLength / (width * channels));
|
|
109
|
+
const samples = new Float32Array(frames);
|
|
110
|
+
for (let frame = 0; frame < frames; frame++) {
|
|
111
|
+
let sum = 0;
|
|
112
|
+
for (let channel = 0; channel < channels; channel++) {
|
|
113
|
+
const offset = dataAt + (frame * channels + channel) * width;
|
|
114
|
+
if (float)
|
|
115
|
+
sum += view.getFloat32(offset, true);
|
|
116
|
+
else if (bits === 8)
|
|
117
|
+
sum += bytes[offset] / 128 - 1;
|
|
118
|
+
else if (bits === 16)
|
|
119
|
+
sum += view.getInt16(offset, true) / 32768;
|
|
120
|
+
else if (bits === 24)
|
|
121
|
+
sum += ((bytes[offset + 2] << 24) | (bytes[offset + 1] << 16) | (bytes[offset] << 8)) / 2147483648;
|
|
122
|
+
else
|
|
123
|
+
sum += view.getInt32(offset, true) / 2147483648;
|
|
124
|
+
}
|
|
125
|
+
samples[frame] = sum / channels;
|
|
126
|
+
}
|
|
127
|
+
return { rate, channels, samples };
|
|
128
|
+
}
|
|
129
|
+
/**
|
|
130
|
+
* Samples at one rate, at another. Down is an average over each output
|
|
131
|
+
* sample's span, which is a crude low-pass and enough for speech; up is a
|
|
132
|
+
* straight line between neighbours. Whisper resamples nothing itself.
|
|
133
|
+
*/
|
|
134
|
+
export function resample(samples, from, to) {
|
|
135
|
+
if (from === to || samples.length === 0)
|
|
136
|
+
return samples;
|
|
137
|
+
const ratio = from / to;
|
|
138
|
+
const length = Math.max(1, Math.round(samples.length / ratio));
|
|
139
|
+
const out = new Float32Array(length);
|
|
140
|
+
if (ratio > 1) {
|
|
141
|
+
for (let i = 0; i < length; i++) {
|
|
142
|
+
const start = Math.floor(i * ratio);
|
|
143
|
+
const end = Math.min(samples.length, Math.max(start + 1, Math.floor((i + 1) * ratio)));
|
|
144
|
+
let sum = 0;
|
|
145
|
+
for (let j = start; j < end; j++)
|
|
146
|
+
sum += samples[j];
|
|
147
|
+
out[i] = sum / (end - start);
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
else {
|
|
151
|
+
for (let i = 0; i < length; i++) {
|
|
152
|
+
const position = i * ratio;
|
|
153
|
+
const left = Math.floor(position);
|
|
154
|
+
const right = Math.min(samples.length - 1, left + 1);
|
|
155
|
+
const mix = position - left;
|
|
156
|
+
out[i] = samples[left] * (1 - mix) + samples[right] * mix;
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
return out;
|
|
160
|
+
}
|
|
161
|
+
/** 16-bit mono PCM WAV bytes from samples: what the CLI's tests, and anybody, can send. */
|
|
162
|
+
export function encodeWav(samples, rate = RATE) {
|
|
163
|
+
const bytes = new Uint8Array(44 + samples.length * 2);
|
|
164
|
+
const view = new DataView(bytes.buffer);
|
|
165
|
+
const ascii = (at, text) => {
|
|
166
|
+
for (let i = 0; i < text.length; i++)
|
|
167
|
+
bytes[at + i] = text.charCodeAt(i);
|
|
168
|
+
};
|
|
169
|
+
ascii(0, "RIFF");
|
|
170
|
+
view.setUint32(4, 36 + samples.length * 2, true);
|
|
171
|
+
ascii(8, "WAVE");
|
|
172
|
+
ascii(12, "fmt ");
|
|
173
|
+
view.setUint32(16, 16, true);
|
|
174
|
+
view.setUint16(20, 1, true);
|
|
175
|
+
view.setUint16(22, 1, true);
|
|
176
|
+
view.setUint32(24, rate, true);
|
|
177
|
+
view.setUint32(28, rate * 2, true);
|
|
178
|
+
view.setUint16(32, 2, true);
|
|
179
|
+
view.setUint16(34, 16, true);
|
|
180
|
+
ascii(36, "data");
|
|
181
|
+
view.setUint32(40, samples.length * 2, true);
|
|
182
|
+
for (let i = 0; i < samples.length; i++) {
|
|
183
|
+
const clipped = Math.max(-1, Math.min(1, samples[i]));
|
|
184
|
+
view.setInt16(44 + i * 2, clipped < 0 ? clipped * 32768 : clipped * 32767, true);
|
|
185
|
+
}
|
|
186
|
+
return bytes;
|
|
187
|
+
}
|
|
188
|
+
/** Whisper's own spacing and blank-audio tokens, tidied into a line. */
|
|
189
|
+
export function tidy(text) {
|
|
190
|
+
return text.replace(/\[[A-Z_ ]+\]|\([A-Za-z ]+\)/g, " ").replace(/\s+/g, " ").trim();
|
|
191
|
+
}
|
|
192
|
+
/** A two-letter language code, or nothing: Whisper guesses when not told. */
|
|
193
|
+
export function languageOf(value) {
|
|
194
|
+
if (typeof value !== "string")
|
|
195
|
+
return undefined;
|
|
196
|
+
const code = value.trim().toLowerCase();
|
|
197
|
+
return /^[a-z]{2}$/.test(code) ? code : undefined;
|
|
198
|
+
}
|
|
199
|
+
async function loadWhisper(model, cacheDir) {
|
|
200
|
+
// By name held in a variable: an optional dependency that is not on disk
|
|
201
|
+
// must fail here, at the first ask, and not when the file is imported.
|
|
202
|
+
const name = "@huggingface/transformers";
|
|
203
|
+
let transformers;
|
|
204
|
+
try {
|
|
205
|
+
transformers = (await import(__rewriteRelativeImportExtension(name)));
|
|
206
|
+
}
|
|
207
|
+
catch {
|
|
208
|
+
throw new SpeechError("this nixamp cannot hear: @huggingface/transformers is not installed here. nixamp.com can.", 503);
|
|
209
|
+
}
|
|
210
|
+
transformers.env.cacheDir = cacheDir;
|
|
211
|
+
const recognize = await transformers.pipeline("automatic-speech-recognition", model, { dtype: "q8" });
|
|
212
|
+
return async (pcm, { language }) => {
|
|
213
|
+
const heard = await recognize(pcm, {
|
|
214
|
+
// Whisper hears thirty seconds at a time; longer is heard in overlapping pieces.
|
|
215
|
+
chunk_length_s: 30,
|
|
216
|
+
stride_length_s: 5,
|
|
217
|
+
...(language ? { language, task: "transcribe" } : {}),
|
|
218
|
+
});
|
|
219
|
+
return { text: Array.isArray(heard) ? heard.map((piece) => piece.text).join(" ") : heard.text };
|
|
220
|
+
};
|
|
221
|
+
}
|
|
222
|
+
export class Speech {
|
|
223
|
+
model;
|
|
224
|
+
cacheDir;
|
|
225
|
+
load;
|
|
226
|
+
now;
|
|
227
|
+
recognizer = null;
|
|
228
|
+
/** One at a time: the model is CPU-bound, and two at once is slower than two in turn. */
|
|
229
|
+
tail = Promise.resolve();
|
|
230
|
+
waiting = 0;
|
|
231
|
+
asked = new Map();
|
|
232
|
+
constructor(options = {}) {
|
|
233
|
+
this.model = options.model ?? process.env["NIXAMP_STT_MODEL"] ?? DEFAULT_MODEL;
|
|
234
|
+
this.cacheDir = options.cacheDir ?? process.env["NIXAMP_STT_CACHE"] ?? join(stateDir(), "models");
|
|
235
|
+
this.load = options.load ?? loadWhisper;
|
|
236
|
+
this.now = options.now ?? (() => Date.now());
|
|
237
|
+
}
|
|
238
|
+
ear() {
|
|
239
|
+
this.recognizer ??= this.load(this.model, this.cacheDir).catch((error) => {
|
|
240
|
+
// A failed load is tried again next time, not remembered forever.
|
|
241
|
+
this.recognizer = null;
|
|
242
|
+
throw error;
|
|
243
|
+
});
|
|
244
|
+
return this.recognizer;
|
|
245
|
+
}
|
|
246
|
+
/**
|
|
247
|
+
* Load the model now, so the first person to speak is not the one who
|
|
248
|
+
* waits for the download. Says whether it could; never throws.
|
|
249
|
+
*/
|
|
250
|
+
async warm() {
|
|
251
|
+
try {
|
|
252
|
+
await this.ear();
|
|
253
|
+
return true;
|
|
254
|
+
}
|
|
255
|
+
catch {
|
|
256
|
+
return false;
|
|
257
|
+
}
|
|
258
|
+
}
|
|
259
|
+
/** Whether this account may have this much heard now, and the bookkeeping if so. */
|
|
260
|
+
allow(accountId, seconds) {
|
|
261
|
+
const minute = Math.floor(this.now() / 60_000);
|
|
262
|
+
const record = this.asked.get(accountId) ?? { minute, seconds: 0 };
|
|
263
|
+
if (record.minute !== minute) {
|
|
264
|
+
record.minute = minute;
|
|
265
|
+
record.seconds = 0;
|
|
266
|
+
}
|
|
267
|
+
if (record.seconds + seconds > SECONDS_PER_MINUTE) {
|
|
268
|
+
throw new SpeechError(`${SECONDS_PER_MINUTE} seconds of sound a minute is plenty; try again in a moment`, 429);
|
|
269
|
+
}
|
|
270
|
+
record.seconds += seconds;
|
|
271
|
+
this.asked.set(accountId, record);
|
|
272
|
+
if (this.asked.size > 5000) {
|
|
273
|
+
for (const [id, one] of this.asked)
|
|
274
|
+
if (one.minute !== minute)
|
|
275
|
+
this.asked.delete(id);
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
/**
|
|
279
|
+
* The words in a WAV. Refuses what is not a WAV, or is too long, or more
|
|
280
|
+
* than the account may have heard this minute, with a status.
|
|
281
|
+
*/
|
|
282
|
+
async transcribe(bytes, options = {}) {
|
|
283
|
+
if (bytes.length > MAX_BYTES)
|
|
284
|
+
throw new SpeechError(`that is too much sound: ${MAX_SECONDS} seconds at most`, 413);
|
|
285
|
+
const wav = decodeWav(bytes);
|
|
286
|
+
const seconds = wav.samples.length / wav.rate;
|
|
287
|
+
if (seconds > MAX_SECONDS)
|
|
288
|
+
throw new SpeechError(`that is ${Math.round(seconds)} seconds; ${MAX_SECONDS} at most`, 413);
|
|
289
|
+
if (options.by)
|
|
290
|
+
this.allow(options.by, seconds);
|
|
291
|
+
if (wav.samples.length < wav.rate / 10)
|
|
292
|
+
return { text: "", seconds };
|
|
293
|
+
const pcm = resample(wav.samples, wav.rate, RATE);
|
|
294
|
+
if (this.waiting >= QUEUE_LIMIT)
|
|
295
|
+
throw new SpeechError("too many people are talking at once; try again in a moment", 503);
|
|
296
|
+
this.waiting += 1;
|
|
297
|
+
const turn = this.tail.then(async () => {
|
|
298
|
+
const recognize = await this.ear();
|
|
299
|
+
return recognize(pcm, options.language ? { language: options.language } : {});
|
|
300
|
+
});
|
|
301
|
+
// The queue moves on whether or not this one was heard.
|
|
302
|
+
this.tail = turn.catch(() => undefined);
|
|
303
|
+
try {
|
|
304
|
+
const heard = await turn;
|
|
305
|
+
return { text: tidy(heard.text), seconds };
|
|
306
|
+
}
|
|
307
|
+
finally {
|
|
308
|
+
this.waiting -= 1;
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
}
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
import { type Tools } from "./audio.ts";
|
|
2
|
+
import { type Session } from "./session.ts";
|
|
3
|
+
/** Where the sound may be sent, as a request. */
|
|
4
|
+
export interface Ask {
|
|
5
|
+
wav: Uint8Array;
|
|
6
|
+
language?: string;
|
|
7
|
+
/** A room to post the words to: the server's address, and its channel. */
|
|
8
|
+
server?: string;
|
|
9
|
+
channel?: string;
|
|
10
|
+
}
|
|
11
|
+
export interface Heard {
|
|
12
|
+
text: string;
|
|
13
|
+
seconds: number;
|
|
14
|
+
model?: string;
|
|
15
|
+
/** The trollbox line, when a room was named. */
|
|
16
|
+
message?: {
|
|
17
|
+
id: string;
|
|
18
|
+
handle: string;
|
|
19
|
+
body: string;
|
|
20
|
+
createdAt: string;
|
|
21
|
+
};
|
|
22
|
+
}
|
|
23
|
+
export type Answer = {
|
|
24
|
+
ok: true;
|
|
25
|
+
heard: Heard;
|
|
26
|
+
} | {
|
|
27
|
+
ok: false;
|
|
28
|
+
status: number;
|
|
29
|
+
error: string;
|
|
30
|
+
};
|
|
31
|
+
/**
|
|
32
|
+
* The file as a WAV: as it is when it already is one, through ffmpeg to
|
|
33
|
+
* 16 kHz mono otherwise. Throws a sentence when neither is possible.
|
|
34
|
+
*/
|
|
35
|
+
export declare function wavOf(path: string, tools?: () => Pick<Tools, "ffmpeg" | "carries">): Uint8Array;
|
|
36
|
+
/** The ask, made: one POST to the site the session belongs to. */
|
|
37
|
+
export declare function askToHear(session: Pick<Session, "site" | "token">, ask: Ask, fetcher?: typeof fetch, site?: string): Promise<Answer>;
|
|
38
|
+
export interface TranscribeDeps {
|
|
39
|
+
fetcher?: typeof fetch;
|
|
40
|
+
wavOf?: typeof wavOf;
|
|
41
|
+
session?: Pick<Session, "site" | "token"> | null;
|
|
42
|
+
}
|
|
43
|
+
export declare function transcribe(argv: string[], deps?: TranscribeDeps): Promise<number>;
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `nixamp transcribe` -- a recording in, the words out, and into a room.
|
|
3
|
+
*
|
|
4
|
+
* The ear is nixamp.com's (see speech.ts): this machine sends a WAV and
|
|
5
|
+
* gets text back, signed in as whoever it is signed in as. Anything that
|
|
6
|
+
* is not already a WAV goes through the ffmpeg nixamp plays with, to 16 kHz
|
|
7
|
+
* mono, which is what the ear listens at and the smallest thing to send.
|
|
8
|
+
*
|
|
9
|
+
* With `--say SERVER`, the words are posted to that server's trollbox as
|
|
10
|
+
* this account, by the same rules as typing them: one line a second, and
|
|
11
|
+
* signed with the public handle. The MCP tool of the same name is this
|
|
12
|
+
* function with a different front.
|
|
13
|
+
*/
|
|
14
|
+
import { spawnSync } from "node:child_process";
|
|
15
|
+
import { readFileSync } from "node:fs";
|
|
16
|
+
import { detectTools } from "./audio.js";
|
|
17
|
+
import { readSession } from "./session.js";
|
|
18
|
+
import { RATE, isWav } from "./speech.js";
|
|
19
|
+
const HELP = `nixamp transcribe — say it, and have it written down.
|
|
20
|
+
|
|
21
|
+
nixamp transcribe FILE the words in a recording
|
|
22
|
+
nixamp transcribe FILE --say SERVER and post them to that server's trollbox
|
|
23
|
+
nixamp transcribe FILE --say SERVER --channel ID to one channel's room (default: live)
|
|
24
|
+
nixamp transcribe FILE --language de when Whisper should not guess
|
|
25
|
+
nixamp transcribe FILE --json the answer as JSON
|
|
26
|
+
|
|
27
|
+
FILE is any recording ffmpeg can read; a WAV needs no ffmpeg at all. The
|
|
28
|
+
hearing is done by nixamp.com with an open-source model on its own CPU, so
|
|
29
|
+
this needs a sign-in (\`nixamp login\`) and nothing else. Up to a minute at
|
|
30
|
+
a time.
|
|
31
|
+
|
|
32
|
+
SERVER is the address of the nixamp whose room it is, as in its share link:
|
|
33
|
+
https://server1.chovy.nixamp.com:4321. The room is that server's own stream
|
|
34
|
+
unless --channel names one of its channels.
|
|
35
|
+
`;
|
|
36
|
+
/**
|
|
37
|
+
* The file as a WAV: as it is when it already is one, through ffmpeg to
|
|
38
|
+
* 16 kHz mono otherwise. Throws a sentence when neither is possible.
|
|
39
|
+
*/
|
|
40
|
+
export function wavOf(path, tools = detectTools) {
|
|
41
|
+
let bytes;
|
|
42
|
+
try {
|
|
43
|
+
bytes = new Uint8Array(readFileSync(path));
|
|
44
|
+
}
|
|
45
|
+
catch {
|
|
46
|
+
throw new Error(`cannot read ${path}`);
|
|
47
|
+
}
|
|
48
|
+
if (isWav(bytes))
|
|
49
|
+
return bytes;
|
|
50
|
+
const found = tools();
|
|
51
|
+
if (found.carries === false)
|
|
52
|
+
throw new Error(`${path} is not a WAV, and there is no ffmpeg here to convert it. Install ffmpeg, or record a WAV.`);
|
|
53
|
+
const [command, ...prefix] = found.ffmpeg;
|
|
54
|
+
const run = spawnSync(command, [...prefix, "-v", "error", "-i", path, "-vn", "-ac", "1", "-ar", String(RATE), "-f", "wav", "-"], {
|
|
55
|
+
maxBuffer: 64 * 1024 * 1024,
|
|
56
|
+
});
|
|
57
|
+
if (run.error || run.status !== 0 || !run.stdout || run.stdout.length < 44) {
|
|
58
|
+
const why = run.stderr ? run.stderr.toString("utf8").trim().split("\n").pop() : run.error?.message;
|
|
59
|
+
throw new Error(`ffmpeg could not read ${path}${why ? `: ${why}` : ""}`);
|
|
60
|
+
}
|
|
61
|
+
return new Uint8Array(run.stdout);
|
|
62
|
+
}
|
|
63
|
+
/** The ask, made: one POST to the site the session belongs to. */
|
|
64
|
+
export async function askToHear(session, ask, fetcher = fetch, site = session.site) {
|
|
65
|
+
const url = new URL(`${site.replace(/\/+$/, "")}/api/v1/speech/transcribe`);
|
|
66
|
+
if (ask.language)
|
|
67
|
+
url.searchParams.set("language", ask.language);
|
|
68
|
+
if (ask.server) {
|
|
69
|
+
url.searchParams.set("server", ask.server);
|
|
70
|
+
url.searchParams.set("channel", ask.channel || "live");
|
|
71
|
+
}
|
|
72
|
+
let response;
|
|
73
|
+
try {
|
|
74
|
+
response = await fetcher(url.toString(), {
|
|
75
|
+
method: "POST",
|
|
76
|
+
headers: { authorization: `Bearer ${session.token}`, "content-type": "audio/wav" },
|
|
77
|
+
body: new Blob([ask.wav.buffer.slice(ask.wav.byteOffset, ask.wav.byteOffset + ask.wav.byteLength)]),
|
|
78
|
+
});
|
|
79
|
+
}
|
|
80
|
+
catch (error) {
|
|
81
|
+
return { ok: false, status: 0, error: `could not reach ${url.origin}: ${error.message}` };
|
|
82
|
+
}
|
|
83
|
+
const body = (await response.json().catch(() => ({})));
|
|
84
|
+
if (!response.ok)
|
|
85
|
+
return { ok: false, status: response.status, error: body.error ?? `nixamp answered ${response.status}` };
|
|
86
|
+
return {
|
|
87
|
+
ok: true,
|
|
88
|
+
heard: {
|
|
89
|
+
text: body.text ?? "",
|
|
90
|
+
seconds: body.seconds ?? 0,
|
|
91
|
+
...(body.model ? { model: body.model } : {}),
|
|
92
|
+
...(body.message ? { message: body.message } : {}),
|
|
93
|
+
},
|
|
94
|
+
};
|
|
95
|
+
}
|
|
96
|
+
function flag(argv, name) {
|
|
97
|
+
const at = argv.indexOf(name);
|
|
98
|
+
return at === -1 ? undefined : argv[at + 1];
|
|
99
|
+
}
|
|
100
|
+
export async function transcribe(argv, deps = {}) {
|
|
101
|
+
if (argv.length === 0 || argv.includes("--help") || argv.includes("-h") || argv[0] === "help") {
|
|
102
|
+
console.log(HELP);
|
|
103
|
+
return argv.length === 0 ? 64 : 0;
|
|
104
|
+
}
|
|
105
|
+
const session = deps.session === undefined ? readSession() : deps.session;
|
|
106
|
+
if (session === null) {
|
|
107
|
+
console.error("nixamp: not signed in. Try `nixamp login`.");
|
|
108
|
+
return 1;
|
|
109
|
+
}
|
|
110
|
+
const withValue = new Set(["--say", "--channel", "--language", "--site"]);
|
|
111
|
+
const file = argv.find((one, at) => !one.startsWith("-") && !(at > 0 && withValue.has(argv[at - 1])));
|
|
112
|
+
if (!file) {
|
|
113
|
+
console.error("nixamp: which recording? `nixamp transcribe clip.m4a`.");
|
|
114
|
+
return 64;
|
|
115
|
+
}
|
|
116
|
+
const server = flag(argv, "--say");
|
|
117
|
+
if (argv.includes("--say") && !server) {
|
|
118
|
+
console.error("nixamp: --say needs the server's address, as in its share link.");
|
|
119
|
+
return 64;
|
|
120
|
+
}
|
|
121
|
+
let wav;
|
|
122
|
+
try {
|
|
123
|
+
wav = (deps.wavOf ?? wavOf)(file);
|
|
124
|
+
}
|
|
125
|
+
catch (error) {
|
|
126
|
+
console.error(`nixamp: ${error.message}`);
|
|
127
|
+
return 1;
|
|
128
|
+
}
|
|
129
|
+
const ask = {
|
|
130
|
+
wav,
|
|
131
|
+
...(flag(argv, "--language") ? { language: flag(argv, "--language") } : {}),
|
|
132
|
+
...(server ? { server, channel: flag(argv, "--channel") ?? "live" } : {}),
|
|
133
|
+
};
|
|
134
|
+
const answer = await askToHear(session, ask, deps.fetcher ?? fetch, flag(argv, "--site") ?? session.site);
|
|
135
|
+
if (!answer.ok) {
|
|
136
|
+
console.error(`nixamp: ${answer.error}`);
|
|
137
|
+
return 1;
|
|
138
|
+
}
|
|
139
|
+
if (argv.includes("--json")) {
|
|
140
|
+
console.log(JSON.stringify(answer.heard, null, 2));
|
|
141
|
+
return 0;
|
|
142
|
+
}
|
|
143
|
+
if (answer.heard.text === "") {
|
|
144
|
+
console.error("nixamp: heard nothing in that.");
|
|
145
|
+
return 1;
|
|
146
|
+
}
|
|
147
|
+
console.log(answer.heard.text);
|
|
148
|
+
if (answer.heard.message) {
|
|
149
|
+
console.error(` Said in the room for ${ask.channel} at ${server} as ${answer.heard.message.handle}.`);
|
|
150
|
+
}
|
|
151
|
+
else if (server) {
|
|
152
|
+
console.error(" Nothing was posted: there were no words to post.");
|
|
153
|
+
}
|
|
154
|
+
return 0;
|
|
155
|
+
}
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
export interface TranscriptLine {
|
|
2
|
+
channel: string;
|
|
3
|
+
at: number;
|
|
4
|
+
until: number;
|
|
5
|
+
text: string;
|
|
6
|
+
}
|
|
7
|
+
export interface TranscriptAnswer {
|
|
8
|
+
channel: string;
|
|
9
|
+
backlog: number;
|
|
10
|
+
now: number;
|
|
11
|
+
on: boolean;
|
|
12
|
+
lines: number;
|
|
13
|
+
error: string;
|
|
14
|
+
/** The recent lines, oldest first. Named `lines` on the wire; renamed here so the count above keeps its name. */
|
|
15
|
+
recent: TranscriptLine[];
|
|
16
|
+
}
|
|
17
|
+
export type Fetched = {
|
|
18
|
+
ok: true;
|
|
19
|
+
answer: TranscriptAnswer;
|
|
20
|
+
} | {
|
|
21
|
+
ok: false;
|
|
22
|
+
status: number;
|
|
23
|
+
error: string;
|
|
24
|
+
};
|
|
25
|
+
/** One ask for a channel's recent lines, after a moment when given. */
|
|
26
|
+
export declare function readTranscript(target: {
|
|
27
|
+
url: string;
|
|
28
|
+
key: string | null;
|
|
29
|
+
}, channel: string, after?: number, fetcher?: typeof fetch): Promise<Fetched>;
|
|
30
|
+
/** A line as the terminal prints it: the time its sound was heard, then the words. */
|
|
31
|
+
export declare function printed(line: TranscriptLine): string;
|
|
32
|
+
export interface TranscriptDeps {
|
|
33
|
+
fetcher?: typeof fetch;
|
|
34
|
+
/** How long --follow waits between asks. Short in the tests. */
|
|
35
|
+
everyMs?: number;
|
|
36
|
+
/** How many asks --follow makes before it stops. Forever, except in the tests. */
|
|
37
|
+
polls?: number;
|
|
38
|
+
sleep?: (ms: number) => Promise<void>;
|
|
39
|
+
}
|
|
40
|
+
export declare function transcript(argv: string[], deps?: TranscriptDeps): Promise<number>;
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `nixamp transcript` -- what a channel is saying, in the terminal.
|
|
3
|
+
*
|
|
4
|
+
* The server carrying the channel captions it (see captions.ts) and keeps
|
|
5
|
+
* the recent lines; this asks for them, and with --follow keeps asking, a
|
|
6
|
+
* poll every few seconds with the last line's time so nothing is printed
|
|
7
|
+
* twice. A poll rather than the SSE the page uses because each ask also
|
|
8
|
+
* keeps the captioner alive, which is what a terminal that is still
|
|
9
|
+
* reading wants, and there is nothing to parse.
|
|
10
|
+
*
|
|
11
|
+
* Pointed at a server the way `nixamp admin` is: --url and --key from its
|
|
12
|
+
* share link, or the daemon on this machine when neither is given.
|
|
13
|
+
*/
|
|
14
|
+
import { resolveTarget } from "./admin.js";
|
|
15
|
+
import { KEY_HEADER } from "./share.js";
|
|
16
|
+
const HELP = `nixamp transcript — what a channel is saying, written down.
|
|
17
|
+
|
|
18
|
+
nixamp transcript --channel ID the recent lines from this machine's daemon
|
|
19
|
+
nixamp transcript --url URL --key K --channel ID from another server, with its share link
|
|
20
|
+
nixamp transcript ... --follow and keep printing as it speaks
|
|
21
|
+
nixamp transcript ... --json the lines as JSON
|
|
22
|
+
|
|
23
|
+
ID is the channel's id as the server names it (the address bar says it when
|
|
24
|
+
you are watching one). The server captions a channel while somebody is asking
|
|
25
|
+
for the transcript, with nixamp.com's ear; it needs an ffmpeg and a sign-in
|
|
26
|
+
(\`nixamp login\`) on that server.
|
|
27
|
+
`;
|
|
28
|
+
/** One ask for a channel's recent lines, after a moment when given. */
|
|
29
|
+
export async function readTranscript(target, channel, after = 0, fetcher = fetch) {
|
|
30
|
+
const url = new URL(`${target.url.replace(/\/+$/, "")}/api/channels/${encodeURIComponent(channel)}/transcript`);
|
|
31
|
+
if (after > 0)
|
|
32
|
+
url.searchParams.set("after", String(after));
|
|
33
|
+
let response;
|
|
34
|
+
try {
|
|
35
|
+
response = await fetcher(url.toString(), { headers: target.key ? { [KEY_HEADER]: target.key } : {} });
|
|
36
|
+
}
|
|
37
|
+
catch (error) {
|
|
38
|
+
return { ok: false, status: 0, error: `could not reach ${url.origin}: ${error.message}` };
|
|
39
|
+
}
|
|
40
|
+
const body = (await response.json().catch(() => ({})));
|
|
41
|
+
if (!response.ok)
|
|
42
|
+
return { ok: false, status: response.status, error: body.error ?? `the server answered ${response.status}` };
|
|
43
|
+
const recent = Array.isArray(body.lines) ? body.lines : [];
|
|
44
|
+
return {
|
|
45
|
+
ok: true,
|
|
46
|
+
answer: {
|
|
47
|
+
channel: body.channel ?? channel,
|
|
48
|
+
backlog: body.backlog ?? 0,
|
|
49
|
+
now: body.now ?? Date.now(),
|
|
50
|
+
on: body.on ?? false,
|
|
51
|
+
lines: recent.length,
|
|
52
|
+
error: body.error ?? "",
|
|
53
|
+
recent,
|
|
54
|
+
},
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
/** A line as the terminal prints it: the time its sound was heard, then the words. */
|
|
58
|
+
export function printed(line) {
|
|
59
|
+
const at = new Date(line.at);
|
|
60
|
+
const clock = Number.isNaN(at.getTime()) ? "--:--:--" : at.toLocaleTimeString([], { hour: "2-digit", minute: "2-digit", second: "2-digit" });
|
|
61
|
+
return `${clock} ${line.text}`;
|
|
62
|
+
}
|
|
63
|
+
function flag(argv, name) {
|
|
64
|
+
const at = argv.indexOf(name);
|
|
65
|
+
return at === -1 ? undefined : argv[at + 1];
|
|
66
|
+
}
|
|
67
|
+
export async function transcript(argv, deps = {}) {
|
|
68
|
+
if (argv.includes("--help") || argv.includes("-h") || argv[0] === "help") {
|
|
69
|
+
console.log(HELP);
|
|
70
|
+
return 0;
|
|
71
|
+
}
|
|
72
|
+
let target;
|
|
73
|
+
try {
|
|
74
|
+
target = resolveTarget(argv);
|
|
75
|
+
}
|
|
76
|
+
catch (error) {
|
|
77
|
+
console.error(error.message);
|
|
78
|
+
return 1;
|
|
79
|
+
}
|
|
80
|
+
const channel = flag(argv, "--channel") ?? "main";
|
|
81
|
+
const follow = argv.includes("--follow") || argv.includes("-f");
|
|
82
|
+
const asJson = argv.includes("--json");
|
|
83
|
+
const fetcher = deps.fetcher ?? fetch;
|
|
84
|
+
const sleep = deps.sleep ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms)));
|
|
85
|
+
let after = 0;
|
|
86
|
+
let polls = 0;
|
|
87
|
+
for (;;) {
|
|
88
|
+
const got = await readTranscript(target, channel, after, fetcher);
|
|
89
|
+
if (!got.ok) {
|
|
90
|
+
console.error(`nixamp: ${got.error}`);
|
|
91
|
+
return 1;
|
|
92
|
+
}
|
|
93
|
+
if (asJson) {
|
|
94
|
+
console.log(JSON.stringify(got.answer.recent, null, 2));
|
|
95
|
+
if (!follow)
|
|
96
|
+
return 0;
|
|
97
|
+
}
|
|
98
|
+
else {
|
|
99
|
+
if (after === 0 && got.answer.recent.length === 0 && !follow) {
|
|
100
|
+
console.log(got.answer.error
|
|
101
|
+
? `Nothing yet: ${got.answer.error}`
|
|
102
|
+
: "Nothing said yet. The server has just started listening; ask again in a few seconds, or --follow.");
|
|
103
|
+
return 0;
|
|
104
|
+
}
|
|
105
|
+
for (const line of got.answer.recent)
|
|
106
|
+
console.log(printed(line));
|
|
107
|
+
if (!follow)
|
|
108
|
+
return 0;
|
|
109
|
+
if (after === 0 && got.answer.error)
|
|
110
|
+
console.error(`nixamp: ${got.answer.error}`);
|
|
111
|
+
}
|
|
112
|
+
const last = got.answer.recent[got.answer.recent.length - 1];
|
|
113
|
+
if (last)
|
|
114
|
+
after = last.at;
|
|
115
|
+
polls += 1;
|
|
116
|
+
if (deps.polls !== undefined && polls >= deps.polls)
|
|
117
|
+
return 0;
|
|
118
|
+
await sleep(deps.everyMs ?? 5000);
|
|
119
|
+
}
|
|
120
|
+
}
|