nixamp 0.21.4 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +75 -0
- package/dist/captions.d.ts +75 -0
- package/dist/captions.js +308 -0
- package/dist/main.js +47 -2
- package/dist/mcp.d.ts +3 -0
- package/dist/mcp.js +134 -2
- package/dist/server.d.ts +6 -0
- package/dist/server.js +151 -1
- package/dist/speech.d.ts +96 -0
- package/dist/speech.js +311 -0
- package/dist/transcribe.d.ts +43 -0
- package/dist/transcribe.js +155 -0
- package/dist/transcript.d.ts +40 -0
- package/dist/transcript.js +120 -0
- package/package.json +5 -2
- package/src/captions.ts +334 -0
- package/src/main.ts +47 -2
- package/src/mcp.ts +145 -2
- package/src/server.ts +153 -1
- package/src/speech.ts +324 -0
- package/src/transcribe.ts +177 -0
- package/src/transcript.ts +150 -0
- package/web/dist/.well-known/openaccess.json +29 -0
- package/web/dist/assets/{hls-3VKVEQE3-RSItUqmr.js → hls-3VKVEQE3-DOUPX_xZ.js} +1 -1
- package/web/dist/assets/index-BIL7sVli.js +1 -0
- package/web/dist/assets/index-D3Is8ykj.css +1 -0
- package/web/dist/assets/{mpegts-LO6RVLD6-B5VNNeMl.js → mpegts-LO6RVLD6-C9gfFFrM.js} +1 -1
- package/web/dist/assets/{mpegts-CGsP3zeI.js → mpegts-XL4bjo7s.js} +1 -1
- package/web/dist/index.html +21 -2
- package/web/dist/sw.js +7 -6
- package/web/dist/assets/index-DMVUa6vY.js +0 -1
- package/web/dist/assets/index-jOXfym7D.css +0 -1
package/dist/mcp.js
CHANGED
|
@@ -8,7 +8,9 @@
|
|
|
8
8
|
*
|
|
9
9
|
* What it offers is the watch party, because that is the part of nixamp an
|
|
10
10
|
* agent can usefully do something with: find the party, say where it is, put
|
|
11
|
-
* one on the air, move everybody to the same second.
|
|
11
|
+
* one on the air, move everybody to the same second. And the room: hear a
|
|
12
|
+
* recording (nixamp.com's own ear, see speech.ts), say a line in a trollbox,
|
|
13
|
+
* read one back. It signs in as whoever
|
|
12
14
|
* this machine is signed in as -- the session on disk, or NIXAMP_TOKEN --
|
|
13
15
|
* because an agent holding its own credential is a credential nobody revokes.
|
|
14
16
|
*
|
|
@@ -18,6 +20,8 @@
|
|
|
18
20
|
import { createInterface } from "node:readline";
|
|
19
21
|
import { clock } from "./party.js";
|
|
20
22
|
import { readSession } from "./session.js";
|
|
23
|
+
import { askToHear, wavOf } from "./transcribe.js";
|
|
24
|
+
import { readTranscript } from "./transcript.js";
|
|
21
25
|
export const PROTOCOL_VERSION = "2025-06-18";
|
|
22
26
|
const STRING = { type: "string" };
|
|
23
27
|
export const TOOLS = [
|
|
@@ -74,7 +78,64 @@ export const TOOLS = [
|
|
|
74
78
|
description: "End a watch party. Only its host may.",
|
|
75
79
|
inputSchema: { type: "object", properties: { code: STRING }, required: ["code"] },
|
|
76
80
|
},
|
|
81
|
+
{
|
|
82
|
+
name: "transcribe_audio",
|
|
83
|
+
description: "The words in a recording on this machine, heard by nixamp.com's own open-source ear (Whisper). Any format ffmpeg reads; up to a minute. Given a server, the words are also posted to that server's trollbox as this account.",
|
|
84
|
+
inputSchema: {
|
|
85
|
+
type: "object",
|
|
86
|
+
properties: {
|
|
87
|
+
path: { ...STRING, description: "The recording's path on this machine." },
|
|
88
|
+
language: { ...STRING, description: "A two-letter language code, when Whisper should not guess." },
|
|
89
|
+
server: { ...STRING, description: "Post the words to this nixamp's trollbox: its address, as in its share link." },
|
|
90
|
+
channel: { ...STRING, description: "Which of that server's channels; its own stream (live) by default." },
|
|
91
|
+
},
|
|
92
|
+
required: ["path"],
|
|
93
|
+
},
|
|
94
|
+
},
|
|
95
|
+
{
|
|
96
|
+
name: "trollbox_say",
|
|
97
|
+
description: "Say a line in a live room's trollbox, as this account and under its public handle. A room is a nixamp server's address and one of its channels (or `live`, the server's own stream).",
|
|
98
|
+
inputSchema: {
|
|
99
|
+
type: "object",
|
|
100
|
+
properties: {
|
|
101
|
+
server: { ...STRING, description: "The nixamp server's address, as in its share link." },
|
|
102
|
+
channel: { ...STRING, description: "The channel's id, or live (default)." },
|
|
103
|
+
text: { ...STRING, description: "The line. At most 500 characters." },
|
|
104
|
+
},
|
|
105
|
+
required: ["server", "text"],
|
|
106
|
+
},
|
|
107
|
+
},
|
|
108
|
+
{
|
|
109
|
+
name: "transcript_read",
|
|
110
|
+
description: "What a live channel is saying: the recent lines of its transcript, oldest first, each with when its sound was heard. The server carrying the channel captions it while somebody asks. Pass the server's address and share key, and the channel's id.",
|
|
111
|
+
inputSchema: {
|
|
112
|
+
type: "object",
|
|
113
|
+
properties: {
|
|
114
|
+
url: { ...STRING, description: "The nixamp server's address, e.g. https://server1.chovy.nixamp.com:4321." },
|
|
115
|
+
key: { ...STRING, description: "The share key from its link, when it has one." },
|
|
116
|
+
channel: { ...STRING, description: "The channel's id on that server (default: main)." },
|
|
117
|
+
after: { type: "number", description: "Only lines heard after this moment (ms since the epoch)." },
|
|
118
|
+
},
|
|
119
|
+
required: ["url"],
|
|
120
|
+
},
|
|
121
|
+
},
|
|
122
|
+
{
|
|
123
|
+
name: "trollbox_read",
|
|
124
|
+
description: "The recent lines in a live room's trollbox, oldest first: who said what, and when.",
|
|
125
|
+
inputSchema: {
|
|
126
|
+
type: "object",
|
|
127
|
+
properties: {
|
|
128
|
+
server: { ...STRING, description: "The nixamp server's address, as in its share link." },
|
|
129
|
+
channel: { ...STRING, description: "The channel's id, or live (default)." },
|
|
130
|
+
after: { ...STRING, description: "Only lines after this moment (an ISO timestamp)." },
|
|
131
|
+
},
|
|
132
|
+
required: ["server"],
|
|
133
|
+
},
|
|
134
|
+
},
|
|
77
135
|
];
|
|
136
|
+
function said(line) {
|
|
137
|
+
return `${line.createdAt} ${line.handle}: ${line.body}`;
|
|
138
|
+
}
|
|
78
139
|
function text(value) {
|
|
79
140
|
return { content: [{ type: "text", text: value }] };
|
|
80
141
|
}
|
|
@@ -172,6 +233,77 @@ export async function callTool(name, args, options = {}) {
|
|
|
172
233
|
return failed(await answerOf(response));
|
|
173
234
|
return text(`Ended ${code}.`);
|
|
174
235
|
}
|
|
236
|
+
const server = typeof args["server"] === "string" ? args["server"].trim() : "";
|
|
237
|
+
const channel = typeof args["channel"] === "string" && args["channel"].trim() ? args["channel"].trim() : "live";
|
|
238
|
+
if (name === "transcribe_audio") {
|
|
239
|
+
const path = typeof args["path"] === "string" ? args["path"] : "";
|
|
240
|
+
if (!path)
|
|
241
|
+
return failed("Which recording? Pass its path.");
|
|
242
|
+
let wav;
|
|
243
|
+
try {
|
|
244
|
+
wav = (options.wavOf ?? wavOf)(path);
|
|
245
|
+
}
|
|
246
|
+
catch (error) {
|
|
247
|
+
return failed(error.message);
|
|
248
|
+
}
|
|
249
|
+
const answer = await askToHear(session, {
|
|
250
|
+
wav,
|
|
251
|
+
...(typeof args["language"] === "string" ? { language: args["language"] } : {}),
|
|
252
|
+
...(server ? { server, channel } : {}),
|
|
253
|
+
}, send, site);
|
|
254
|
+
if (!answer.ok)
|
|
255
|
+
return failed(answer.error);
|
|
256
|
+
if (answer.heard.text === "")
|
|
257
|
+
return text("Heard nothing in that recording.");
|
|
258
|
+
return text(answer.heard.message
|
|
259
|
+
? `${answer.heard.text}\n\nSaid in the room for ${channel} at ${server} as ${answer.heard.message.handle}.`
|
|
260
|
+
: answer.heard.text);
|
|
261
|
+
}
|
|
262
|
+
if (name === "trollbox_say") {
|
|
263
|
+
if (!server)
|
|
264
|
+
return failed("Which room? Pass the server's address.");
|
|
265
|
+
const line = typeof args["text"] === "string" ? args["text"] : "";
|
|
266
|
+
if (!line.trim())
|
|
267
|
+
return failed("Say what? Pass the text.");
|
|
268
|
+
const response = await send(`${site}/api/v1/trollbox`, {
|
|
269
|
+
method: "POST",
|
|
270
|
+
headers,
|
|
271
|
+
body: JSON.stringify({ server, channel, body: line }),
|
|
272
|
+
});
|
|
273
|
+
if (!response.ok)
|
|
274
|
+
return failed(await answerOf(response));
|
|
275
|
+
const body = (await response.json());
|
|
276
|
+
return text(body.message ? `Said, as ${body.message.handle}: ${body.message.body}` : "Said.");
|
|
277
|
+
}
|
|
278
|
+
if (name === "transcript_read") {
|
|
279
|
+
const url = typeof args["url"] === "string" ? args["url"].trim() : "";
|
|
280
|
+
if (!url)
|
|
281
|
+
return failed("Which server? Pass its address.");
|
|
282
|
+
const got = await readTranscript({ url, key: typeof args["key"] === "string" && args["key"] ? args["key"] : null }, channel === "live" ? "main" : channel, typeof args["after"] === "number" ? args["after"] : 0, send);
|
|
283
|
+
if (!got.ok)
|
|
284
|
+
return failed(got.error);
|
|
285
|
+
if (got.answer.recent.length === 0) {
|
|
286
|
+
return text(got.answer.error
|
|
287
|
+
? `Nothing yet: ${got.answer.error}`
|
|
288
|
+
: "Nothing said yet. The server has just started listening; ask again in a few seconds.");
|
|
289
|
+
}
|
|
290
|
+
return text(got.answer.recent.map((line) => `${new Date(line.at).toISOString()} ${line.text}`).join("\n"));
|
|
291
|
+
}
|
|
292
|
+
if (name === "trollbox_read") {
|
|
293
|
+
if (!server)
|
|
294
|
+
return failed("Which room? Pass the server's address.");
|
|
295
|
+
const url = new URL(`${site}/api/v1/trollbox`);
|
|
296
|
+
url.searchParams.set("server", server);
|
|
297
|
+
url.searchParams.set("channel", channel);
|
|
298
|
+
if (typeof args["after"] === "string" && args["after"])
|
|
299
|
+
url.searchParams.set("after", args["after"]);
|
|
300
|
+
const response = await send(url.toString(), { headers });
|
|
301
|
+
if (!response.ok)
|
|
302
|
+
return failed(await answerOf(response));
|
|
303
|
+
const body = (await response.json());
|
|
304
|
+
const lines = body.messages ?? [];
|
|
305
|
+
return text(lines.length === 0 ? `Nobody has said anything in the room for ${channel} at ${server}.` : lines.map(said).join("\n"));
|
|
306
|
+
}
|
|
175
307
|
}
|
|
176
308
|
catch (error) {
|
|
177
309
|
return failed(`Could not reach ${site}: ${error.message}`);
|
|
@@ -186,7 +318,7 @@ export async function handleMessage(message, options = {}) {
|
|
|
186
318
|
return reply({
|
|
187
319
|
protocolVersion: PROTOCOL_VERSION,
|
|
188
320
|
capabilities: { tools: { listChanged: false } },
|
|
189
|
-
serverInfo: { name: "nixamp", title: "nixamp watch parties", version: "1" },
|
|
321
|
+
serverInfo: { name: "nixamp", title: "nixamp: watch parties and rooms", version: "1" },
|
|
190
322
|
instructions: "Watch parties on nixamp. A party lives on the site hosting the film and is bridged here as a room every nixamp client can join. Codes are the ones that site shows; positions are seconds into the film.",
|
|
191
323
|
});
|
|
192
324
|
}
|
package/dist/server.d.ts
CHANGED
|
@@ -28,6 +28,8 @@ import { Tickets } from "./tickets.ts";
|
|
|
28
28
|
import { Layouts } from "./layouts.ts";
|
|
29
29
|
import { Rooms } from "./rooms.ts";
|
|
30
30
|
import { Trollbox } from "./trollbox.ts";
|
|
31
|
+
import { Speech } from "./speech.ts";
|
|
32
|
+
import { Captions } from "./captions.ts";
|
|
31
33
|
import { type Codecs } from "./audio.ts";
|
|
32
34
|
import { type Tools, type Track } from "./audio.ts";
|
|
33
35
|
import { type Command, type RemoteTrack, type Snapshot } from "./protocol.ts";
|
|
@@ -511,6 +513,8 @@ export interface HandlerOptions {
|
|
|
511
513
|
ytdlp?: string[] | null;
|
|
512
514
|
/** Channels as HLS, for Safari on a phone, which plays a live stream no other way. */
|
|
513
515
|
hls?: HlsPackagers;
|
|
516
|
+
/** Captions for a channel: its sound, as lines, as they are heard. Needs ffmpeg and a sign-in. */
|
|
517
|
+
captions?: Captions;
|
|
514
518
|
/** Lossless relay compression, its policies, diagnostics and static representations. */
|
|
515
519
|
compression?: CompressionService;
|
|
516
520
|
/** What a name is -- a film, a channel, a fixture -- asked of nichedb.dev and remembered. */
|
|
@@ -658,6 +662,8 @@ export interface HandlerOptions {
|
|
|
658
662
|
rooms?: Rooms;
|
|
659
663
|
/** The trollbox: a chat per live room, kept at nixamp.com. */
|
|
660
664
|
trollbox?: Trollbox;
|
|
665
|
+
/** Speech to text: a line said out loud, heard here. Needs the optional model. */
|
|
666
|
+
speech?: Speech;
|
|
661
667
|
/** Tickets: a paid pass to one event's room. Absent means every show is free. */
|
|
662
668
|
tickets?: Tickets;
|
|
663
669
|
/**
|
package/dist/server.js
CHANGED
|
@@ -19,7 +19,7 @@ import { readFileSync } from "node:fs";
|
|
|
19
19
|
import { Connections } from "./connections.js";
|
|
20
20
|
import { Broadcaster, DEFAULT_ENCODER, PRESETS, redact, } from "./broadcast.js";
|
|
21
21
|
import { Ingest, normaliseFormat } from "./ingest.js";
|
|
22
|
-
import { Channels, cleanId, generatedId, rememberChannels, rememberedChannels, rememberedNow, } from "./channels.js";
|
|
22
|
+
import { BACKLOG_SECONDS, Channels, cleanId, generatedId, rememberChannels, rememberedChannels, rememberedNow, } from "./channels.js";
|
|
23
23
|
import { RtmpListeners } from "./rtmp-in.js";
|
|
24
24
|
import { Accounts, clearedCookie, sessionCookie, tokenFrom } from "./accounts.js";
|
|
25
25
|
import { anonymousHandle, Handles } from "./handles.js";
|
|
@@ -63,6 +63,8 @@ import { Tickets, needsTicket, ticketFrom, ticketsFromEnv } from "./tickets.js";
|
|
|
63
63
|
import { Layouts } from "./layouts.js";
|
|
64
64
|
import { Rooms } from "./rooms.js";
|
|
65
65
|
import { Trollbox, TrollboxError, fallbackHandle, roomFor } from "./trollbox.js";
|
|
66
|
+
import { MAX_BYTES as SPEECH_BYTES, Speech, SpeechError, isWav, languageOf } from "./speech.js";
|
|
67
|
+
import { Captions } from "./captions.js";
|
|
66
68
|
import { confirm, DEFAULT_DIRECTORY, Publisher } from "./publish.js";
|
|
67
69
|
import { applyRemoteConfig, createPaywall, FREE_LISTENERS, paywallFromEnv, } from "./paywall.js";
|
|
68
70
|
import { isRemote, isTransportStream, playsInBrowser, sourceLabel } from "./sources.js";
|
|
@@ -1182,6 +1184,19 @@ async function readBody(request, limit = 64 * 1024) {
|
|
|
1182
1184
|
}
|
|
1183
1185
|
return Buffer.concat(chunks).toString("utf8");
|
|
1184
1186
|
}
|
|
1187
|
+
/** The body as bytes: sound, not text. */
|
|
1188
|
+
async function readBytes(request, limit) {
|
|
1189
|
+
const chunks = [];
|
|
1190
|
+
let size = 0;
|
|
1191
|
+
for await (const chunk of request) {
|
|
1192
|
+
const buffer = chunk;
|
|
1193
|
+
size += buffer.length;
|
|
1194
|
+
if (size > limit)
|
|
1195
|
+
throw new SpeechError("that is too much sound", 413);
|
|
1196
|
+
chunks.push(buffer);
|
|
1197
|
+
}
|
|
1198
|
+
return new Uint8Array(Buffer.concat(chunks));
|
|
1199
|
+
}
|
|
1185
1200
|
/**
|
|
1186
1201
|
* The whole HTTP surface, as a plain function of a request — so a test can
|
|
1187
1202
|
* drive it with a real socket and no ffmpeg in sight.
|
|
@@ -2127,6 +2142,56 @@ export function createHandler(engine, options) {
|
|
|
2127
2142
|
}
|
|
2128
2143
|
return;
|
|
2129
2144
|
}
|
|
2145
|
+
/*
|
|
2146
|
+
* Speech to text: a WAV in, the words out. Signed in only -- the ear
|
|
2147
|
+
* costs CPU and a trollbox line needs a name anyway -- and, given a
|
|
2148
|
+
* room, the words go straight into that room's trollbox as a line by
|
|
2149
|
+
* whoever spoke them. The page, the CLI and the MCP tools all come here.
|
|
2150
|
+
*/
|
|
2151
|
+
if (path === "/api/v1/speech/transcribe" && options.speech && options.accounts) {
|
|
2152
|
+
const speech = options.speech;
|
|
2153
|
+
try {
|
|
2154
|
+
if (request.method !== "POST") {
|
|
2155
|
+
json(response, 405, { error: "POST a WAV" });
|
|
2156
|
+
return;
|
|
2157
|
+
}
|
|
2158
|
+
const who = await options.accounts.whoIs(tokenFrom(request.headers));
|
|
2159
|
+
if (who === null) {
|
|
2160
|
+
json(response, 401, { error: "sign in to nixamp.com to dictate" });
|
|
2161
|
+
return;
|
|
2162
|
+
}
|
|
2163
|
+
const bytes = await readBytes(request, SPEECH_BYTES);
|
|
2164
|
+
if (!isWav(bytes)) {
|
|
2165
|
+
json(response, 415, { error: "send a WAV: 16-bit PCM, mono, 16 kHz is ideal" });
|
|
2166
|
+
return;
|
|
2167
|
+
}
|
|
2168
|
+
// A room named is a line posted, by the same rules as typing it.
|
|
2169
|
+
const roomAsked = url.searchParams.has("server") || url.searchParams.has("channel");
|
|
2170
|
+
const where = roomAsked ? roomFor(url.searchParams.get("server"), url.searchParams.get("channel")) : null;
|
|
2171
|
+
if (roomAsked && !where) {
|
|
2172
|
+
json(response, 400, { error: "a room is a server address and a channel" });
|
|
2173
|
+
return;
|
|
2174
|
+
}
|
|
2175
|
+
const heard = await speech.transcribe(bytes, { language: languageOf(url.searchParams.get("language")), by: who.id });
|
|
2176
|
+
if (where && options.trollbox && heard.text !== "") {
|
|
2177
|
+
const handle = (await options.handles?.of(who.id)) || fallbackHandle(who.id);
|
|
2178
|
+
const line = await options.trollbox.post(where, who.id, handle, heard.text);
|
|
2179
|
+
json(response, 201, {
|
|
2180
|
+
text: heard.text, seconds: heard.seconds, model: speech.model,
|
|
2181
|
+
message: { id: line.id, handle: line.handle, body: line.body, createdAt: line.createdAt, mine: true },
|
|
2182
|
+
});
|
|
2183
|
+
return;
|
|
2184
|
+
}
|
|
2185
|
+
json(response, 200, { text: heard.text, seconds: heard.seconds, model: speech.model });
|
|
2186
|
+
}
|
|
2187
|
+
catch (error) {
|
|
2188
|
+
if (error instanceof SpeechError || error instanceof TrollboxError)
|
|
2189
|
+
json(response, error.status, { error: error.message });
|
|
2190
|
+
else
|
|
2191
|
+
throw error;
|
|
2192
|
+
}
|
|
2193
|
+
return;
|
|
2194
|
+
}
|
|
2130
2195
|
if (path === "/api/v1/me/handle" && options.handles && options.accounts) {
|
|
2131
2196
|
const handles = options.handles;
|
|
2132
2197
|
const who = await options.accounts.whoIs(tokenFrom(request.headers));
|
|
@@ -3182,6 +3247,67 @@ export function createHandler(engine, options) {
|
|
|
3182
3247
|
createReadStream(segment).pipe(response);
|
|
3183
3248
|
return;
|
|
3184
3249
|
}
|
|
3250
|
+
/*
|
|
3251
|
+
* The channel's captions: what it is saying, as lines, as they are
|
|
3252
|
+
* heard by nixamp.com's ear on this server's behalf. Read the way the
|
|
3253
|
+
* sound is read -- the key gate above has already passed -- and
|
|
3254
|
+
* started by the first person asking. `captions` is the live stream
|
|
3255
|
+
* of lines; `transcript` is the recent ones as JSON, for a poll.
|
|
3256
|
+
*/
|
|
3257
|
+
if ((action === "captions" || action === "transcript") && request.method === "GET") {
|
|
3258
|
+
const captions = options.captions;
|
|
3259
|
+
if (!captions) {
|
|
3260
|
+
json(response, 503, { error: "this server cannot caption: it has no ffmpeg" });
|
|
3261
|
+
return;
|
|
3262
|
+
}
|
|
3263
|
+
if (!captions.available()) {
|
|
3264
|
+
json(response, 503, { error: "this server is not signed in to nixamp.com, so it cannot caption; run `nixamp login` on it" });
|
|
3265
|
+
return;
|
|
3266
|
+
}
|
|
3267
|
+
if (!channels.has(id)) {
|
|
3268
|
+
json(response, 404, { error: "nothing is playing on that channel" });
|
|
3269
|
+
return;
|
|
3270
|
+
}
|
|
3271
|
+
// Lines after a moment: a poll's ?after=, or the last line a
|
|
3272
|
+
// reconnecting EventSource saw, which the browser sends by itself.
|
|
3273
|
+
const lastSeen = request.headers["last-event-id"];
|
|
3274
|
+
const after = Number(url.searchParams.get("after") ?? (Array.isArray(lastSeen) ? lastSeen[0] : lastSeen) ?? 0) || 0;
|
|
3275
|
+
if (action === "transcript") {
|
|
3276
|
+
// Asking keeps the captioner up: it stops a minute after the last ask.
|
|
3277
|
+
captions.subscribe(id, () => undefined)?.();
|
|
3278
|
+
json(response, 200, {
|
|
3279
|
+
channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), ...captions.status(id), lines: captions.recent(id, after),
|
|
3280
|
+
});
|
|
3281
|
+
return;
|
|
3282
|
+
}
|
|
3283
|
+
response.writeHead(200, {
|
|
3284
|
+
...CORS,
|
|
3285
|
+
"content-type": "text/event-stream; charset=utf-8",
|
|
3286
|
+
"cache-control": "no-store",
|
|
3287
|
+
connection: "keep-alive",
|
|
3288
|
+
"x-accel-buffering": "no",
|
|
3289
|
+
});
|
|
3290
|
+
const write = (event, data, eventId) => {
|
|
3291
|
+
response.write(`${eventId === undefined ? "" : `id: ${eventId}\n`}event: ${event}\ndata: ${JSON.stringify(data)}\n\n`);
|
|
3292
|
+
};
|
|
3293
|
+
const off = captions.subscribe(id, (line) => write("line", line, line.at));
|
|
3294
|
+
if (off === null) {
|
|
3295
|
+
response.end();
|
|
3296
|
+
return;
|
|
3297
|
+
}
|
|
3298
|
+
// How far behind the live edge a newcomer's playback starts, so the
|
|
3299
|
+
// page can hold each line until its own sound gets there.
|
|
3300
|
+
write("hello", { channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), ...captions.status(id), lines: captions.recent(id, after) });
|
|
3301
|
+
const beat = setInterval(() => response.write(": beat\n\n"), 20_000);
|
|
3302
|
+
beat.unref?.();
|
|
3303
|
+
const done = () => {
|
|
3304
|
+
clearInterval(beat);
|
|
3305
|
+
off();
|
|
3306
|
+
};
|
|
3307
|
+
request.on("close", done);
|
|
3308
|
+
response.on("close", done);
|
|
3309
|
+
return;
|
|
3310
|
+
}
|
|
3185
3311
|
if (action === undefined && request.method === "GET") {
|
|
3186
3312
|
// Listening. The response is the fan-out target: whatever ffmpeg
|
|
3187
3313
|
// produces for this channel is written to it until one end goes away.
|
|
@@ -4450,6 +4576,16 @@ export async function serve(argv, version = "0.1.0") {
|
|
|
4450
4576
|
secret: process.env["NIXAMP_JWT_SECRET"] ?? "",
|
|
4451
4577
|
})
|
|
4452
4578
|
: undefined;
|
|
4579
|
+
// The ear lives where the accounts live. Loaded now, in the background, so
|
|
4580
|
+
// the first person to speak does not wait for the model to arrive; a box
|
|
4581
|
+
// without the optional model never says it is ready, and answers 503.
|
|
4582
|
+
const speech = accounts && process.env["NIXAMP_STT"] !== "off" ? new Speech() : undefined;
|
|
4583
|
+
if (speech) {
|
|
4584
|
+
void speech.warm().then((ready) => {
|
|
4585
|
+
if (ready)
|
|
4586
|
+
console.error(`nixamp: hearing with ${speech.model}`);
|
|
4587
|
+
});
|
|
4588
|
+
}
|
|
4453
4589
|
// nixamp as an OAuth 2.1 authorization server, and the watch parties a
|
|
4454
4590
|
// client site bridges through it. Both need the same three things -- a
|
|
4455
4591
|
// database, accounts, and a site to be the issuer of -- so both appear or
|
|
@@ -4627,6 +4763,18 @@ export async function serve(argv, version = "0.1.0") {
|
|
|
4627
4763
|
void durable.sweep(Date.now() - ENDED_TTL_MS, new Date(Date.now() - 30 * 24 * 60 * 60 * 1000));
|
|
4628
4764
|
}
|
|
4629
4765
|
const session = readSession();
|
|
4766
|
+
// Captions for a channel: this server's ffmpeg turns the sound into PCM
|
|
4767
|
+
// and nixamp.com's ear, asked as this server, turns that into lines. A
|
|
4768
|
+
// server without ffmpeg cannot; a server nobody signed in on is refused
|
|
4769
|
+
// by the ear, and says so to whoever asks.
|
|
4770
|
+
const captions = tools.carries
|
|
4771
|
+
? new Captions({
|
|
4772
|
+
ffmpeg: tools.ffmpeg,
|
|
4773
|
+
listen: (id, listener) => channels.listen(id, listener),
|
|
4774
|
+
session: () => readSession(),
|
|
4775
|
+
onEvent: (message) => console.log(` ${message}`),
|
|
4776
|
+
})
|
|
4777
|
+
: undefined;
|
|
4630
4778
|
const owner = new Owner({
|
|
4631
4779
|
ownerId: options.owner || (session?.token ? await ownerIdOf(session) : ""),
|
|
4632
4780
|
site: session?.site ?? DEFAULT_DIRECTORY,
|
|
@@ -4810,6 +4958,7 @@ export async function serve(argv, version = "0.1.0") {
|
|
|
4810
4958
|
carries: tools.carries !== false,
|
|
4811
4959
|
cookies: cookiesFile(),
|
|
4812
4960
|
hls,
|
|
4961
|
+
...(captions ? { captions } : {}),
|
|
4813
4962
|
compression,
|
|
4814
4963
|
enricher,
|
|
4815
4964
|
...(tls ? { tls } : {}),
|
|
@@ -4824,6 +4973,7 @@ export async function serve(argv, version = "0.1.0") {
|
|
|
4824
4973
|
...(layouts ? { layouts } : {}),
|
|
4825
4974
|
...(rooms ? { rooms } : {}),
|
|
4826
4975
|
...(trollbox ? { trollbox } : {}),
|
|
4976
|
+
...(speech ? { speech } : {}),
|
|
4827
4977
|
...(tickets ? { tickets } : {}),
|
|
4828
4978
|
...(authServer ? { authServer } : {}),
|
|
4829
4979
|
...(parties ? { parties } : {}),
|
package/dist/speech.d.ts
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
/** What Whisper listens at. Everything is brought to this before it is heard. */
|
|
2
|
+
export declare const RATE = 16000;
|
|
3
|
+
/** A trollbox line, said out loud, is seconds long; a minute is the ceiling. */
|
|
4
|
+
export declare const MAX_SECONDS = 60;
|
|
5
|
+
/** A minute of 16-bit mono at 16 kHz is under 2 MB; 48 kHz stereo is under 12. */
|
|
6
|
+
export declare const MAX_BYTES: number;
|
|
7
|
+
/**
|
|
8
|
+
* How much one account may have heard in a minute, in seconds of sound.
|
|
9
|
+
* Counted in sound rather than asks because a live channel being captioned
|
|
10
|
+
* asks twelve times a minute for five seconds each, and a person dictating
|
|
11
|
+
* asks twice for thirty: the cost is the sound, not the call. Five minutes
|
|
12
|
+
* of sound a minute is four channels captioned, or a conversation, and
|
|
13
|
+
* not a firehose.
|
|
14
|
+
*/
|
|
15
|
+
export declare const SECONDS_PER_MINUTE = 300;
|
|
16
|
+
/** How many may wait for the one CPU. Past this, the honest answer is "later". */
|
|
17
|
+
export declare const QUEUE_LIMIT = 8;
|
|
18
|
+
export declare const DEFAULT_MODEL = "onnx-community/whisper-base";
|
|
19
|
+
export declare class SpeechError extends Error {
|
|
20
|
+
readonly status: number;
|
|
21
|
+
constructor(message: string, status: number);
|
|
22
|
+
}
|
|
23
|
+
export interface Heard {
|
|
24
|
+
text: string;
|
|
25
|
+
/** How long the audio was, in seconds. */
|
|
26
|
+
seconds: number;
|
|
27
|
+
}
|
|
28
|
+
/** Mono samples in [-1, 1] at RATE -> words. What the loaded model is, to this file. */
|
|
29
|
+
export type Recognizer = (pcm: Float32Array, options: {
|
|
30
|
+
language?: string;
|
|
31
|
+
}) => Promise<{
|
|
32
|
+
text: string;
|
|
33
|
+
}>;
|
|
34
|
+
export interface SpeechOptions {
|
|
35
|
+
/** A Hugging Face model id; NIXAMP_STT_MODEL otherwise; whisper-base by default. */
|
|
36
|
+
model?: string;
|
|
37
|
+
/** Where the model's files are kept between runs. */
|
|
38
|
+
cacheDir?: string;
|
|
39
|
+
/** How the model is loaded. The tests hand in a fake; nothing else does. */
|
|
40
|
+
load?: (model: string, cacheDir: string) => Promise<Recognizer>;
|
|
41
|
+
now?: () => number;
|
|
42
|
+
}
|
|
43
|
+
export interface Wav {
|
|
44
|
+
rate: number;
|
|
45
|
+
channels: number;
|
|
46
|
+
/** Mixed down to one channel, in [-1, 1]. */
|
|
47
|
+
samples: Float32Array;
|
|
48
|
+
}
|
|
49
|
+
/** Whether these bytes are a RIFF/WAVE file, whatever the request called them. */
|
|
50
|
+
export declare function isWav(bytes: Uint8Array): boolean;
|
|
51
|
+
/**
|
|
52
|
+
* A WAV file's samples, mixed to mono. PCM of 8, 16, 24 or 32 bits, or
|
|
53
|
+
* 32-bit float; the WAVE_FORMAT_EXTENSIBLE wrapper around either. That is
|
|
54
|
+
* what every encoder that matters writes, including the one in the page.
|
|
55
|
+
*/
|
|
56
|
+
export declare function decodeWav(bytes: Uint8Array): Wav;
|
|
57
|
+
/**
|
|
58
|
+
* Samples at one rate, at another. Down is an average over each output
|
|
59
|
+
* sample's span, which is a crude low-pass and enough for speech; up is a
|
|
60
|
+
* straight line between neighbours. Whisper resamples nothing itself.
|
|
61
|
+
*/
|
|
62
|
+
export declare function resample(samples: Float32Array, from: number, to: number): Float32Array;
|
|
63
|
+
/** 16-bit mono PCM WAV bytes from samples: what the CLI's tests, and anybody, can send. */
|
|
64
|
+
export declare function encodeWav(samples: Float32Array, rate?: number): Uint8Array;
|
|
65
|
+
/** Whisper's own spacing and blank-audio tokens, tidied into a line. */
|
|
66
|
+
export declare function tidy(text: string): string;
|
|
67
|
+
/** A two-letter language code, or nothing: Whisper guesses when not told. */
|
|
68
|
+
export declare function languageOf(value: unknown): string | undefined;
|
|
69
|
+
export declare class Speech {
|
|
70
|
+
readonly model: string;
|
|
71
|
+
private readonly cacheDir;
|
|
72
|
+
private readonly load;
|
|
73
|
+
private readonly now;
|
|
74
|
+
private recognizer;
|
|
75
|
+
/** One at a time: the model is CPU-bound, and two at once is slower than two in turn. */
|
|
76
|
+
private tail;
|
|
77
|
+
private waiting;
|
|
78
|
+
private readonly asked;
|
|
79
|
+
constructor(options?: SpeechOptions);
|
|
80
|
+
private ear;
|
|
81
|
+
/**
|
|
82
|
+
* Load the model now, so the first person to speak is not the one who
|
|
83
|
+
* waits for the download. Says whether it could; never throws.
|
|
84
|
+
*/
|
|
85
|
+
warm(): Promise<boolean>;
|
|
86
|
+
/** Whether this account may have this much heard now, and the bookkeeping if so. */
|
|
87
|
+
allow(accountId: string, seconds: number): void;
|
|
88
|
+
/**
|
|
89
|
+
* The words in a WAV. Refuses what is not a WAV, or is too long, or more
|
|
90
|
+
* than the account may have heard this minute, with a status.
|
|
91
|
+
*/
|
|
92
|
+
transcribe(bytes: Uint8Array, options?: {
|
|
93
|
+
language?: string;
|
|
94
|
+
by?: string;
|
|
95
|
+
}): Promise<Heard>;
|
|
96
|
+
}
|