nixamp 0.21.4 → 0.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/mcp.js CHANGED
@@ -8,7 +8,9 @@
8
8
  *
9
9
  * What it offers is the watch party, because that is the part of nixamp an
10
10
  * agent can usefully do something with: find the party, say where it is, put
11
- * one on the air, move everybody to the same second. It signs in as whoever
11
+ * one on the air, move everybody to the same second. And the room: hear a
12
+ * recording (nixamp.com's own ear, see speech.ts), say a line in a trollbox,
13
+ * read one back. It signs in as whoever
12
14
  * this machine is signed in as -- the session on disk, or NIXAMP_TOKEN --
13
15
  * because an agent holding its own credential is a credential nobody revokes.
14
16
  *
@@ -18,6 +20,8 @@
18
20
  import { createInterface } from "node:readline";
19
21
  import { clock } from "./party.js";
20
22
  import { readSession } from "./session.js";
23
+ import { askToHear, wavOf } from "./transcribe.js";
24
+ import { readTranscript } from "./transcript.js";
21
25
  export const PROTOCOL_VERSION = "2025-06-18";
22
26
  const STRING = { type: "string" };
23
27
  export const TOOLS = [
@@ -74,7 +78,64 @@ export const TOOLS = [
74
78
  description: "End a watch party. Only its host may.",
75
79
  inputSchema: { type: "object", properties: { code: STRING }, required: ["code"] },
76
80
  },
81
+ {
82
+ name: "transcribe_audio",
83
+ description: "The words in a recording on this machine, heard by nixamp.com's own open-source ear (Whisper). Any format ffmpeg reads; up to a minute. Given a server, the words are also posted to that server's trollbox as this account.",
84
+ inputSchema: {
85
+ type: "object",
86
+ properties: {
87
+ path: { ...STRING, description: "The recording's path on this machine." },
88
+ language: { ...STRING, description: "A two-letter language code, when Whisper should not guess." },
89
+ server: { ...STRING, description: "Post the words to this nixamp's trollbox: its address, as in its share link." },
90
+ channel: { ...STRING, description: "Which of that server's channels; its own stream (live) by default." },
91
+ },
92
+ required: ["path"],
93
+ },
94
+ },
95
+ {
96
+ name: "trollbox_say",
97
+ description: "Say a line in a live room's trollbox, as this account and under its public handle. A room is a nixamp server's address and one of its channels (or `live`, the server's own stream).",
98
+ inputSchema: {
99
+ type: "object",
100
+ properties: {
101
+ server: { ...STRING, description: "The nixamp server's address, as in its share link." },
102
+ channel: { ...STRING, description: "The channel's id, or live (default)." },
103
+ text: { ...STRING, description: "The line. At most 500 characters." },
104
+ },
105
+ required: ["server", "text"],
106
+ },
107
+ },
108
+ {
109
+ name: "transcript_read",
110
+ description: "What a live channel is saying: the recent lines of its transcript, oldest first, each with when its sound was heard. The server carrying the channel captions it while somebody asks. Pass the server's address and share key, and the channel's id.",
111
+ inputSchema: {
112
+ type: "object",
113
+ properties: {
114
+ url: { ...STRING, description: "The nixamp server's address, e.g. https://server1.chovy.nixamp.com:4321." },
115
+ key: { ...STRING, description: "The share key from its link, when it has one." },
116
+ channel: { ...STRING, description: "The channel's id on that server (default: main)." },
117
+ after: { type: "number", description: "Only lines heard after this moment (ms since the epoch)." },
118
+ },
119
+ required: ["url"],
120
+ },
121
+ },
122
+ {
123
+ name: "trollbox_read",
124
+ description: "The recent lines in a live room's trollbox, oldest first: who said what, and when.",
125
+ inputSchema: {
126
+ type: "object",
127
+ properties: {
128
+ server: { ...STRING, description: "The nixamp server's address, as in its share link." },
129
+ channel: { ...STRING, description: "The channel's id, or live (default)." },
130
+ after: { ...STRING, description: "Only lines after this moment (an ISO timestamp)." },
131
+ },
132
+ required: ["server"],
133
+ },
134
+ },
77
135
  ];
136
+ function said(line) {
137
+ return `${line.createdAt} ${line.handle}: ${line.body}`;
138
+ }
78
139
  function text(value) {
79
140
  return { content: [{ type: "text", text: value }] };
80
141
  }
@@ -172,6 +233,77 @@ export async function callTool(name, args, options = {}) {
172
233
  return failed(await answerOf(response));
173
234
  return text(`Ended ${code}.`);
174
235
  }
236
+ const server = typeof args["server"] === "string" ? args["server"].trim() : "";
237
+ const channel = typeof args["channel"] === "string" && args["channel"].trim() ? args["channel"].trim() : "live";
238
+ if (name === "transcribe_audio") {
239
+ const path = typeof args["path"] === "string" ? args["path"] : "";
240
+ if (!path)
241
+ return failed("Which recording? Pass its path.");
242
+ let wav;
243
+ try {
244
+ wav = (options.wavOf ?? wavOf)(path);
245
+ }
246
+ catch (error) {
247
+ return failed(error.message);
248
+ }
249
+ const answer = await askToHear(session, {
250
+ wav,
251
+ ...(typeof args["language"] === "string" ? { language: args["language"] } : {}),
252
+ ...(server ? { server, channel } : {}),
253
+ }, send, site);
254
+ if (!answer.ok)
255
+ return failed(answer.error);
256
+ if (answer.heard.text === "")
257
+ return text("Heard nothing in that recording.");
258
+ return text(answer.heard.message
259
+ ? `${answer.heard.text}\n\nSaid in the room for ${channel} at ${server} as ${answer.heard.message.handle}.`
260
+ : answer.heard.text);
261
+ }
262
+ if (name === "trollbox_say") {
263
+ if (!server)
264
+ return failed("Which room? Pass the server's address.");
265
+ const line = typeof args["text"] === "string" ? args["text"] : "";
266
+ if (!line.trim())
267
+ return failed("Say what? Pass the text.");
268
+ const response = await send(`${site}/api/v1/trollbox`, {
269
+ method: "POST",
270
+ headers,
271
+ body: JSON.stringify({ server, channel, body: line }),
272
+ });
273
+ if (!response.ok)
274
+ return failed(await answerOf(response));
275
+ const body = (await response.json());
276
+ return text(body.message ? `Said, as ${body.message.handle}: ${body.message.body}` : "Said.");
277
+ }
278
+ if (name === "transcript_read") {
279
+ const url = typeof args["url"] === "string" ? args["url"].trim() : "";
280
+ if (!url)
281
+ return failed("Which server? Pass its address.");
282
+ const got = await readTranscript({ url, key: typeof args["key"] === "string" && args["key"] ? args["key"] : null }, channel === "live" ? "main" : channel, typeof args["after"] === "number" ? args["after"] : 0, send);
283
+ if (!got.ok)
284
+ return failed(got.error);
285
+ if (got.answer.recent.length === 0) {
286
+ return text(got.answer.error
287
+ ? `Nothing yet: ${got.answer.error}`
288
+ : "Nothing said yet. The server has just started listening; ask again in a few seconds.");
289
+ }
290
+ return text(got.answer.recent.map((line) => `${new Date(line.at).toISOString()} ${line.text}`).join("\n"));
291
+ }
292
+ if (name === "trollbox_read") {
293
+ if (!server)
294
+ return failed("Which room? Pass the server's address.");
295
+ const url = new URL(`${site}/api/v1/trollbox`);
296
+ url.searchParams.set("server", server);
297
+ url.searchParams.set("channel", channel);
298
+ if (typeof args["after"] === "string" && args["after"])
299
+ url.searchParams.set("after", args["after"]);
300
+ const response = await send(url.toString(), { headers });
301
+ if (!response.ok)
302
+ return failed(await answerOf(response));
303
+ const body = (await response.json());
304
+ const lines = body.messages ?? [];
305
+ return text(lines.length === 0 ? `Nobody has said anything in the room for ${channel} at ${server}.` : lines.map(said).join("\n"));
306
+ }
175
307
  }
176
308
  catch (error) {
177
309
  return failed(`Could not reach ${site}: ${error.message}`);
@@ -186,7 +318,7 @@ export async function handleMessage(message, options = {}) {
186
318
  return reply({
187
319
  protocolVersion: PROTOCOL_VERSION,
188
320
  capabilities: { tools: { listChanged: false } },
189
- serverInfo: { name: "nixamp", title: "nixamp watch parties", version: "1" },
321
+ serverInfo: { name: "nixamp", title: "nixamp: watch parties and rooms", version: "1" },
190
322
  instructions: "Watch parties on nixamp. A party lives on the site hosting the film and is bridged here as a room every nixamp client can join. Codes are the ones that site shows; positions are seconds into the film.",
191
323
  });
192
324
  }
package/dist/server.d.ts CHANGED
@@ -28,6 +28,8 @@ import { Tickets } from "./tickets.ts";
28
28
  import { Layouts } from "./layouts.ts";
29
29
  import { Rooms } from "./rooms.ts";
30
30
  import { Trollbox } from "./trollbox.ts";
31
+ import { Speech } from "./speech.ts";
32
+ import { Captions } from "./captions.ts";
31
33
  import { type Codecs } from "./audio.ts";
32
34
  import { type Tools, type Track } from "./audio.ts";
33
35
  import { type Command, type RemoteTrack, type Snapshot } from "./protocol.ts";
@@ -511,6 +513,8 @@ export interface HandlerOptions {
511
513
  ytdlp?: string[] | null;
512
514
  /** Channels as HLS, for Safari on a phone, which plays a live stream no other way. */
513
515
  hls?: HlsPackagers;
516
+ /** Captions for a channel: its sound, as lines, as they are heard. Needs ffmpeg and a sign-in. */
517
+ captions?: Captions;
514
518
  /** Lossless relay compression, its policies, diagnostics and static representations. */
515
519
  compression?: CompressionService;
516
520
  /** What a name is -- a film, a channel, a fixture -- asked of nichedb.dev and remembered. */
@@ -658,6 +662,8 @@ export interface HandlerOptions {
658
662
  rooms?: Rooms;
659
663
  /** The trollbox: a chat per live room, kept at nixamp.com. */
660
664
  trollbox?: Trollbox;
665
+ /** Speech to text: a line said out loud, heard here. Needs the optional model. */
666
+ speech?: Speech;
661
667
  /** Tickets: a paid pass to one event's room. Absent means every show is free. */
662
668
  tickets?: Tickets;
663
669
  /**
package/dist/server.js CHANGED
@@ -19,7 +19,7 @@ import { readFileSync } from "node:fs";
19
19
  import { Connections } from "./connections.js";
20
20
  import { Broadcaster, DEFAULT_ENCODER, PRESETS, redact, } from "./broadcast.js";
21
21
  import { Ingest, normaliseFormat } from "./ingest.js";
22
- import { Channels, cleanId, generatedId, rememberChannels, rememberedChannels, rememberedNow, } from "./channels.js";
22
+ import { BACKLOG_SECONDS, Channels, cleanId, generatedId, rememberChannels, rememberedChannels, rememberedNow, } from "./channels.js";
23
23
  import { RtmpListeners } from "./rtmp-in.js";
24
24
  import { Accounts, clearedCookie, sessionCookie, tokenFrom } from "./accounts.js";
25
25
  import { anonymousHandle, Handles } from "./handles.js";
@@ -63,6 +63,8 @@ import { Tickets, needsTicket, ticketFrom, ticketsFromEnv } from "./tickets.js";
63
63
  import { Layouts } from "./layouts.js";
64
64
  import { Rooms } from "./rooms.js";
65
65
  import { Trollbox, TrollboxError, fallbackHandle, roomFor } from "./trollbox.js";
66
+ import { MAX_BYTES as SPEECH_BYTES, Speech, SpeechError, isWav, languageOf } from "./speech.js";
67
+ import { Captions } from "./captions.js";
66
68
  import { confirm, DEFAULT_DIRECTORY, Publisher } from "./publish.js";
67
69
  import { applyRemoteConfig, createPaywall, FREE_LISTENERS, paywallFromEnv, } from "./paywall.js";
68
70
  import { isRemote, isTransportStream, playsInBrowser, sourceLabel } from "./sources.js";
@@ -1182,6 +1184,19 @@ async function readBody(request, limit = 64 * 1024) {
1182
1184
  }
1183
1185
  return Buffer.concat(chunks).toString("utf8");
1184
1186
  }
1187
+ /** The body as bytes: sound, not text. */
1188
+ async function readBytes(request, limit) {
1189
+ const chunks = [];
1190
+ let size = 0;
1191
+ for await (const chunk of request) {
1192
+ const buffer = chunk;
1193
+ size += buffer.length;
1194
+ if (size > limit)
1195
+ throw new SpeechError("that is too much sound", 413);
1196
+ chunks.push(buffer);
1197
+ }
1198
+ return new Uint8Array(Buffer.concat(chunks));
1199
+ }
1185
1200
  /**
1186
1201
  * The whole HTTP surface, as a plain function of a request — so a test can
1187
1202
  * drive it with a real socket and no ffmpeg in sight.
@@ -2127,6 +2142,56 @@ export function createHandler(engine, options) {
2127
2142
  }
2128
2143
  return;
2129
2144
  }
2145
+ /*
2146
+ * Speech to text: a WAV in, the words out. Signed in only -- the ear
2147
+ * costs CPU and a trollbox line needs a name anyway -- and, given a
2148
+ * room, the words go straight into that room's trollbox as a line by
2149
+ * whoever spoke them. The page, the CLI and the MCP tools all come here.
2150
+ */
2151
+ if (path === "/api/v1/speech/transcribe" && options.speech && options.accounts) {
2152
+ const speech = options.speech;
2153
+ try {
2154
+ if (request.method !== "POST") {
2155
+ json(response, 405, { error: "POST a WAV" });
2156
+ return;
2157
+ }
2158
+ const who = await options.accounts.whoIs(tokenFrom(request.headers));
2159
+ if (who === null) {
2160
+ json(response, 401, { error: "sign in to nixamp.com to dictate" });
2161
+ return;
2162
+ }
2163
+ const bytes = await readBytes(request, SPEECH_BYTES);
2164
+ if (!isWav(bytes)) {
2165
+ json(response, 415, { error: "send a WAV: 16-bit PCM, mono, 16 kHz is ideal" });
2166
+ return;
2167
+ }
2168
+ // A room named is a line posted, by the same rules as typing it.
2169
+ const roomAsked = url.searchParams.has("server") || url.searchParams.has("channel");
2170
+ const where = roomAsked ? roomFor(url.searchParams.get("server"), url.searchParams.get("channel")) : null;
2171
+ if (roomAsked && !where) {
2172
+ json(response, 400, { error: "a room is a server address and a channel" });
2173
+ return;
2174
+ }
2175
+ const heard = await speech.transcribe(bytes, { language: languageOf(url.searchParams.get("language")), by: who.id });
2176
+ if (where && options.trollbox && heard.text !== "") {
2177
+ const handle = (await options.handles?.of(who.id)) || fallbackHandle(who.id);
2178
+ const line = await options.trollbox.post(where, who.id, handle, heard.text);
2179
+ json(response, 201, {
2180
+ text: heard.text, seconds: heard.seconds, model: speech.model,
2181
+ message: { id: line.id, handle: line.handle, body: line.body, createdAt: line.createdAt, mine: true },
2182
+ });
2183
+ return;
2184
+ }
2185
+ json(response, 200, { text: heard.text, seconds: heard.seconds, model: speech.model });
2186
+ }
2187
+ catch (error) {
2188
+ if (error instanceof SpeechError || error instanceof TrollboxError)
2189
+ json(response, error.status, { error: error.message });
2190
+ else
2191
+ throw error;
2192
+ }
2193
+ return;
2194
+ }
2130
2195
  if (path === "/api/v1/me/handle" && options.handles && options.accounts) {
2131
2196
  const handles = options.handles;
2132
2197
  const who = await options.accounts.whoIs(tokenFrom(request.headers));
@@ -3182,6 +3247,67 @@ export function createHandler(engine, options) {
3182
3247
  createReadStream(segment).pipe(response);
3183
3248
  return;
3184
3249
  }
3250
+ /*
3251
+ * The channel's captions: what it is saying, as lines, as they are
3252
+ * heard by nixamp.com's ear on this server's behalf. Read the way the
3253
+ * sound is read -- the key gate above has already passed -- and
3254
+ * started by the first person asking. `captions` is the live stream
3255
+ * of lines; `transcript` is the recent ones as JSON, for a poll.
3256
+ */
3257
+ if ((action === "captions" || action === "transcript") && request.method === "GET") {
3258
+ const captions = options.captions;
3259
+ if (!captions) {
3260
+ json(response, 503, { error: "this server cannot caption: it has no ffmpeg" });
3261
+ return;
3262
+ }
3263
+ if (!captions.available()) {
3264
+ json(response, 503, { error: "this server is not signed in to nixamp.com, so it cannot caption; run `nixamp login` on it" });
3265
+ return;
3266
+ }
3267
+ if (!channels.has(id)) {
3268
+ json(response, 404, { error: "nothing is playing on that channel" });
3269
+ return;
3270
+ }
3271
+ // Lines after a moment: a poll's ?after=, or the last line a
3272
+ // reconnecting EventSource saw, which the browser sends by itself.
3273
+ const lastSeen = request.headers["last-event-id"];
3274
+ const after = Number(url.searchParams.get("after") ?? (Array.isArray(lastSeen) ? lastSeen[0] : lastSeen) ?? 0) || 0;
3275
+ if (action === "transcript") {
3276
+ // Asking keeps the captioner up: it stops a minute after the last ask.
3277
+ captions.subscribe(id, () => undefined)?.();
3278
+ json(response, 200, {
3279
+ channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), ...captions.status(id), lines: captions.recent(id, after),
3280
+ });
3281
+ return;
3282
+ }
3283
+ response.writeHead(200, {
3284
+ ...CORS,
3285
+ "content-type": "text/event-stream; charset=utf-8",
3286
+ "cache-control": "no-store",
3287
+ connection: "keep-alive",
3288
+ "x-accel-buffering": "no",
3289
+ });
3290
+ const write = (event, data, eventId) => {
3291
+ response.write(`${eventId === undefined ? "" : `id: ${eventId}\n`}event: ${event}\ndata: ${JSON.stringify(data)}\n\n`);
3292
+ };
3293
+ const off = captions.subscribe(id, (line) => write("line", line, line.at));
3294
+ if (off === null) {
3295
+ response.end();
3296
+ return;
3297
+ }
3298
+ // How far behind the live edge a newcomer's playback starts, so the
3299
+ // page can hold each line until its own sound gets there.
3300
+ write("hello", { channel: id, backlog: BACKLOG_SECONDS, now: Date.now(), ...captions.status(id), lines: captions.recent(id, after) });
3301
+ const beat = setInterval(() => response.write(": beat\n\n"), 20_000);
3302
+ beat.unref?.();
3303
+ const done = () => {
3304
+ clearInterval(beat);
3305
+ off();
3306
+ };
3307
+ request.on("close", done);
3308
+ response.on("close", done);
3309
+ return;
3310
+ }
3185
3311
  if (action === undefined && request.method === "GET") {
3186
3312
  // Listening. The response is the fan-out target: whatever ffmpeg
3187
3313
  // produces for this channel is written to it until one end goes away.
@@ -4450,6 +4576,16 @@ export async function serve(argv, version = "0.1.0") {
4450
4576
  secret: process.env["NIXAMP_JWT_SECRET"] ?? "",
4451
4577
  })
4452
4578
  : undefined;
4579
+ // The ear lives where the accounts live. Loaded now, in the background, so
4580
+ // the first person to speak does not wait for the model to arrive; a box
4581
+ // without the optional model never says it is ready, and answers 503.
4582
+ const speech = accounts && process.env["NIXAMP_STT"] !== "off" ? new Speech() : undefined;
4583
+ if (speech) {
4584
+ void speech.warm().then((ready) => {
4585
+ if (ready)
4586
+ console.error(`nixamp: hearing with ${speech.model}`);
4587
+ });
4588
+ }
4453
4589
  // nixamp as an OAuth 2.1 authorization server, and the watch parties a
4454
4590
  // client site bridges through it. Both need the same three things -- a
4455
4591
  // database, accounts, and a site to be the issuer of -- so both appear or
@@ -4627,6 +4763,18 @@ export async function serve(argv, version = "0.1.0") {
4627
4763
  void durable.sweep(Date.now() - ENDED_TTL_MS, new Date(Date.now() - 30 * 24 * 60 * 60 * 1000));
4628
4764
  }
4629
4765
  const session = readSession();
4766
+ // Captions for a channel: this server's ffmpeg turns the sound into PCM
4767
+ // and nixamp.com's ear, asked as this server, turns that into lines. A
4768
+ // server without ffmpeg cannot; a server nobody signed in on is refused
4769
+ // by the ear, and says so to whoever asks.
4770
+ const captions = tools.carries
4771
+ ? new Captions({
4772
+ ffmpeg: tools.ffmpeg,
4773
+ listen: (id, listener) => channels.listen(id, listener),
4774
+ session: () => readSession(),
4775
+ onEvent: (message) => console.log(` ${message}`),
4776
+ })
4777
+ : undefined;
4630
4778
  const owner = new Owner({
4631
4779
  ownerId: options.owner || (session?.token ? await ownerIdOf(session) : ""),
4632
4780
  site: session?.site ?? DEFAULT_DIRECTORY,
@@ -4810,6 +4958,7 @@ export async function serve(argv, version = "0.1.0") {
4810
4958
  carries: tools.carries !== false,
4811
4959
  cookies: cookiesFile(),
4812
4960
  hls,
4961
+ ...(captions ? { captions } : {}),
4813
4962
  compression,
4814
4963
  enricher,
4815
4964
  ...(tls ? { tls } : {}),
@@ -4824,6 +4973,7 @@ export async function serve(argv, version = "0.1.0") {
4824
4973
  ...(layouts ? { layouts } : {}),
4825
4974
  ...(rooms ? { rooms } : {}),
4826
4975
  ...(trollbox ? { trollbox } : {}),
4976
+ ...(speech ? { speech } : {}),
4827
4977
  ...(tickets ? { tickets } : {}),
4828
4978
  ...(authServer ? { authServer } : {}),
4829
4979
  ...(parties ? { parties } : {}),
@@ -0,0 +1,96 @@
1
+ /** What Whisper listens at. Everything is brought to this before it is heard. */
2
+ export declare const RATE = 16000;
3
+ /** A trollbox line, said out loud, is seconds long; a minute is the ceiling. */
4
+ export declare const MAX_SECONDS = 60;
5
+ /** A minute of 16-bit mono at 16 kHz is under 2 MB; 48 kHz stereo is under 12. */
6
+ export declare const MAX_BYTES: number;
7
+ /**
8
+ * How much one account may have heard in a minute, in seconds of sound.
9
+ * Counted in sound rather than asks because a live channel being captioned
10
+ * asks twelve times a minute for five seconds each, and a person dictating
11
+ * asks twice for thirty: the cost is the sound, not the call. Five minutes
12
+ * of sound a minute is four channels captioned, or a conversation, and
13
+ * not a firehose.
14
+ */
15
+ export declare const SECONDS_PER_MINUTE = 300;
16
+ /** How many may wait for the one CPU. Past this, the honest answer is "later". */
17
+ export declare const QUEUE_LIMIT = 8;
18
+ export declare const DEFAULT_MODEL = "onnx-community/whisper-base";
19
+ export declare class SpeechError extends Error {
20
+ readonly status: number;
21
+ constructor(message: string, status: number);
22
+ }
23
+ export interface Heard {
24
+ text: string;
25
+ /** How long the audio was, in seconds. */
26
+ seconds: number;
27
+ }
28
+ /** Mono samples in [-1, 1] at RATE -> words. What the loaded model is, to this file. */
29
+ export type Recognizer = (pcm: Float32Array, options: {
30
+ language?: string;
31
+ }) => Promise<{
32
+ text: string;
33
+ }>;
34
+ export interface SpeechOptions {
35
+ /** A Hugging Face model id; NIXAMP_STT_MODEL otherwise; whisper-base by default. */
36
+ model?: string;
37
+ /** Where the model's files are kept between runs. */
38
+ cacheDir?: string;
39
+ /** How the model is loaded. The tests hand in a fake; nothing else does. */
40
+ load?: (model: string, cacheDir: string) => Promise<Recognizer>;
41
+ now?: () => number;
42
+ }
43
+ export interface Wav {
44
+ rate: number;
45
+ channels: number;
46
+ /** Mixed down to one channel, in [-1, 1]. */
47
+ samples: Float32Array;
48
+ }
49
+ /** Whether these bytes are a RIFF/WAVE file, whatever the request called them. */
50
+ export declare function isWav(bytes: Uint8Array): boolean;
51
+ /**
52
+ * A WAV file's samples, mixed to mono. PCM of 8, 16, 24 or 32 bits, or
53
+ * 32-bit float; the WAVE_FORMAT_EXTENSIBLE wrapper around either. That is
54
+ * what every encoder that matters writes, including the one in the page.
55
+ */
56
+ export declare function decodeWav(bytes: Uint8Array): Wav;
57
+ /**
58
+ * Samples at one rate, at another. Down is an average over each output
59
+ * sample's span, which is a crude low-pass and enough for speech; up is a
60
+ * straight line between neighbours. Whisper resamples nothing itself.
61
+ */
62
+ export declare function resample(samples: Float32Array, from: number, to: number): Float32Array;
63
+ /** 16-bit mono PCM WAV bytes from samples: what the CLI's tests, and anybody, can send. */
64
+ export declare function encodeWav(samples: Float32Array, rate?: number): Uint8Array;
65
+ /** Whisper's own spacing and blank-audio tokens, tidied into a line. */
66
+ export declare function tidy(text: string): string;
67
+ /** A two-letter language code, or nothing: Whisper guesses when not told. */
68
+ export declare function languageOf(value: unknown): string | undefined;
69
+ export declare class Speech {
70
+ readonly model: string;
71
+ private readonly cacheDir;
72
+ private readonly load;
73
+ private readonly now;
74
+ private recognizer;
75
+ /** One at a time: the model is CPU-bound, and two at once is slower than two in turn. */
76
+ private tail;
77
+ private waiting;
78
+ private readonly asked;
79
+ constructor(options?: SpeechOptions);
80
+ private ear;
81
+ /**
82
+ * Load the model now, so the first person to speak is not the one who
83
+ * waits for the download. Says whether it could; never throws.
84
+ */
85
+ warm(): Promise<boolean>;
86
+ /** Whether this account may have this much heard now, and the bookkeeping if so. */
87
+ allow(accountId: string, seconds: number): void;
88
+ /**
89
+ * The words in a WAV. Refuses what is not a WAV, or is too long, or more
90
+ * than the account may have heard this minute, with a status.
91
+ */
92
+ transcribe(bytes: Uint8Array, options?: {
93
+ language?: string;
94
+ by?: string;
95
+ }): Promise<Heard>;
96
+ }