decibri 3.4.2 → 4.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,15 +4,54 @@ const { Writable } = require('stream');
4
4
  const { DecibriOutputBridge } = require('../index.js');
5
5
  const { wrapNativeError } = require('./errors');
6
6
 
7
- // ─── DecibriOutput (Writable) ───────────────────────────────────────────────
7
+ // The npm package version, reported as `binding` by version(). Read from
8
+ // package.json so it tracks the published package and cannot drift.
9
+ const PACKAGE_VERSION = require('../package.json').version;
8
10
 
9
- class DecibriOutput extends Writable {
11
+ // ─── Speaker (Writable) ─────────────────────────────────────────────────────
12
+
13
+ class Speaker extends Writable {
10
14
  /**
11
- * @param {import('./decibri').DecibriOutputOptions} [options]
15
+ * @param {import('./decibri').SpeakerOptions} [options]
16
+ * @param {{ prepared: object, native: object }} [_internal] Internal: a
17
+ * pre-resolved options bundle and an already-constructed native bridge,
18
+ * passed by the async `Speaker.open()` factory. Not part of the public API.
12
19
  */
13
- constructor(options = {}) {
20
+ constructor(options = {}, _internal = undefined) {
14
21
  super({ highWaterMark: options.highWaterMark || 16384 });
15
22
 
23
+ // Validate and resolve options once. The async factory passes its already
24
+ // resolved bundle through `_internal` to avoid recomputing it.
25
+ const prepared = _internal ? _internal.prepared : Speaker._prepareOptions(options);
26
+
27
+ // ── Store config ───────────────────────────────────────────────────────
28
+
29
+ this._dtype = prepared.dtype;
30
+ this._started = false;
31
+
32
+ // ── Create or adopt native bridge ───────────────────────────────────────
33
+
34
+ if (_internal) {
35
+ // Built off the event loop by Speaker.open(); already wrapped.
36
+ this._native = _internal.native;
37
+ } else {
38
+ try {
39
+ this._native = new DecibriOutputBridge(prepared.nativeOptions);
40
+ } catch (err) {
41
+ throw wrapNativeError(err);
42
+ }
43
+ }
44
+ }
45
+
46
+ /**
47
+ * Validate the constructor options and resolve them into the native options
48
+ * object plus the wrapper-side state. Throws the same `RangeError` /
49
+ * `TypeError` as the constructor on invalid input. Shared by the synchronous
50
+ * constructor and the async `open()` factory.
51
+ * @internal
52
+ * @param {import('./decibri').SpeakerOptions} options
53
+ */
54
+ static _prepareOptions(options) {
16
55
  // ── Validate options ───────────────────────────────────────────────────
17
56
 
18
57
  const sampleRate = options.sampleRate ?? 16000;
@@ -25,33 +64,23 @@ class DecibriOutput extends Writable {
25
64
  throw new RangeError('channels must be between 1 and 32');
26
65
  }
27
66
 
28
- const format = options.format ?? 'int16';
29
- if (format !== 'int16' && format !== 'float32') {
30
- throw new TypeError("format must be 'int16' or 'float32'");
67
+ const dtype = options.dtype ?? 'int16';
68
+ if (dtype !== 'int16' && dtype !== 'float32') {
69
+ throw new TypeError("dtype must be 'int16' or 'float32'");
31
70
  }
32
71
 
33
72
  // ── Resolve device ──────────────────────────────────────────────────────
34
73
 
74
+ // Name and multi-match resolution are delegated to the core, which owns
75
+ // the renamed-vocabulary errors (SpeakerNotFound / MultipleDevicesMatch).
76
+ // A string name and an { id } object are passed straight through to the
77
+ // native addon. Only the numeric index keeps a client-side bounds check,
78
+ // for a clean Node-side RangeError without a round-trip.
35
79
  let resolvedDevice = options.device;
36
- if (typeof options.device === 'string') {
37
- const lower = options.device.toLowerCase();
38
- const matches = DecibriOutputBridge.devices().filter(d =>
39
- d.name.toLowerCase().includes(lower)
40
- );
41
- if (matches.length === 0) {
42
- throw new TypeError(`No audio output device found matching "${options.device}"`);
43
- }
44
- if (matches.length > 1) {
45
- const names = matches.map(d => ` [${d.index}] ${d.name}`).join('\n');
46
- throw new TypeError(
47
- `Multiple devices match "${options.device}":\n${names}\nUse a more specific name or pass the device index directly.`
48
- );
49
- }
50
- resolvedDevice = matches[0].index;
51
- } else if (typeof options.device === 'number') {
80
+ if (typeof options.device === 'number') {
52
81
  const devices = DecibriOutputBridge.devices();
53
82
  if (options.device < 0 || options.device >= devices.length) {
54
- throw new RangeError('device index out of range. Call DecibriOutput.devices() to list available devices');
83
+ throw new RangeError('device index out of range. Call Speaker.devices() to list available devices');
55
84
  }
56
85
  resolvedDevice = options.device;
57
86
  } else if (
@@ -67,23 +96,41 @@ class DecibriOutput extends Writable {
67
96
  resolvedDevice = options.device;
68
97
  }
69
98
 
70
- // ── Store config ───────────────────────────────────────────────────────
71
-
72
- this._format = format;
73
- this._started = false;
74
-
75
- // ── Create native bridge ───────────────────────────────────────────────
76
-
77
- try {
78
- this._native = new DecibriOutputBridge({
99
+ return {
100
+ dtype,
101
+ nativeOptions: {
79
102
  sampleRate,
80
103
  channels,
81
- format,
104
+ format: dtype,
82
105
  device: resolvedDevice,
83
- });
106
+ },
107
+ };
108
+ }
109
+
110
+ /**
111
+ * Construct a Speaker without blocking the event loop.
112
+ *
113
+ * Symmetric with `Microphone.open()` and the Python `AsyncSpeaker.open()`.
114
+ * The speaker loads no model, so the only open work is device resolution and
115
+ * the practical blocking risk is small; this factory exists chiefly so async
116
+ * callers can use one consistent construction pattern across both classes.
117
+ * The synchronous constructor remains available and unchanged.
118
+ *
119
+ * A failed open (unknown device) rejects the returned Promise with the
120
+ * matching error class rather than throwing synchronously.
121
+ *
122
+ * @param {import('./decibri').SpeakerOptions} [options]
123
+ * @returns {Promise<Speaker>}
124
+ */
125
+ static async open(options = {}) {
126
+ const prepared = Speaker._prepareOptions(options);
127
+ let native;
128
+ try {
129
+ native = await DecibriOutputBridge.openAsync(prepared.nativeOptions);
84
130
  } catch (err) {
85
131
  throw wrapNativeError(err);
86
132
  }
133
+ return new Speaker(options, { prepared, native });
87
134
  }
88
135
 
89
136
  /** @internal */
@@ -110,6 +157,56 @@ class DecibriOutput extends Writable {
110
157
  }
111
158
  }
112
159
 
160
+ /**
161
+ * Write PCM audio without blocking the event loop.
162
+ *
163
+ * The blocking part of a write is the backpressure wait when the native
164
+ * playback queue is full; the synchronous stream path (`write()` / `pipe()`)
165
+ * performs that wait on the event loop. This method performs it on the native
166
+ * thread pool and resolves when the samples are queued. The audio stream is
167
+ * created on the first call (a fast device open) and stays on its own thread;
168
+ * only the queue handoff runs off the event loop.
169
+ *
170
+ * Additive and non-blocking: the synchronous `write()` / `pipe()` stream
171
+ * interface is unchanged. This is a direct, opt-in alternative that bypasses
172
+ * the Writable buffer, so do not interleave it with `write()` / `pipe()` on
173
+ * the same instance; pick one path per instance. Await calls sequentially to
174
+ * preserve sample order. An empty buffer resolves immediately. A failed write
175
+ * (a closed or stopped stream) rejects with the matching error class.
176
+ *
177
+ * @param {Buffer} chunk PCM samples in the configured `dtype`.
178
+ * @returns {Promise<void>}
179
+ */
180
+ async writeAsync(chunk) {
181
+ try {
182
+ await this._native.writeAsync(chunk);
183
+ } catch (err) {
184
+ throw wrapNativeError(err);
185
+ }
186
+ }
187
+
188
+ /**
189
+ * Wait for all queued audio to finish playing without blocking the event
190
+ * loop.
191
+ *
192
+ * The synchronous drain (run by `end()` / `_final`) polls for completion on
193
+ * the event loop for the full playback tail; this method runs that wait on the
194
+ * native thread pool and resolves when the buffer has drained. If nothing has
195
+ * been written yet it resolves immediately.
196
+ *
197
+ * Additive and non-blocking: the synchronous drain via `end()` is unchanged.
198
+ * Pair this with `writeAsync()` for a fully non-blocking playback path.
199
+ *
200
+ * @returns {Promise<void>}
201
+ */
202
+ async drainAsync() {
203
+ try {
204
+ await this._native.drainAsync();
205
+ } catch (err) {
206
+ throw wrapNativeError(err);
207
+ }
208
+ }
209
+
113
210
  /**
114
211
  * Immediate stop. Discards remaining buffered audio.
115
212
  */
@@ -134,12 +231,13 @@ class DecibriOutput extends Writable {
134
231
  }
135
232
 
136
233
  /**
137
- * Version information for decibri and the audio runtime.
138
- * @returns {{ decibri: string, portaudio: string }}
234
+ * Version information for decibri, the audio backend, and this binding.
235
+ * @returns {{ decibri: string, audioBackend: string, binding: string }}
139
236
  */
140
237
  static version() {
141
- return DecibriOutputBridge.version();
238
+ const v = DecibriOutputBridge.version();
239
+ return { decibri: v.decibri, audioBackend: v.audioBackend, binding: PACKAGE_VERSION };
142
240
  }
143
241
  }
144
242
 
145
- module.exports = DecibriOutput;
243
+ module.exports = Speaker;
package/src/decibri.d.ts CHANGED
@@ -1,7 +1,7 @@
1
1
  import { Readable, ReadableOptions, Writable, WritableOptions } from 'stream';
2
2
 
3
3
  /** Information about an available audio input device. */
4
- export interface DeviceInfo {
4
+ export interface MicrophoneInfo {
5
5
  /** Device index (pass to constructor as `device`). */
6
6
  index: number;
7
7
  /** Human-readable device name from the OS. */
@@ -23,16 +23,18 @@ export interface DeviceInfo {
23
23
  isDefault: boolean;
24
24
  }
25
25
 
26
- /** Version strings returned by `Decibri.version()`. */
26
+ /** Version strings returned by `Microphone.version()`. */
27
27
  export interface VersionInfo {
28
- /** decibri package version (e.g. `"3.0.0"`). */
28
+ /** decibri core version. */
29
29
  decibri: string;
30
- /** Audio runtime version string (e.g. `"cpal 0.17"`). */
31
- portaudio: string;
30
+ /** Audio backend version string (e.g. `"cpal 0.17"`). */
31
+ audioBackend: string;
32
+ /** This binding's npm package version. */
33
+ binding: string;
32
34
  }
33
35
 
34
- /** Constructor options for `Decibri`. */
35
- export interface DecibriOptions extends ReadableOptions {
36
+ /** Constructor options for `Microphone`. */
37
+ export interface MicrophoneOptions extends ReadableOptions {
36
38
  /**
37
39
  * Sample rate in Hz.
38
40
  * @default 16000
@@ -57,53 +59,50 @@ export interface DecibriOptions extends ReadableOptions {
57
59
 
58
60
  /**
59
61
  * Audio input device. One of:
60
- * - numeric index (from `DeviceInfo.index`)
62
+ * - numeric index (from `MicrophoneInfo.index`)
61
63
  * - case-insensitive name substring
62
- * - `{ id: string }` for stable per-host device ID from `DeviceInfo.id`
64
+ * - `{ id: string }` for stable per-host device ID from `MicrophoneInfo.id`
63
65
  *
64
66
  * Omit to use the system default input device.
65
67
  */
66
68
  device?: number | string | { id: string };
67
69
 
68
70
  /**
69
- * Sample encoding format.
71
+ * Sample encoding data type.
70
72
  * - `'int16'`: 16-bit signed integer, little-endian (2 bytes per sample)
71
73
  * - `'float32'`: 32-bit IEEE 754 float, little-endian (4 bytes per sample)
72
74
  * @default 'int16'
73
75
  */
74
- format?: 'int16' | 'float32';
76
+ dtype?: 'int16' | 'float32';
75
77
 
76
78
  /**
77
- * Enable energy-based voice activity detection.
78
- * When enabled, emits `'speech'` and `'silence'` events.
79
+ * Voice activity detection mode. One of:
80
+ * - `false`: disabled (default)
81
+ * - `'silero'`: Silero VAD v5 ML model (more accurate, ~1ms inference)
82
+ * - `'energy'`: RMS energy threshold (lightweight)
83
+ *
84
+ * When enabled, emits `'speech'` and `'silence'` events and updates `vadScore`.
85
+ * The legacy `vad: true` form is rejected; specify the mode explicitly.
79
86
  * @default false
80
87
  */
81
- vad?: boolean;
88
+ vad?: false | 'silero' | 'energy';
82
89
 
83
90
  /**
84
- * RMS energy threshold for speech detection (VAD mode only).
85
- * @default 0.01
91
+ * Speech-detection threshold for the active VAD mode.
92
+ * @default 0.5 for `'silero'`, 0.01 for `'energy'`
86
93
  * @range 0–1
87
94
  */
88
95
  vadThreshold?: number;
89
96
 
90
97
  /**
91
- * Milliseconds of sub-threshold audio before emitting `'silence'` (VAD mode only).
98
+ * Milliseconds of sub-threshold audio before emitting `'silence'`.
92
99
  * @default 300
93
100
  */
94
101
  vadHoldoff?: number;
95
102
 
96
- /**
97
- * VAD engine to use.
98
- * - `'energy'`: RMS energy threshold (default, lightweight)
99
- * - `'silero'`: Silero VAD v5 ML model (more accurate, ~1ms inference)
100
- * @default 'energy'
101
- */
102
- vadMode?: 'energy' | 'silero';
103
-
104
103
  /**
105
104
  * Path to the Silero VAD ONNX model file.
106
- * Only used when `vadMode` is `'silero'`.
105
+ * Only used when `vad` is `'silero'`.
107
106
  * Defaults to `models/silero_vad.onnx` relative to the package.
108
107
  */
109
108
  modelPath?: string;
@@ -114,16 +113,37 @@ export interface DecibriOptions extends ReadableOptions {
114
113
  *
115
114
  * @example
116
115
  * ```js
117
- * const Decibri = require('decibri');
118
- * const mic = new Decibri({ sampleRate: 16000, channels: 1 });
116
+ * const { Microphone } = require('decibri');
117
+ * const mic = new Microphone({ sampleRate: 16000, channels: 1 });
119
118
  * mic.on('data', (chunk) => {
120
119
  * // chunk is a Buffer of Int16 LE PCM samples
121
120
  * });
122
121
  * setTimeout(() => mic.stop(), 5000);
123
122
  * ```
124
123
  */
125
- declare class Decibri extends Readable {
126
- constructor(options?: DecibriOptions);
124
+ export declare class Microphone extends Readable {
125
+ constructor(options?: MicrophoneOptions);
126
+
127
+ /**
128
+ * Construct a Microphone without blocking the event loop on the open work.
129
+ *
130
+ * The synchronous constructor loads the Silero VAD model inline when
131
+ * `vad: 'silero'` is set, blocking the event loop for roughly 100 to 500 ms
132
+ * on a cold cache. This factory runs that load (and device resolution) on the
133
+ * native thread pool and resolves to a ready instance. The synchronous
134
+ * constructor remains available and unchanged.
135
+ *
136
+ * Options are identical to the constructor. A failed open rejects the Promise
137
+ * with the matching error: `RangeError` / `TypeError` for invalid options, or
138
+ * a `DeviceError` / `OrtError` / `OrtPathError` for native failures.
139
+ *
140
+ * @example
141
+ * ```js
142
+ * const mic = await Microphone.open({ vad: 'silero' });
143
+ * mic.on('data', (chunk) => { ... });
144
+ * ```
145
+ */
146
+ static open(options?: MicrophoneOptions): Promise<Microphone>;
127
147
 
128
148
  /** Stop microphone capture and end the stream. Safe to call multiple times. */
129
149
  stop(): void;
@@ -131,8 +151,15 @@ declare class Decibri extends Readable {
131
151
  /** Whether the microphone is currently capturing audio. */
132
152
  readonly isOpen: boolean;
133
153
 
154
+ /**
155
+ * Most recent VAD score for the active mode: the Silero speech probability in
156
+ * `'silero'` mode, the normalized RMS of the last chunk in `'energy'` mode.
157
+ * 0 when VAD is disabled or before the first chunk is processed.
158
+ */
159
+ readonly vadScore: number;
160
+
134
161
  /** List all available audio input devices. */
135
- static devices(): DeviceInfo[];
162
+ static devices(): MicrophoneInfo[];
136
163
 
137
164
  /** Version information for decibri and the audio runtime. */
138
165
  static version(): VersionInfo;
@@ -162,13 +189,13 @@ declare class Decibri extends Readable {
162
189
  }
163
190
 
164
191
  /** Information about an available audio output device. */
165
- export interface OutputDeviceInfo {
192
+ export interface SpeakerInfo {
166
193
  /** Device index (pass to constructor as `device`). */
167
194
  index: number;
168
195
  /** Human-readable device name from the OS. */
169
196
  name: string;
170
197
  /**
171
- * Stable per-host device ID. See `DeviceInfo.id` for format and fallback
198
+ * Stable per-host device ID. See `MicrophoneInfo.id` for format and fallback
172
199
  * semantics; identical rules for output devices.
173
200
  */
174
201
  id: string;
@@ -180,8 +207,8 @@ export interface OutputDeviceInfo {
180
207
  isDefault: boolean;
181
208
  }
182
209
 
183
- /** Constructor options for `DecibriOutput`. */
184
- export interface DecibriOutputOptions extends WritableOptions {
210
+ /** Constructor options for `Speaker`. */
211
+ export interface SpeakerOptions extends WritableOptions {
185
212
  /**
186
213
  * Sample rate in Hz.
187
214
  * @default 16000
@@ -197,18 +224,18 @@ export interface DecibriOutputOptions extends WritableOptions {
197
224
  channels?: number;
198
225
 
199
226
  /**
200
- * Sample encoding format of incoming data.
227
+ * Sample encoding data type of incoming data.
201
228
  * - `'int16'`: 16-bit signed integer, little-endian (2 bytes per sample)
202
229
  * - `'float32'`: 32-bit IEEE 754 float, little-endian (4 bytes per sample)
203
230
  * @default 'int16'
204
231
  */
205
- format?: 'int16' | 'float32';
232
+ dtype?: 'int16' | 'float32';
206
233
 
207
234
  /**
208
235
  * Audio output device. One of:
209
- * - numeric index (from `OutputDeviceInfo.index`)
236
+ * - numeric index (from `SpeakerInfo.index`)
210
237
  * - case-insensitive name substring
211
- * - `{ id: string }` for stable per-host device ID from `OutputDeviceInfo.id`
238
+ * - `{ id: string }` for stable per-host device ID from `SpeakerInfo.id`
212
239
  *
213
240
  * Omit to use the system default output device.
214
241
  */
@@ -220,14 +247,54 @@ export interface DecibriOutputOptions extends WritableOptions {
220
247
  *
221
248
  * @example
222
249
  * ```js
223
- * const { DecibriOutput } = require('decibri');
224
- * const speaker = new DecibriOutput({ sampleRate: 16000, channels: 1 });
250
+ * const { Speaker } = require('decibri');
251
+ * const speaker = new Speaker({ sampleRate: 16000, channels: 1 });
225
252
  * speaker.write(pcmBuffer);
226
253
  * speaker.end();
227
254
  * ```
228
255
  */
229
- declare class DecibriOutput extends Writable {
230
- constructor(options?: DecibriOutputOptions);
256
+ export declare class Speaker extends Writable {
257
+ constructor(options?: SpeakerOptions);
258
+
259
+ /**
260
+ * Construct a Speaker without blocking the event loop. Symmetric with
261
+ * `Microphone.open()`. The speaker loads no model, so the only open work is
262
+ * device resolution; this factory is provided so async callers can use one
263
+ * consistent construction pattern across both classes. The synchronous
264
+ * constructor remains available and unchanged.
265
+ *
266
+ * A failed open (unknown device) rejects the Promise with the matching error.
267
+ *
268
+ * @example
269
+ * ```js
270
+ * const speaker = await Speaker.open({ sampleRate: 24000 });
271
+ * speaker.write(pcmBuffer);
272
+ * ```
273
+ */
274
+ static open(options?: SpeakerOptions): Promise<Speaker>;
275
+
276
+ /**
277
+ * Write PCM audio without blocking the event loop. Performs the backpressure
278
+ * wait (when the native playback queue is full) on the native thread pool and
279
+ * resolves when the samples are queued.
280
+ *
281
+ * Additive: the synchronous `write()` / `pipe()` stream interface is
282
+ * unchanged. This is a direct, opt-in alternative that bypasses the Writable
283
+ * buffer; do not interleave it with `write()` / `pipe()` on the same instance.
284
+ * Await calls sequentially to preserve sample order. An empty buffer resolves
285
+ * immediately; a closed or stopped stream rejects with the matching error.
286
+ */
287
+ writeAsync(chunk: Buffer): Promise<void>;
288
+
289
+ /**
290
+ * Wait for all queued audio to finish playing without blocking the event
291
+ * loop. Runs the drain wait on the native thread pool and resolves when the
292
+ * buffer has drained; resolves immediately if nothing was written.
293
+ *
294
+ * Additive: the synchronous drain via `end()` is unchanged. Pair with
295
+ * `writeAsync()` for a fully non-blocking playback path.
296
+ */
297
+ drainAsync(): Promise<void>;
231
298
 
232
299
  /** Immediate stop. Discards remaining buffered audio. */
233
300
  stop(): void;
@@ -236,7 +303,7 @@ declare class DecibriOutput extends Writable {
236
303
  readonly isPlaying: boolean;
237
304
 
238
305
  /** List all available audio output devices. */
239
- static devices(): OutputDeviceInfo[];
306
+ static devices(): SpeakerInfo[];
240
307
 
241
308
  /** Version information for decibri and the audio runtime. */
242
309
  static version(): VersionInfo;
@@ -252,8 +319,36 @@ declare class DecibriOutput extends Writable {
252
319
  on(event: string | symbol, listener: (...args: any[]) => void): this;
253
320
  }
254
321
 
255
- export = Decibri;
322
+ /** List all available audio input devices. */
323
+ export declare function inputDevices(): MicrophoneInfo[];
324
+
325
+ /** List all available audio output devices. */
326
+ export declare function outputDevices(): SpeakerInfo[];
256
327
 
257
- declare namespace Decibri {
258
- export { DecibriOutput };
328
+ /** Version information for decibri and the audio runtime. */
329
+ export declare function version(): VersionInfo;
330
+
331
+ /**
332
+ * Base class for errors raised by the decibri native bindings.
333
+ *
334
+ * Catch this to handle any decibri device or ONNX Runtime failure generically;
335
+ * catch a subclass for finer control. Argument validation (bad sample rate,
336
+ * channels, frames, dtype, vad) throws built-in `RangeError` / `TypeError`, not
337
+ * a `DecibriError`.
338
+ */
339
+ export declare class DecibriError extends Error {
340
+ /** Stable string code identifying the specific failure. */
341
+ readonly code: string;
259
342
  }
343
+
344
+ /**
345
+ * Device enumeration or selection failure: an unmatched device name, an
346
+ * ambiguous name match, or missing hardware.
347
+ */
348
+ export declare class DeviceError extends DecibriError {}
349
+
350
+ /** ONNX Runtime setup or inference failure (Silero VAD). */
351
+ export declare class OrtError extends DecibriError {}
352
+
353
+ /** A specific ONNX Runtime library path could not be loaded. */
354
+ export declare class OrtPathError extends OrtError {}