decibri 4.4.2 → 5.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/decibri.d.ts CHANGED
@@ -33,6 +33,34 @@ export interface VersionInfo {
33
33
  binding: string;
34
34
  }
35
35
 
36
+ /**
37
+ * Voice-activity-detection config object, passed on the `vad` option to tune
38
+ * the detector's threshold and holdoff. The bare `vad: 'silero'` / `vad:
39
+ * 'energy'` shorthand selects a mode with its default policy; pass this object
40
+ * to override the threshold or holdoff.
41
+ */
42
+ export interface VadOptions {
43
+ /**
44
+ * Which detector to run.
45
+ * - `'silero'`: Silero VAD v5 ML model (more accurate, ~1ms inference)
46
+ * - `'energy'`: RMS energy threshold (lightweight, no model)
47
+ */
48
+ model: 'silero' | 'energy';
49
+
50
+ /**
51
+ * Speech-detection threshold for the active mode.
52
+ * @default 0.5 for `'silero'`, 0.01 for `'energy'`
53
+ * @range 0–1
54
+ */
55
+ threshold?: number;
56
+
57
+ /**
58
+ * Milliseconds of sub-threshold audio before emitting `'silence'`.
59
+ * @default 300
60
+ */
61
+ holdoffMs?: number;
62
+ }
63
+
36
64
  /** Constructor options for `Microphone`. */
37
65
  export interface MicrophoneOptions extends ReadableOptions {
38
66
  /**
@@ -43,9 +71,12 @@ export interface MicrophoneOptions extends ReadableOptions {
43
71
  sampleRate?: number;
44
72
 
45
73
  /**
46
- * Number of input channels.
74
+ * Number of input channels. Mono only: the only accepted value is `1`, and a
75
+ * value greater than `1` throws a `RangeError` (multichannel capture is not
76
+ * supported) rather than being silently downmixed. The option is kept for
77
+ * forward compatibility: a future release may accept a value greater than `1`
78
+ * by delivering true interleaved multichannel.
47
79
  * @default 1
48
- * @range 1–32
49
80
  */
50
81
  channels?: number;
51
82
 
@@ -76,36 +107,85 @@ export interface MicrophoneOptions extends ReadableOptions {
76
107
  dtype?: 'int16' | 'float32';
77
108
 
78
109
  /**
79
- * Voice activity detection mode. One of:
110
+ * Voice activity detection. One of:
80
111
  * - `false`: disabled (default)
81
112
  * - `'silero'`: Silero VAD v5 ML model (more accurate, ~1ms inference)
82
113
  * - `'energy'`: RMS energy threshold (lightweight)
114
+ * - a `VadOptions` config object `{ model, threshold?, holdoffMs? }` to tune
115
+ * the threshold and holdoff for the chosen model
83
116
  *
84
- * When enabled, emits `'speech'` and `'silence'` events and updates `vadScore`.
85
- * The legacy `vad: true` form is rejected; specify the mode explicitly.
117
+ * The string shorthand uses the mode's default threshold (0.5 for `'silero'`,
118
+ * 0.01 for `'energy'`) and a 300 ms holdoff; pass a `VadOptions` object to
119
+ * override them. When enabled, emits `'speech'` and `'silence'` events and
120
+ * updates `vadScore`. The legacy `vad: true` form is rejected; specify the
121
+ * mode explicitly.
86
122
  * @default false
87
123
  */
88
- vad?: false | 'silero' | 'energy';
124
+ vad?: false | 'silero' | 'energy' | VadOptions;
89
125
 
90
126
  /**
91
- * Speech-detection threshold for the active VAD mode.
92
- * @default 0.5 for `'silero'`, 0.01 for `'energy'`
93
- * @range 0–1
127
+ * Path to the Silero VAD ONNX model file.
128
+ * Only used when `vad` is `'silero'`.
129
+ * Defaults to `models/silero_vad.onnx` relative to the package.
94
130
  */
95
- vadThreshold?: number;
131
+ modelPath?: string;
96
132
 
97
133
  /**
98
- * Milliseconds of sub-threshold audio before emitting `'silence'`.
99
- * @default 300
134
+ * Remove a constant (DC) offset from the captured audio with a one-pole
135
+ * DC-blocking high-pass. Set `true` to enable it; omit or set `false` to
136
+ * leave it off (the default), which keeps the capture path byte-identical.
137
+ * Runs first in the chain, before denoise, and is same-length with no added
138
+ * latency, so `vadScore` and the `speech` / `silence` events are unaffected.
139
+ * Pure DSP: no bundled file or download is needed.
140
+ * @default undefined
100
141
  */
101
- vadHoldoff?: number;
142
+ dcRemoval?: boolean;
102
143
 
103
144
  /**
104
- * Path to the Silero VAD ONNX model file.
105
- * Only used when `vad` is `'silero'`.
106
- * Defaults to `models/silero_vad.onnx` relative to the package.
145
+ * Single-channel speech enhancement (denoise) model applied to the captured
146
+ * audio. The only accepted value is `'fastenhancer-t'`; omit to leave denoise
147
+ * off (the default), which keeps the capture path unchanged. The bundled
148
+ * model ships with the package; no path is required.
149
+ *
150
+ * When set, the captured audio is denoised before delivery and the `'data'`
151
+ * chunks carry the enhanced signal. VAD reads the pre-enhancement signal, so
152
+ * `vadScore` and the `speech` / `silence` events are unaffected.
153
+ * @default undefined
107
154
  */
108
- modelPath?: string;
155
+ denoise?: 'fastenhancer-t';
156
+
157
+ /**
158
+ * High-pass filter cutoff in Hz applied to the captured audio, removing
159
+ * low-frequency rumble below the voice band. The accepted values are `80` (an
160
+ * 80 Hz second-order Butterworth high-pass) and `100` (a 100 Hz one); omit to
161
+ * leave the high-pass off (the default), which keeps the capture path
162
+ * full-range. Runs after denoise in the chain. The closed value set is
163
+ * designed to grow (further cutoffs are additive) the way `denoise` grows.
164
+ * Out-of-set values raise a `RangeError`.
165
+ * @default undefined
166
+ */
167
+ highpass?: 80 | 100;
168
+
169
+ /**
170
+ * Automatic gain control target level in dBFS applied to the captured audio.
171
+ * Drives the running level toward this target with a smoothed, rate-limited
172
+ * gain. An integer in the range -40 to -3 (typical -18); omit to leave AGC
173
+ * off (the default), which keeps the level untouched. Runs after the
174
+ * high-pass step. Out-of-range values raise a `RangeError`.
175
+ * @default undefined
176
+ */
177
+ agc?: number;
178
+
179
+ /**
180
+ * Peak limiter ceiling in dBFS (sample-peak) applied to the captured audio.
181
+ * Holds the signal at or below this ceiling, the safety net that catches a
182
+ * transient the AGC's gain would let exceed full scale. A number in the range
183
+ * -3.0 to 0.0 (typical -1.0); omit to leave the limiter off (the default),
184
+ * which keeps the level untouched. Runs last in the chain, after the AGC step.
185
+ * Out-of-range values raise a `RangeError`.
186
+ * @default undefined
187
+ */
188
+ limiter?: number;
109
189
  }
110
190
 
111
191
  /**
@@ -196,6 +276,235 @@ export declare class Microphone extends Readable {
196
276
  }
197
277
 
198
278
  /** Information about an available audio output device. */
279
+ /** Constructor options for `File`. */
280
+ export interface FileOptions extends ReadableOptions {
281
+ /**
282
+ * Target output rate in Hz: the rate every delivered chunk carries. The
283
+ * source's input rate (from the WAV header, or `inputRate` for
284
+ * `File.buffer`) is resampled to this rate, so a 44.1 kHz recording comes
285
+ * out at 16 kHz unless you set `sampleRate`. The same meaning the option
286
+ * has on `Microphone`.
287
+ * @default 16000
288
+ * @range 1000–384000
289
+ */
290
+ sampleRate?: number;
291
+
292
+ /**
293
+ * Sample encoding data type of the delivered chunks.
294
+ * - `'int16'`: 16-bit signed integer, little-endian (2 bytes per sample)
295
+ * - `'float32'`: 32-bit IEEE 754 float, little-endian (4 bytes per sample)
296
+ * @default 'int16'
297
+ */
298
+ dtype?: 'int16' | 'float32';
299
+
300
+ /**
301
+ * Voice activity detection, opt-in exactly as on `Microphone`: `false`
302
+ * (default), `'silero'`, `'energy'`, or a `VadOptions` config object.
303
+ * When enabled, per-chunk detection runs alongside the stream (the
304
+ * `'speech'` / `'silence'` events and `vadScore`, with the holdoff
305
+ * measured in FILE time rather than wall-clock time) and `'silero'`
306
+ * additionally enables the whole-file `analyze()`. With no `vad` set the
307
+ * File simply conditions audio: no scores, no segments, no speech events.
308
+ * @default false
309
+ */
310
+ vad?: false | 'silero' | 'energy' | VadOptions;
311
+
312
+ /**
313
+ * Path to the Silero VAD ONNX model file.
314
+ * Only used when `vad` is `'silero'`.
315
+ * Defaults to `models/silero_vad.onnx` relative to the package.
316
+ */
317
+ modelPath?: string;
318
+
319
+ /**
320
+ * Remove a constant (DC) offset with a one-pole DC-blocking high-pass,
321
+ * exactly as on `Microphone`.
322
+ * @default undefined
323
+ */
324
+ dcRemoval?: boolean;
325
+
326
+ /**
327
+ * Single-channel speech enhancement (denoise) model, exactly as on
328
+ * `Microphone`. The only accepted value is `'fastenhancer-t'`.
329
+ * @default undefined
330
+ */
331
+ denoise?: 'fastenhancer-t';
332
+
333
+ /**
334
+ * High-pass filter cutoff in Hz (`80` or `100`), exactly as on
335
+ * `Microphone`.
336
+ * @default undefined
337
+ */
338
+ highpass?: 80 | 100;
339
+
340
+ /**
341
+ * Automatic gain control target level in dBFS, exactly as on `Microphone`.
342
+ * @default undefined
343
+ * @range -40 to -3
344
+ */
345
+ agc?: number;
346
+
347
+ /**
348
+ * Peak limiter ceiling in dBFS (sample-peak), exactly as on `Microphone`.
349
+ * @default undefined
350
+ * @range -3.0 to 0.0
351
+ */
352
+ limiter?: number;
353
+ }
354
+
355
+ /** Options for `File.buffer`: `FileOptions` plus the samples' native rate. */
356
+ export interface FileBufferOptions extends FileOptions {
357
+ /**
358
+ * The native rate of the in-memory samples in Hz. Required: raw samples
359
+ * carry no header to read a rate from. The samples are resampled from this
360
+ * rate to `sampleRate`.
361
+ * @range 1000–384000
362
+ */
363
+ inputRate: number;
364
+ }
365
+
366
+ /**
367
+ * One scored voice-activity window of a recording, produced by
368
+ * `File.analyze()`. Windows tile the recording from the start in fixed steps
369
+ * (512 samples at 16 kHz, 32 ms per window); a trailing remainder shorter
370
+ * than one window is not scored, exactly as live detection leaves a
371
+ * sub-window remainder unscored.
372
+ */
373
+ export interface VadWindow {
374
+ /** Window start, in seconds of file time. */
375
+ start: number;
376
+ /** Window end, in seconds of file time. */
377
+ end: number;
378
+ /**
379
+ * Speech probability for this window (0 to 1). The same quantity the live
380
+ * per-chunk `vadScore` reports, here per window across the whole recording.
381
+ */
382
+ vadScore: number;
383
+ /**
384
+ * Whether `vadScore` meets the configured threshold. The raw per-window
385
+ * test, not the debounced speaking state.
386
+ */
387
+ isSpeech: boolean;
388
+ }
389
+
390
+ /**
391
+ * One merged speech region of a recording, produced by `File.analyze()`:
392
+ * consecutive speech windows whose silence gaps are within the configured
393
+ * holdoff collapse into one segment. The segment ends at the last speech
394
+ * window, not at the holdoff expiry.
395
+ */
396
+ export interface Segment {
397
+ /** Region start, in seconds of file time. */
398
+ start: number;
399
+ /** Region end, in seconds of file time. */
400
+ end: number;
401
+ }
402
+
403
+ /** The whole-recording voice-activity analysis `File.analyze()` resolves to. */
404
+ export interface VadReport {
405
+ /** Per-window speech scores across the whole recording, in file order. */
406
+ scores: VadWindow[];
407
+ /** Merged speech regions across the whole recording, in file order. */
408
+ segments: Segment[];
409
+ }
410
+
411
+ /**
412
+ * Offline audio source: conditions a recording or in-memory samples through
413
+ * the same chain as the live `Microphone`, delivered as a finite Readable
414
+ * stream of conditioned chunks that ends at EOF (after the chain's
415
+ * end-of-stream tail). Because a `File` is a complete recording, it can also
416
+ * analyze the whole recording for speech with `analyze()` / `analyse()`,
417
+ * which a live stream cannot do.
418
+ *
419
+ * Construction: `new File(path)` reads the WAV synchronously (fine for a
420
+ * script; it blocks the event loop on disk I/O), `await File.open(path)` reads
421
+ * it off the event loop (the recommended form, mirroring `Microphone.open`),
422
+ * and `File.buffer(samples, { inputRate })` wraps a `Float32Array` of samples
423
+ * you already hold (a raw `Buffer` of bytes is rejected as ambiguous).
424
+ *
425
+ * Iteration and analysis are separate single passes: each consumes the source
426
+ * once, so construct one `File` per operation.
427
+ *
428
+ * Note: Node also has a global `File` (the web File API). Import decibri's
429
+ * explicitly (`const { File } = require('decibri')`) or reference it as
430
+ * `decibri.File` to avoid shadowing surprises.
431
+ *
432
+ * @example
433
+ * const { File } = require('decibri');
434
+ * const file = await File.open('clip.wav', { denoise: 'fastenhancer-t' });
435
+ * file.on('data', (chunk) => { /* Buffer of conditioned Int16 PCM *\/ });
436
+ * file.on('end', () => console.log('done'));
437
+ *
438
+ * @example
439
+ * // Where is the speech?
440
+ * const f = await File.open('clip.wav', { vad: 'silero' });
441
+ * const report = await f.analyze();
442
+ * for (const s of report.segments) console.log(s.start, s.end);
443
+ */
444
+ export declare class File extends Readable {
445
+ /**
446
+ * Open a WAV file synchronously (blocks on disk I/O; prefer `File.open`
447
+ * in servers). Supports 16-bit PCM and 32-bit float WAV files; the input
448
+ * rate and channel count come from the header.
449
+ */
450
+ constructor(path: string, options?: FileOptions);
451
+
452
+ /**
453
+ * Open a WAV file without blocking the event loop: the disk read, WAV
454
+ * parse, and chain construction run on the native thread pool. The
455
+ * recommended form, mirroring `Microphone.open`.
456
+ */
457
+ static open(path: string, options?: FileOptions): Promise<File>;
458
+
459
+ /**
460
+ * Wrap in-memory samples as an offline source. `samples` must be a
461
+ * `Float32Array` of mono samples in [-1.0, 1.0]; a raw `Buffer` of PCM
462
+ * bytes is rejected as ambiguous. `inputRate` is required (raw samples
463
+ * carry no header). Synchronous: no I/O is involved.
464
+ */
465
+ static buffer(samples: Float32Array, options: FileBufferOptions): File;
466
+
467
+ /**
468
+ * Most recent per-chunk VAD score for the active mode: the Silero speech
469
+ * probability in `'silero'` mode, the normalized RMS of the
470
+ * pre-conditioning signal in `'energy'` mode. 0 when VAD is disabled or
471
+ * before the first chunk.
472
+ */
473
+ readonly vadScore: number;
474
+
475
+ /**
476
+ * Analyze the whole recording for speech, off the event loop. Resolves to
477
+ * a `VadReport` with per-window `scores` and merged speech `segments`,
478
+ * all in seconds of file time. Consumes the source (a `File` is a single
479
+ * pass). Requires `vad: 'silero'`: a File opened without `vad` rejects
480
+ * with the core's "analysis requires VAD" error, and the energy mode has
481
+ * no whole-file analysis.
482
+ */
483
+ analyze(): Promise<VadReport>;
484
+
485
+ /** The same whole-recording analysis under the international spelling. */
486
+ analyse(): Promise<VadReport>;
487
+
488
+ /** Release the source. Idempotent; a closed File reads as ended. */
489
+ close(): void;
490
+
491
+ on(event: 'data', listener: (chunk: Buffer) => void): this;
492
+ on(event: 'end', listener: () => void): this;
493
+ on(event: 'error', listener: (err: Error) => void): this;
494
+ /** Speech detected (VAD enabled), at a FILE-time boundary. */
495
+ on(event: 'speech', listener: () => void): this;
496
+ /** Silence holdoff elapsed (VAD enabled), in FILE time. */
497
+ on(event: 'silence', listener: () => void): this;
498
+ on(event: string | symbol, listener: (...args: any[]) => void): this;
499
+
500
+ once(event: 'data', listener: (chunk: Buffer) => void): this;
501
+ once(event: 'end', listener: () => void): this;
502
+ once(event: 'error', listener: (err: Error) => void): this;
503
+ once(event: 'speech', listener: () => void): this;
504
+ once(event: 'silence', listener: () => void): this;
505
+ once(event: string | symbol, listener: (...args: any[]) => void): this;
506
+ }
507
+
199
508
  export interface SpeakerInfo {
200
509
  /** Device index (pass to constructor as `device`). */
201
510
  index: number;