decibri 4.4.2 → 5.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +33 -3
- package/MIGRATION.md +70 -0
- package/README.md +73 -8
- package/examples/README.md +4 -3
- package/examples/decibri.browser.js +18 -6
- package/index.d.ts +126 -0
- package/index.js +53 -52
- package/models/README.md +149 -0
- package/models/fastenhancer_t.onnx +0 -0
- package/package.json +5 -5
- package/src/browser/decibri-browser.js +33 -13
- package/src/browser/index.d.ts +30 -18
- package/src/decibri.d.ts +326 -17
- package/src/decibri.js +573 -56
- package/src/errors.js +16 -1
package/src/decibri.d.ts
CHANGED
|
@@ -33,6 +33,34 @@ export interface VersionInfo {
|
|
|
33
33
|
binding: string;
|
|
34
34
|
}
|
|
35
35
|
|
|
36
|
+
/**
|
|
37
|
+
* Voice-activity-detection config object, passed on the `vad` option to tune
|
|
38
|
+
* the detector's threshold and holdoff. The bare `vad: 'silero'` / `vad:
|
|
39
|
+
* 'energy'` shorthand selects a mode with its default policy; pass this object
|
|
40
|
+
* to override the threshold or holdoff.
|
|
41
|
+
*/
|
|
42
|
+
export interface VadOptions {
|
|
43
|
+
/**
|
|
44
|
+
* Which detector to run.
|
|
45
|
+
* - `'silero'`: Silero VAD v5 ML model (more accurate, ~1ms inference)
|
|
46
|
+
* - `'energy'`: RMS energy threshold (lightweight, no model)
|
|
47
|
+
*/
|
|
48
|
+
model: 'silero' | 'energy';
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Speech-detection threshold for the active mode.
|
|
52
|
+
* @default 0.5 for `'silero'`, 0.01 for `'energy'`
|
|
53
|
+
* @range 0–1
|
|
54
|
+
*/
|
|
55
|
+
threshold?: number;
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Milliseconds of sub-threshold audio before emitting `'silence'`.
|
|
59
|
+
* @default 300
|
|
60
|
+
*/
|
|
61
|
+
holdoffMs?: number;
|
|
62
|
+
}
|
|
63
|
+
|
|
36
64
|
/** Constructor options for `Microphone`. */
|
|
37
65
|
export interface MicrophoneOptions extends ReadableOptions {
|
|
38
66
|
/**
|
|
@@ -43,9 +71,12 @@ export interface MicrophoneOptions extends ReadableOptions {
|
|
|
43
71
|
sampleRate?: number;
|
|
44
72
|
|
|
45
73
|
/**
|
|
46
|
-
* Number of input channels.
|
|
74
|
+
* Number of input channels. Mono only: the only accepted value is `1`, and a
|
|
75
|
+
* value greater than `1` throws a `RangeError` (multichannel capture is not
|
|
76
|
+
* supported) rather than being silently downmixed. The option is kept for
|
|
77
|
+
* forward compatibility: a future release may accept a value greater than `1`
|
|
78
|
+
* by delivering true interleaved multichannel.
|
|
47
79
|
* @default 1
|
|
48
|
-
* @range 1–32
|
|
49
80
|
*/
|
|
50
81
|
channels?: number;
|
|
51
82
|
|
|
@@ -76,36 +107,85 @@ export interface MicrophoneOptions extends ReadableOptions {
|
|
|
76
107
|
dtype?: 'int16' | 'float32';
|
|
77
108
|
|
|
78
109
|
/**
|
|
79
|
-
* Voice activity detection
|
|
110
|
+
* Voice activity detection. One of:
|
|
80
111
|
* - `false`: disabled (default)
|
|
81
112
|
* - `'silero'`: Silero VAD v5 ML model (more accurate, ~1ms inference)
|
|
82
113
|
* - `'energy'`: RMS energy threshold (lightweight)
|
|
114
|
+
* - a `VadOptions` config object `{ model, threshold?, holdoffMs? }` to tune
|
|
115
|
+
* the threshold and holdoff for the chosen model
|
|
83
116
|
*
|
|
84
|
-
*
|
|
85
|
-
*
|
|
117
|
+
* The string shorthand uses the mode's default threshold (0.5 for `'silero'`,
|
|
118
|
+
* 0.01 for `'energy'`) and a 300 ms holdoff; pass a `VadOptions` object to
|
|
119
|
+
* override them. When enabled, emits `'speech'` and `'silence'` events and
|
|
120
|
+
* updates `vadScore`. The legacy `vad: true` form is rejected; specify the
|
|
121
|
+
* mode explicitly.
|
|
86
122
|
* @default false
|
|
87
123
|
*/
|
|
88
|
-
vad?: false | 'silero' | 'energy';
|
|
124
|
+
vad?: false | 'silero' | 'energy' | VadOptions;
|
|
89
125
|
|
|
90
126
|
/**
|
|
91
|
-
*
|
|
92
|
-
*
|
|
93
|
-
*
|
|
127
|
+
* Path to the Silero VAD ONNX model file.
|
|
128
|
+
* Only used when `vad` is `'silero'`.
|
|
129
|
+
* Defaults to `models/silero_vad.onnx` relative to the package.
|
|
94
130
|
*/
|
|
95
|
-
|
|
131
|
+
modelPath?: string;
|
|
96
132
|
|
|
97
133
|
/**
|
|
98
|
-
*
|
|
99
|
-
*
|
|
134
|
+
* Remove a constant (DC) offset from the captured audio with a one-pole
|
|
135
|
+
* DC-blocking high-pass. Set `true` to enable it; omit or set `false` to
|
|
136
|
+
* leave it off (the default), which keeps the capture path byte-identical.
|
|
137
|
+
* Runs first in the chain, before denoise, and is same-length with no added
|
|
138
|
+
* latency, so `vadScore` and the `speech` / `silence` events are unaffected.
|
|
139
|
+
* Pure DSP: no bundled file or download is needed.
|
|
140
|
+
* @default undefined
|
|
100
141
|
*/
|
|
101
|
-
|
|
142
|
+
dcRemoval?: boolean;
|
|
102
143
|
|
|
103
144
|
/**
|
|
104
|
-
*
|
|
105
|
-
*
|
|
106
|
-
*
|
|
145
|
+
* Single-channel speech enhancement (denoise) model applied to the captured
|
|
146
|
+
* audio. The only accepted value is `'fastenhancer-t'`; omit to leave denoise
|
|
147
|
+
* off (the default), which keeps the capture path unchanged. The bundled
|
|
148
|
+
* model ships with the package; no path is required.
|
|
149
|
+
*
|
|
150
|
+
* When set, the captured audio is denoised before delivery and the `'data'`
|
|
151
|
+
* chunks carry the enhanced signal. VAD reads the pre-enhancement signal, so
|
|
152
|
+
* `vadScore` and the `speech` / `silence` events are unaffected.
|
|
153
|
+
* @default undefined
|
|
107
154
|
*/
|
|
108
|
-
|
|
155
|
+
denoise?: 'fastenhancer-t';
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* High-pass filter cutoff in Hz applied to the captured audio, removing
|
|
159
|
+
* low-frequency rumble below the voice band. The accepted values are `80` (an
|
|
160
|
+
* 80 Hz second-order Butterworth high-pass) and `100` (a 100 Hz one); omit to
|
|
161
|
+
* leave the high-pass off (the default), which keeps the capture path
|
|
162
|
+
* full-range. Runs after denoise in the chain. The closed value set is
|
|
163
|
+
* designed to grow (further cutoffs are additive) the way `denoise` grows.
|
|
164
|
+
* Out-of-set values raise a `RangeError`.
|
|
165
|
+
* @default undefined
|
|
166
|
+
*/
|
|
167
|
+
highpass?: 80 | 100;
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* Automatic gain control target level in dBFS applied to the captured audio.
|
|
171
|
+
* Drives the running level toward this target with a smoothed, rate-limited
|
|
172
|
+
* gain. An integer in the range -40 to -3 (typical -18); omit to leave AGC
|
|
173
|
+
* off (the default), which keeps the level untouched. Runs after the
|
|
174
|
+
* high-pass step. Out-of-range values raise a `RangeError`.
|
|
175
|
+
* @default undefined
|
|
176
|
+
*/
|
|
177
|
+
agc?: number;
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* Peak limiter ceiling in dBFS (sample-peak) applied to the captured audio.
|
|
181
|
+
* Holds the signal at or below this ceiling, the safety net that catches a
|
|
182
|
+
* transient the AGC's gain would let exceed full scale. A number in the range
|
|
183
|
+
* -3.0 to 0.0 (typical -1.0); omit to leave the limiter off (the default),
|
|
184
|
+
* which keeps the level untouched. Runs last in the chain, after the AGC step.
|
|
185
|
+
* Out-of-range values raise a `RangeError`.
|
|
186
|
+
* @default undefined
|
|
187
|
+
*/
|
|
188
|
+
limiter?: number;
|
|
109
189
|
}
|
|
110
190
|
|
|
111
191
|
/**
|
|
@@ -196,6 +276,235 @@ export declare class Microphone extends Readable {
|
|
|
196
276
|
}
|
|
197
277
|
|
|
198
278
|
/** Information about an available audio output device. */
|
|
279
|
+
/** Constructor options for `File`. */
|
|
280
|
+
export interface FileOptions extends ReadableOptions {
|
|
281
|
+
/**
|
|
282
|
+
* Target output rate in Hz: the rate every delivered chunk carries. The
|
|
283
|
+
* source's input rate (from the WAV header, or `inputRate` for
|
|
284
|
+
* `File.buffer`) is resampled to this rate, so a 44.1 kHz recording comes
|
|
285
|
+
* out at 16 kHz unless you set `sampleRate`. The same meaning the option
|
|
286
|
+
* has on `Microphone`.
|
|
287
|
+
* @default 16000
|
|
288
|
+
* @range 1000–384000
|
|
289
|
+
*/
|
|
290
|
+
sampleRate?: number;
|
|
291
|
+
|
|
292
|
+
/**
|
|
293
|
+
* Sample encoding data type of the delivered chunks.
|
|
294
|
+
* - `'int16'`: 16-bit signed integer, little-endian (2 bytes per sample)
|
|
295
|
+
* - `'float32'`: 32-bit IEEE 754 float, little-endian (4 bytes per sample)
|
|
296
|
+
* @default 'int16'
|
|
297
|
+
*/
|
|
298
|
+
dtype?: 'int16' | 'float32';
|
|
299
|
+
|
|
300
|
+
/**
|
|
301
|
+
* Voice activity detection, opt-in exactly as on `Microphone`: `false`
|
|
302
|
+
* (default), `'silero'`, `'energy'`, or a `VadOptions` config object.
|
|
303
|
+
* When enabled, per-chunk detection runs alongside the stream (the
|
|
304
|
+
* `'speech'` / `'silence'` events and `vadScore`, with the holdoff
|
|
305
|
+
* measured in FILE time rather than wall-clock time) and `'silero'`
|
|
306
|
+
* additionally enables the whole-file `analyze()`. With no `vad` set the
|
|
307
|
+
* File simply conditions audio: no scores, no segments, no speech events.
|
|
308
|
+
* @default false
|
|
309
|
+
*/
|
|
310
|
+
vad?: false | 'silero' | 'energy' | VadOptions;
|
|
311
|
+
|
|
312
|
+
/**
|
|
313
|
+
* Path to the Silero VAD ONNX model file.
|
|
314
|
+
* Only used when `vad` is `'silero'`.
|
|
315
|
+
* Defaults to `models/silero_vad.onnx` relative to the package.
|
|
316
|
+
*/
|
|
317
|
+
modelPath?: string;
|
|
318
|
+
|
|
319
|
+
/**
|
|
320
|
+
* Remove a constant (DC) offset with a one-pole DC-blocking high-pass,
|
|
321
|
+
* exactly as on `Microphone`.
|
|
322
|
+
* @default undefined
|
|
323
|
+
*/
|
|
324
|
+
dcRemoval?: boolean;
|
|
325
|
+
|
|
326
|
+
/**
|
|
327
|
+
* Single-channel speech enhancement (denoise) model, exactly as on
|
|
328
|
+
* `Microphone`. The only accepted value is `'fastenhancer-t'`.
|
|
329
|
+
* @default undefined
|
|
330
|
+
*/
|
|
331
|
+
denoise?: 'fastenhancer-t';
|
|
332
|
+
|
|
333
|
+
/**
|
|
334
|
+
* High-pass filter cutoff in Hz (`80` or `100`), exactly as on
|
|
335
|
+
* `Microphone`.
|
|
336
|
+
* @default undefined
|
|
337
|
+
*/
|
|
338
|
+
highpass?: 80 | 100;
|
|
339
|
+
|
|
340
|
+
/**
|
|
341
|
+
* Automatic gain control target level in dBFS, exactly as on `Microphone`.
|
|
342
|
+
* @default undefined
|
|
343
|
+
* @range -40 to -3
|
|
344
|
+
*/
|
|
345
|
+
agc?: number;
|
|
346
|
+
|
|
347
|
+
/**
|
|
348
|
+
* Peak limiter ceiling in dBFS (sample-peak), exactly as on `Microphone`.
|
|
349
|
+
* @default undefined
|
|
350
|
+
* @range -3.0 to 0.0
|
|
351
|
+
*/
|
|
352
|
+
limiter?: number;
|
|
353
|
+
}
|
|
354
|
+
|
|
355
|
+
/** Options for `File.buffer`: `FileOptions` plus the samples' native rate. */
|
|
356
|
+
export interface FileBufferOptions extends FileOptions {
|
|
357
|
+
/**
|
|
358
|
+
* The native rate of the in-memory samples in Hz. Required: raw samples
|
|
359
|
+
* carry no header to read a rate from. The samples are resampled from this
|
|
360
|
+
* rate to `sampleRate`.
|
|
361
|
+
* @range 1000–384000
|
|
362
|
+
*/
|
|
363
|
+
inputRate: number;
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
/**
|
|
367
|
+
* One scored voice-activity window of a recording, produced by
|
|
368
|
+
* `File.analyze()`. Windows tile the recording from the start in fixed steps
|
|
369
|
+
* (512 samples at 16 kHz, 32 ms per window); a trailing remainder shorter
|
|
370
|
+
* than one window is not scored, exactly as live detection leaves a
|
|
371
|
+
* sub-window remainder unscored.
|
|
372
|
+
*/
|
|
373
|
+
export interface VadWindow {
|
|
374
|
+
/** Window start, in seconds of file time. */
|
|
375
|
+
start: number;
|
|
376
|
+
/** Window end, in seconds of file time. */
|
|
377
|
+
end: number;
|
|
378
|
+
/**
|
|
379
|
+
* Speech probability for this window (0 to 1). The same quantity the live
|
|
380
|
+
* per-chunk `vadScore` reports, here per window across the whole recording.
|
|
381
|
+
*/
|
|
382
|
+
vadScore: number;
|
|
383
|
+
/**
|
|
384
|
+
* Whether `vadScore` meets the configured threshold. The raw per-window
|
|
385
|
+
* test, not the debounced speaking state.
|
|
386
|
+
*/
|
|
387
|
+
isSpeech: boolean;
|
|
388
|
+
}
|
|
389
|
+
|
|
390
|
+
/**
|
|
391
|
+
* One merged speech region of a recording, produced by `File.analyze()`:
|
|
392
|
+
* consecutive speech windows whose silence gaps are within the configured
|
|
393
|
+
* holdoff collapse into one segment. The segment ends at the last speech
|
|
394
|
+
* window, not at the holdoff expiry.
|
|
395
|
+
*/
|
|
396
|
+
export interface Segment {
|
|
397
|
+
/** Region start, in seconds of file time. */
|
|
398
|
+
start: number;
|
|
399
|
+
/** Region end, in seconds of file time. */
|
|
400
|
+
end: number;
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
/** The whole-recording voice-activity analysis `File.analyze()` resolves to. */
|
|
404
|
+
export interface VadReport {
|
|
405
|
+
/** Per-window speech scores across the whole recording, in file order. */
|
|
406
|
+
scores: VadWindow[];
|
|
407
|
+
/** Merged speech regions across the whole recording, in file order. */
|
|
408
|
+
segments: Segment[];
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
/**
|
|
412
|
+
* Offline audio source: conditions a recording or in-memory samples through
|
|
413
|
+
* the same chain as the live `Microphone`, delivered as a finite Readable
|
|
414
|
+
* stream of conditioned chunks that ends at EOF (after the chain's
|
|
415
|
+
* end-of-stream tail). Because a `File` is a complete recording, it can also
|
|
416
|
+
* analyze the whole recording for speech with `analyze()` / `analyse()`,
|
|
417
|
+
* which a live stream cannot do.
|
|
418
|
+
*
|
|
419
|
+
* Construction: `new File(path)` reads the WAV synchronously (fine for a
|
|
420
|
+
* script; it blocks the event loop on disk I/O), `await File.open(path)` reads
|
|
421
|
+
* it off the event loop (the recommended form, mirroring `Microphone.open`),
|
|
422
|
+
* and `File.buffer(samples, { inputRate })` wraps a `Float32Array` of samples
|
|
423
|
+
* you already hold (a raw `Buffer` of bytes is rejected as ambiguous).
|
|
424
|
+
*
|
|
425
|
+
* Iteration and analysis are separate single passes: each consumes the source
|
|
426
|
+
* once, so construct one `File` per operation.
|
|
427
|
+
*
|
|
428
|
+
* Note: Node also has a global `File` (the web File API). Import decibri's
|
|
429
|
+
* explicitly (`const { File } = require('decibri')`) or reference it as
|
|
430
|
+
* `decibri.File` to avoid shadowing surprises.
|
|
431
|
+
*
|
|
432
|
+
* @example
|
|
433
|
+
* const { File } = require('decibri');
|
|
434
|
+
* const file = await File.open('clip.wav', { denoise: 'fastenhancer-t' });
|
|
435
|
+
* file.on('data', (chunk) => { /* Buffer of conditioned Int16 PCM *\/ });
|
|
436
|
+
* file.on('end', () => console.log('done'));
|
|
437
|
+
*
|
|
438
|
+
* @example
|
|
439
|
+
* // Where is the speech?
|
|
440
|
+
* const f = await File.open('clip.wav', { vad: 'silero' });
|
|
441
|
+
* const report = await f.analyze();
|
|
442
|
+
* for (const s of report.segments) console.log(s.start, s.end);
|
|
443
|
+
*/
|
|
444
|
+
export declare class File extends Readable {
|
|
445
|
+
/**
|
|
446
|
+
* Open a WAV file synchronously (blocks on disk I/O; prefer `File.open`
|
|
447
|
+
* in servers). Supports 16-bit PCM and 32-bit float WAV files; the input
|
|
448
|
+
* rate and channel count come from the header.
|
|
449
|
+
*/
|
|
450
|
+
constructor(path: string, options?: FileOptions);
|
|
451
|
+
|
|
452
|
+
/**
|
|
453
|
+
* Open a WAV file without blocking the event loop: the disk read, WAV
|
|
454
|
+
* parse, and chain construction run on the native thread pool. The
|
|
455
|
+
* recommended form, mirroring `Microphone.open`.
|
|
456
|
+
*/
|
|
457
|
+
static open(path: string, options?: FileOptions): Promise<File>;
|
|
458
|
+
|
|
459
|
+
/**
|
|
460
|
+
* Wrap in-memory samples as an offline source. `samples` must be a
|
|
461
|
+
* `Float32Array` of mono samples in [-1.0, 1.0]; a raw `Buffer` of PCM
|
|
462
|
+
* bytes is rejected as ambiguous. `inputRate` is required (raw samples
|
|
463
|
+
* carry no header). Synchronous: no I/O is involved.
|
|
464
|
+
*/
|
|
465
|
+
static buffer(samples: Float32Array, options: FileBufferOptions): File;
|
|
466
|
+
|
|
467
|
+
/**
|
|
468
|
+
* Most recent per-chunk VAD score for the active mode: the Silero speech
|
|
469
|
+
* probability in `'silero'` mode, the normalized RMS of the
|
|
470
|
+
* pre-conditioning signal in `'energy'` mode. 0 when VAD is disabled or
|
|
471
|
+
* before the first chunk.
|
|
472
|
+
*/
|
|
473
|
+
readonly vadScore: number;
|
|
474
|
+
|
|
475
|
+
/**
|
|
476
|
+
* Analyze the whole recording for speech, off the event loop. Resolves to
|
|
477
|
+
* a `VadReport` with per-window `scores` and merged speech `segments`,
|
|
478
|
+
* all in seconds of file time. Consumes the source (a `File` is a single
|
|
479
|
+
* pass). Requires `vad: 'silero'`: a File opened without `vad` rejects
|
|
480
|
+
* with the core's "analysis requires VAD" error, and the energy mode has
|
|
481
|
+
* no whole-file analysis.
|
|
482
|
+
*/
|
|
483
|
+
analyze(): Promise<VadReport>;
|
|
484
|
+
|
|
485
|
+
/** The same whole-recording analysis under the international spelling. */
|
|
486
|
+
analyse(): Promise<VadReport>;
|
|
487
|
+
|
|
488
|
+
/** Release the source. Idempotent; a closed File reads as ended. */
|
|
489
|
+
close(): void;
|
|
490
|
+
|
|
491
|
+
on(event: 'data', listener: (chunk: Buffer) => void): this;
|
|
492
|
+
on(event: 'end', listener: () => void): this;
|
|
493
|
+
on(event: 'error', listener: (err: Error) => void): this;
|
|
494
|
+
/** Speech detected (VAD enabled), at a FILE-time boundary. */
|
|
495
|
+
on(event: 'speech', listener: () => void): this;
|
|
496
|
+
/** Silence holdoff elapsed (VAD enabled), in FILE time. */
|
|
497
|
+
on(event: 'silence', listener: () => void): this;
|
|
498
|
+
on(event: string | symbol, listener: (...args: any[]) => void): this;
|
|
499
|
+
|
|
500
|
+
once(event: 'data', listener: (chunk: Buffer) => void): this;
|
|
501
|
+
once(event: 'end', listener: () => void): this;
|
|
502
|
+
once(event: 'error', listener: (err: Error) => void): this;
|
|
503
|
+
once(event: 'speech', listener: () => void): this;
|
|
504
|
+
once(event: 'silence', listener: () => void): this;
|
|
505
|
+
once(event: string | symbol, listener: (...args: any[]) => void): this;
|
|
506
|
+
}
|
|
507
|
+
|
|
199
508
|
export interface SpeakerInfo {
|
|
200
509
|
/** Device index (pass to constructor as `device`). */
|
|
201
510
|
index: number;
|