decibri 5.2.5 → 5.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +32 -0
- package/LICENSE +191 -0
- package/README.md +1 -1
- package/examples/decibri.browser.js +1 -1
- package/index.d.ts +133 -5
- package/index.js +52 -52
- package/package.json +5 -5
- package/src/browser/decibri-browser.js +1 -1
- package/src/decibri.d.ts +267 -8
- package/src/decibri.js +330 -10
- package/src/errors.js +11 -1
package/src/decibri.d.ts
CHANGED
|
@@ -61,6 +61,108 @@ export interface VadOptions {
|
|
|
61
61
|
holdoffMs?: number;
|
|
62
62
|
}
|
|
63
63
|
|
|
64
|
+
/**
|
|
65
|
+
* Acoustic echo cancellation config object, passed on the `aec` option to tune
|
|
66
|
+
* the canceller. The bare `aec: 'tau'` shorthand selects the model with its
|
|
67
|
+
* defaults; pass this object to override them.
|
|
68
|
+
*
|
|
69
|
+
* The canceller's behaviour while a lost delay alignment is being reacquired
|
|
70
|
+
* is fixed: decibri applies the canceller's graded output transition and does
|
|
71
|
+
* not expose a setting for it.
|
|
72
|
+
*/
|
|
73
|
+
export interface AecOptions {
|
|
74
|
+
/**
|
|
75
|
+
* Which echo canceller model to run. The accepted set is owned by the
|
|
76
|
+
* canceller and grows the way `denoise` grows; today it is `'tau'`, a
|
|
77
|
+
* classical adaptive canceller with no model file. An unknown name is
|
|
78
|
+
* rejected with the canceller's own message naming the accepted set.
|
|
79
|
+
*/
|
|
80
|
+
model: 'tau';
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Adaptive filter tail length in milliseconds: how much echo delay spread
|
|
84
|
+
* the canceller can model.
|
|
85
|
+
* @default the canceller's own default (200)
|
|
86
|
+
* @range 16 to 500
|
|
87
|
+
*/
|
|
88
|
+
tailMs?: number;
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* Residual echo suppression policy. `'conservative'` attenuates the residual
|
|
92
|
+
* echo the linear canceller leaves behind while keeping the near-end voice
|
|
93
|
+
* intact; `'off'` delivers the linear canceller output as-is.
|
|
94
|
+
* @default 'conservative'
|
|
95
|
+
*/
|
|
96
|
+
suppression?: 'conservative' | 'off';
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Sample rate in Hz of the far-end reference pushed through
|
|
100
|
+
* `pushAecReference`. When it names a rate other than `sampleRate`, decibri
|
|
101
|
+
* converts the reference before the canceller sees it: a reference at an
|
|
102
|
+
* undeclared different rate cancels nothing and reports no error, so the
|
|
103
|
+
* conversion is decibri's rather than the caller's.
|
|
104
|
+
* @default the capture `sampleRate`
|
|
105
|
+
* @range 1000 to 384000
|
|
106
|
+
*/
|
|
107
|
+
referenceSampleRate?: number;
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* The echo canceller's transport and cancellation metrics, returned by
|
|
112
|
+
* `Microphone.aecMetrics()`. One object carries the canceller's own report and
|
|
113
|
+
* the reference queue's counters.
|
|
114
|
+
*/
|
|
115
|
+
export interface AecMetrics {
|
|
116
|
+
/**
|
|
117
|
+
* The active delay alignment in samples, or `null` while the estimator is
|
|
118
|
+
* still searching. Staying `null` while `acquisitionParked` climbs is the
|
|
119
|
+
* signature of a canceller with no usable reference: none pushed, not at the
|
|
120
|
+
* declared rate, or not the signal that produced the echo.
|
|
121
|
+
*/
|
|
122
|
+
delaySamples: number | null;
|
|
123
|
+
/**
|
|
124
|
+
* Smoothed echo-return-loss-enhancement estimate in dB: how much echo the
|
|
125
|
+
* canceller is currently removing. 0 before the filter has converged.
|
|
126
|
+
*/
|
|
127
|
+
erleDb: number;
|
|
128
|
+
/**
|
|
129
|
+
* Whether the double-talk detector currently believes the near-end talker is
|
|
130
|
+
* active; adaptation is held while true.
|
|
131
|
+
*/
|
|
132
|
+
doubleTalk: boolean;
|
|
133
|
+
/**
|
|
134
|
+
* Near-end samples the canceller could find no far-end sample for while an
|
|
135
|
+
* alignment was active. decibri keeps the far-end stream level with the
|
|
136
|
+
* capture, so this stays 0 for a caller who simply stops pushing; a non-zero
|
|
137
|
+
* count means the caller ran further ahead of the capture than the
|
|
138
|
+
* canceller's far-end history reaches.
|
|
139
|
+
*/
|
|
140
|
+
referenceStarved: number;
|
|
141
|
+
/**
|
|
142
|
+
* Near-end samples processed while no delay alignment was active: the
|
|
143
|
+
* searching span, not a transport failure.
|
|
144
|
+
*/
|
|
145
|
+
acquisitionParked: number;
|
|
146
|
+
/**
|
|
147
|
+
* Times the canceller inferred a capture discontinuity and rebuilt its
|
|
148
|
+
* alignment from the reference frontier.
|
|
149
|
+
*/
|
|
150
|
+
referenceReanchors: number;
|
|
151
|
+
/**
|
|
152
|
+
* Far-end samples discarded because a single push exceeded the reference
|
|
153
|
+
* queue's bound, at the declared reference rate. The span they occupied is
|
|
154
|
+
* still represented as silence, so a discard costs the cancellation of that
|
|
155
|
+
* span alone.
|
|
156
|
+
*/
|
|
157
|
+
referenceDropped: number;
|
|
158
|
+
/**
|
|
159
|
+
* Far-end samples decibri supplied as silence because the caller had pushed
|
|
160
|
+
* none for them, at the capture rate. An accounting figure, not a fault:
|
|
161
|
+
* while nothing is playing, the far end is silence.
|
|
162
|
+
*/
|
|
163
|
+
referenceSilence: number;
|
|
164
|
+
}
|
|
165
|
+
|
|
64
166
|
/** Constructor options for `Microphone`. */
|
|
65
167
|
export interface MicrophoneOptions extends ReadableOptions {
|
|
66
168
|
/**
|
|
@@ -186,6 +288,26 @@ export interface MicrophoneOptions extends ReadableOptions {
|
|
|
186
288
|
* @default undefined
|
|
187
289
|
*/
|
|
188
290
|
limiter?: number;
|
|
291
|
+
|
|
292
|
+
/**
|
|
293
|
+
* Acoustic echo cancellation applied to the captured audio, removing the
|
|
294
|
+
* echo of far-end audio the caller pushes through `pushAecReference`. The
|
|
295
|
+
* `'tau'` shorthand names the model; an `AecOptions` object tunes it. Omit
|
|
296
|
+
* to leave echo cancellation off (the default), which keeps the capture
|
|
297
|
+
* path unchanged.
|
|
298
|
+
*
|
|
299
|
+
* Runs before the detector tap: with it on, `vadScore` and the `speech` /
|
|
300
|
+
* `silence` events read the echo-removed signal, so playback stops
|
|
301
|
+
* triggering detection. Requires `sampleRate` in 8000 to 48000, narrower
|
|
302
|
+
* than the range the option otherwise accepts. With no reference pushed,
|
|
303
|
+
* the captured audio passes through unchanged.
|
|
304
|
+
*
|
|
305
|
+
* Native capture only: the browser entry does not take this option, because
|
|
306
|
+
* browser capture already carries the platform's own echo cancellation
|
|
307
|
+
* through its `echoCancellation` constraint, on by default.
|
|
308
|
+
* @default undefined
|
|
309
|
+
*/
|
|
310
|
+
aec?: 'tau' | AecOptions;
|
|
189
311
|
}
|
|
190
312
|
|
|
191
313
|
/**
|
|
@@ -245,6 +367,27 @@ export declare class Microphone extends Readable {
|
|
|
245
367
|
*/
|
|
246
368
|
readonly overrunCount: number;
|
|
247
369
|
|
|
370
|
+
/**
|
|
371
|
+
* Queue far-end reference audio for the echo canceller: the audio being
|
|
372
|
+
* played out, pushed as it is played, in played order. Accepts the same
|
|
373
|
+
* input shapes `Speaker.write` accepts (a `Buffer`, any TypedArray, or a
|
|
374
|
+
* `DataView` of PCM bytes in this microphone's `dtype`), mono, at the
|
|
375
|
+
* declared `referenceSampleRate` (the capture rate when unset).
|
|
376
|
+
*
|
|
377
|
+
* Never blocks and never throws on a full queue: samples that do not fit
|
|
378
|
+
* are discarded and counted by `aecMetrics().referenceDropped`. Silence
|
|
379
|
+
* between played audio need not be pushed. A push while capture is not
|
|
380
|
+
* running, or with the `aec` option unset, is a no-op.
|
|
381
|
+
*/
|
|
382
|
+
pushAecReference(data: Buffer | NodeJS.ArrayBufferView): void;
|
|
383
|
+
|
|
384
|
+
/**
|
|
385
|
+
* The echo canceller's transport and cancellation metrics, merged with the
|
|
386
|
+
* reference queue's counters, or `null` when the `aec` option is unset or
|
|
387
|
+
* capture is not running.
|
|
388
|
+
*/
|
|
389
|
+
aecMetrics(): AecMetrics | null;
|
|
390
|
+
|
|
248
391
|
/** List all available audio input devices. */
|
|
249
392
|
static devices(): MicrophoneInfo[];
|
|
250
393
|
|
|
@@ -280,7 +423,7 @@ export declare class Microphone extends Readable {
|
|
|
280
423
|
export interface FileOptions extends ReadableOptions {
|
|
281
424
|
/**
|
|
282
425
|
* Target output rate in Hz: the rate every delivered chunk carries. The
|
|
283
|
-
* source's input rate (from the
|
|
426
|
+
* source's input rate (from the file's header, or `inputRate` for
|
|
284
427
|
* `File.buffer`) is resampled to this rate, so a 44.1 kHz recording comes
|
|
285
428
|
* out at 16 kHz unless you set `sampleRate`. The same meaning the option
|
|
286
429
|
* has on `Microphone`.
|
|
@@ -408,6 +551,46 @@ export interface VadReport {
|
|
|
408
551
|
segments: Segment[];
|
|
409
552
|
}
|
|
410
553
|
|
|
554
|
+
/** Options for `File.save` (also accepted by `AudioWriter`). */
|
|
555
|
+
export interface SaveOptions {
|
|
556
|
+
/**
|
|
557
|
+
* The container format to write. When not given it comes from the path's
|
|
558
|
+
* extension: `.wav`, `.aiff`, `.aif`, `.aifc` or `.flac`. decibri reads a
|
|
559
|
+
* file by its content and writes one by its name; an extension it does not
|
|
560
|
+
* recognise is an error, never a silent default.
|
|
561
|
+
*/
|
|
562
|
+
format?: 'wav' | 'aiff' | 'flac';
|
|
563
|
+
|
|
564
|
+
/**
|
|
565
|
+
* FLAC compression level. Higher levels search harder for a smaller file;
|
|
566
|
+
* every level decodes to identical audio. Applies only to FLAC; ignored
|
|
567
|
+
* for WAV and AIFF.
|
|
568
|
+
* @default 5
|
|
569
|
+
* @range 0 to 8
|
|
570
|
+
*/
|
|
571
|
+
compression?: number;
|
|
572
|
+
}
|
|
573
|
+
|
|
574
|
+
/**
|
|
575
|
+
* What a save did to the samples on their way into the file, resolved by
|
|
576
|
+
* `File.save` and carried by `AudioWriter.report`.
|
|
577
|
+
*/
|
|
578
|
+
export interface SaveReport {
|
|
579
|
+
/**
|
|
580
|
+
* Finite samples outside full scale, clamped to [-1.0, 1.0]. Conditioned
|
|
581
|
+
* audio can exceed full scale (AGC or AEC without a limiter), and 16-bit
|
|
582
|
+
* PCM cannot hold that, so the overshoot clips and this count says how
|
|
583
|
+
* much. The count is a statement about integer encodings: a float encoding
|
|
584
|
+
* would preserve the overshoot instead, and would report zero.
|
|
585
|
+
*/
|
|
586
|
+
clippedSamples: number;
|
|
587
|
+
/**
|
|
588
|
+
* Non-finite samples replaced before writing: NaN with silence, an
|
|
589
|
+
* infinity with full scale. The same replacement on every format.
|
|
590
|
+
*/
|
|
591
|
+
nonFiniteSamples: number;
|
|
592
|
+
}
|
|
593
|
+
|
|
411
594
|
/**
|
|
412
595
|
* Offline audio source: conditions a recording or in-memory samples through
|
|
413
596
|
* the same chain as the live `Microphone`, delivered as a finite Readable
|
|
@@ -416,7 +599,7 @@ export interface VadReport {
|
|
|
416
599
|
* analyze the whole recording for speech with `analyze()` / `analyse()`,
|
|
417
600
|
* which a live stream cannot do.
|
|
418
601
|
*
|
|
419
|
-
* Construction: `new File(path)` reads the
|
|
602
|
+
* Construction: `new File(path)` reads the file synchronously (fine for a
|
|
420
603
|
* script; it blocks the event loop on disk I/O), `await File.open(path)` reads
|
|
421
604
|
* it off the event loop (the recommended form, mirroring `Microphone.open`),
|
|
422
605
|
* and `File.buffer(samples, { inputRate })` wraps a `Float32Array` of samples
|
|
@@ -443,15 +626,16 @@ export interface VadReport {
|
|
|
443
626
|
*/
|
|
444
627
|
export declare class File extends Readable {
|
|
445
628
|
/**
|
|
446
|
-
* Open
|
|
447
|
-
* in servers).
|
|
448
|
-
*
|
|
629
|
+
* Open an audio file synchronously (blocks on disk I/O; prefer `File.open`
|
|
630
|
+
* in servers). Reads WAV, AIFF, AIFF-C and FLAC, identified from the
|
|
631
|
+
* file's own bytes rather than its extension; the input rate and channel
|
|
632
|
+
* count come from the header.
|
|
449
633
|
*/
|
|
450
634
|
constructor(path: string, options?: FileOptions);
|
|
451
635
|
|
|
452
636
|
/**
|
|
453
|
-
* Open
|
|
454
|
-
*
|
|
637
|
+
* Open an audio file without blocking the event loop: the disk read,
|
|
638
|
+
* decode, and chain construction run on the native thread pool. The
|
|
455
639
|
* recommended form, mirroring `Microphone.open`.
|
|
456
640
|
*/
|
|
457
641
|
static open(path: string, options?: FileOptions): Promise<File>;
|
|
@@ -479,7 +663,7 @@ export declare class File extends Readable {
|
|
|
479
663
|
readonly sampleRate: number;
|
|
480
664
|
|
|
481
665
|
/**
|
|
482
|
-
* The source's own rate, taken from the
|
|
666
|
+
* The source's own rate, taken from the file's header or from the
|
|
483
667
|
* `inputRate` passed to `File.buffer`. Differs from `sampleRate` when the
|
|
484
668
|
* source was resampled.
|
|
485
669
|
*/
|
|
@@ -503,6 +687,26 @@ export declare class File extends Readable {
|
|
|
503
687
|
/** The same whole-recording analysis under the international spelling. */
|
|
504
688
|
analyse(): Promise<VadReport>;
|
|
505
689
|
|
|
690
|
+
/**
|
|
691
|
+
* Write the conditioned recording to disk, off the event loop. Runs the
|
|
692
|
+
* recording once through the same conditioning pass iteration delivers,
|
|
693
|
+
* whole, and writes it as 16-bit PCM mono at `sampleRate`. The container
|
|
694
|
+
* comes from the path's extension (`.wav`, `.aiff`, `.aif`, `.aifc` or
|
|
695
|
+
* `.flac`), or from `options.format`: decibri reads a file by its content
|
|
696
|
+
* and writes one by its name. Consumes the source (a `File` is a single
|
|
697
|
+
* pass).
|
|
698
|
+
*
|
|
699
|
+
* Resolves to a `SaveReport`: how many samples were clamped to full scale
|
|
700
|
+
* and how many non-finite samples were replaced (NaN as silence, an
|
|
701
|
+
* infinity as full scale).
|
|
702
|
+
*
|
|
703
|
+
* Requires a File that is not already being streamed: once the stream has
|
|
704
|
+
* been engaged this rejects with a `DecibriError` carrying the code
|
|
705
|
+
* `'FILE_ENGAGED'`. Every failure detected before the pass begins leaves
|
|
706
|
+
* the File usable; a failure during the pass consumes the source.
|
|
707
|
+
*/
|
|
708
|
+
save(path: string, options?: SaveOptions): Promise<SaveReport>;
|
|
709
|
+
|
|
506
710
|
/** Release the source. Idempotent; a closed File reads as ended. */
|
|
507
711
|
close(): void;
|
|
508
712
|
|
|
@@ -523,6 +727,61 @@ export declare class File extends Readable {
|
|
|
523
727
|
once(event: string | symbol, listener: (...args: any[]) => void): this;
|
|
524
728
|
}
|
|
525
729
|
|
|
730
|
+
/** Constructor options for `AudioWriter`: the save options plus the stream's own description. */
|
|
731
|
+
export interface AudioWriterOptions extends SaveOptions, WritableOptions {
|
|
732
|
+
/**
|
|
733
|
+
* The rate of the incoming samples in Hz, written into the file's header.
|
|
734
|
+
* Required: raw audio carries no header to read a rate from.
|
|
735
|
+
* @range 1000 to 384000
|
|
736
|
+
*/
|
|
737
|
+
sampleRate: number;
|
|
738
|
+
|
|
739
|
+
/**
|
|
740
|
+
* Number of channels. Audio is written mono; only `1` is accepted.
|
|
741
|
+
* @default 1
|
|
742
|
+
*/
|
|
743
|
+
channels?: 1;
|
|
744
|
+
|
|
745
|
+
/**
|
|
746
|
+
* Sample encoding of the incoming bytes.
|
|
747
|
+
* - `'int16'`: 16-bit signed integer, little-endian, what a `File` or
|
|
748
|
+
* `Microphone` emits by default
|
|
749
|
+
* - `'float32'`: 32-bit IEEE 754 float, little-endian
|
|
750
|
+
* @default 'int16'
|
|
751
|
+
*/
|
|
752
|
+
dtype?: 'int16' | 'float32';
|
|
753
|
+
}
|
|
754
|
+
|
|
755
|
+
/**
|
|
756
|
+
* A file sink for PCM audio: the Writable to pair with decibri's Readable
|
|
757
|
+
* sources, and with any other stream of PCM bytes (a TTS engine, a decoded
|
|
758
|
+
* network stream). Collects the whole stream, then writes it as one audio
|
|
759
|
+
* file when the stream finishes, exactly as `File.save` writes: the same
|
|
760
|
+
* containers from the same extension rule, the same 16-bit PCM encoding, the
|
|
761
|
+
* same clamp and non-finite handling, the same bytes.
|
|
762
|
+
*
|
|
763
|
+
* `'finish'` fires after the file is on disk, and `report` then carries the
|
|
764
|
+
* `SaveReport` the write produced. A failure destroys the stream with the
|
|
765
|
+
* error.
|
|
766
|
+
*
|
|
767
|
+
* @example
|
|
768
|
+
* const { pipeline } = require('node:stream/promises');
|
|
769
|
+
* const { File, AudioWriter } = require('decibri');
|
|
770
|
+
* await pipeline(
|
|
771
|
+
* new File('noisy.wav', { denoise: 'fastenhancer-t' }),
|
|
772
|
+
* new AudioWriter('clean.flac', { sampleRate: 16000 }),
|
|
773
|
+
* );
|
|
774
|
+
*/
|
|
775
|
+
export declare class AudioWriter extends Writable {
|
|
776
|
+
constructor(path: string, options: AudioWriterOptions);
|
|
777
|
+
|
|
778
|
+
/**
|
|
779
|
+
* The `SaveReport` of the completed write, exactly as `File.save` resolves
|
|
780
|
+
* it. `null` until `'finish'` has fired.
|
|
781
|
+
*/
|
|
782
|
+
readonly report: SaveReport | null;
|
|
783
|
+
}
|
|
784
|
+
|
|
526
785
|
export interface SpeakerInfo {
|
|
527
786
|
/** Device index (pass to constructor as `device`). */
|
|
528
787
|
index: number;
|