decibri 5.3.0 → 5.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +48 -0
- package/LICENSE +191 -0
- package/README.md +11 -3
- package/examples/decibri.browser.js +3 -1
- package/index.d.ts +45 -5
- package/index.js +52 -52
- package/models/README.md +21 -103
- package/models/THIRD-PARTY-NOTICES.md +110 -0
- package/package.json +5 -5
- package/src/browser/decibri-browser.js +14 -1
- package/src/browser/decibri-output-browser.js +6 -0
- package/src/browser/index.d.ts +5 -2
- package/src/decibri-output.js +6 -2
- package/src/decibri.d.ts +164 -14
- package/src/decibri.js +228 -17
- package/src/errors.js +9 -2
package/src/decibri-output.js
CHANGED
|
@@ -67,9 +67,13 @@ class Speaker extends Writable {
|
|
|
67
67
|
throw new RangeError('sample rate must be between 1000 and 384000');
|
|
68
68
|
}
|
|
69
69
|
|
|
70
|
+
// Bounded below only. How many output channels can be carried is the
|
|
71
|
+
// device's answer, so any count above zero is passed through and a device
|
|
72
|
+
// that refuses it throws a DecibriError with code
|
|
73
|
+
// 'SPEAKER_CHANNELS_UNSUPPORTED' naming the count the device reports.
|
|
70
74
|
const channels = options.channels ?? 1;
|
|
71
|
-
if (channels < 1
|
|
72
|
-
throw new RangeError('channels must be
|
|
75
|
+
if (channels < 1) {
|
|
76
|
+
throw new RangeError('channels must be at least 1');
|
|
73
77
|
}
|
|
74
78
|
|
|
75
79
|
const dtype = options.dtype ?? 'int16';
|
package/src/decibri.d.ts
CHANGED
|
@@ -42,7 +42,7 @@ export interface VersionInfo {
|
|
|
42
42
|
export interface VadOptions {
|
|
43
43
|
/**
|
|
44
44
|
* Which detector to run.
|
|
45
|
-
* - `'silero'`: Silero VAD
|
|
45
|
+
* - `'silero'`: Silero VAD v6.2 ML model (more accurate, ~1ms inference)
|
|
46
46
|
* - `'energy'`: RMS energy threshold (lightweight, no model)
|
|
47
47
|
*/
|
|
48
48
|
model: 'silero' | 'energy';
|
|
@@ -105,6 +105,32 @@ export interface AecOptions {
|
|
|
105
105
|
* @range 1000 to 384000
|
|
106
106
|
*/
|
|
107
107
|
referenceSampleRate?: number;
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Number of channels in the far-end reference pushed through
|
|
111
|
+
* `pushAecReference`, frame-interleaved. When it names a count above 1,
|
|
112
|
+
* decibri averages each frame to one mono sample before the canceller sees
|
|
113
|
+
* it: a multichannel reference pushed without declaring the count cancels
|
|
114
|
+
* nothing and reports no error, so the collapse is decibri's rather than
|
|
115
|
+
* the caller's.
|
|
116
|
+
*
|
|
117
|
+
* The declared count must match the buffer actually pushed. The reference
|
|
118
|
+
* arrives as flat PCM whose true channel count is not recoverable from its
|
|
119
|
+
* length, so a mismatch is not detected and raises no error: the frames
|
|
120
|
+
* are misread, nothing is cancelled, and the observable signature is
|
|
121
|
+
* `aecMetrics().delaySamples` staying `null` while the canceller reports
|
|
122
|
+
* no fault.
|
|
123
|
+
*
|
|
124
|
+
* The canceller itself reads one mono reference. Against playback through
|
|
125
|
+
* more than one loudspeaker that is a cancellation ceiling: the echo
|
|
126
|
+
* reaching the microphone is the sum of different room responses driven by
|
|
127
|
+
* different signals, and a single-reference canceller models one response
|
|
128
|
+
* applied to their average, so a placement where those paths differ leaves
|
|
129
|
+
* a residual that no amount of adaptation removes.
|
|
130
|
+
* @default 1 (mono)
|
|
131
|
+
* @range at least 1; no upper bound
|
|
132
|
+
*/
|
|
133
|
+
referenceChannels?: number;
|
|
108
134
|
}
|
|
109
135
|
|
|
110
136
|
/**
|
|
@@ -211,7 +237,7 @@ export interface MicrophoneOptions extends ReadableOptions {
|
|
|
211
237
|
/**
|
|
212
238
|
* Voice activity detection. One of:
|
|
213
239
|
* - `false`: disabled (default)
|
|
214
|
-
* - `'silero'`: Silero VAD
|
|
240
|
+
* - `'silero'`: Silero VAD v6.2 ML model (more accurate, ~1ms inference)
|
|
215
241
|
* - `'energy'`: RMS energy threshold (lightweight)
|
|
216
242
|
* - a `VadOptions` config object `{ model, threshold?, holdoffMs? }` to tune
|
|
217
243
|
* the threshold and holdoff for the chosen model
|
|
@@ -371,8 +397,14 @@ export declare class Microphone extends Readable {
|
|
|
371
397
|
* Queue far-end reference audio for the echo canceller: the audio being
|
|
372
398
|
* played out, pushed as it is played, in played order. Accepts the same
|
|
373
399
|
* input shapes `Speaker.write` accepts (a `Buffer`, any TypedArray, or a
|
|
374
|
-
* `DataView` of PCM bytes in this microphone's `dtype`),
|
|
375
|
-
*
|
|
400
|
+
* `DataView` of PCM bytes in this microphone's `dtype`), at the declared
|
|
401
|
+
* `referenceSampleRate` (the capture rate when unset), interleaved at the
|
|
402
|
+
* declared `referenceChannels` (mono when unset). With `referenceChannels`
|
|
403
|
+
* above 1, each frame is averaged to one mono sample before the canceller
|
|
404
|
+
* sees it. The declared count must match this buffer's actual
|
|
405
|
+
* interleaving: a mismatch is not detected and raises no error, and shows
|
|
406
|
+
* up only as `aecMetrics().delaySamples` staying `null` with no fault
|
|
407
|
+
* reported.
|
|
376
408
|
*
|
|
377
409
|
* Never blocks and never throws on a full queue: samples that do not fit
|
|
378
410
|
* are discarded and counted by `aecMetrics().referenceDropped`. Silence
|
|
@@ -423,7 +455,7 @@ export declare class Microphone extends Readable {
|
|
|
423
455
|
export interface FileOptions extends ReadableOptions {
|
|
424
456
|
/**
|
|
425
457
|
* Target output rate in Hz: the rate every delivered chunk carries. The
|
|
426
|
-
* source's input rate (from the
|
|
458
|
+
* source's input rate (from the file's header, or `inputRate` for
|
|
427
459
|
* `File.buffer`) is resampled to this rate, so a 44.1 kHz recording comes
|
|
428
460
|
* out at 16 kHz unless you set `sampleRate`. The same meaning the option
|
|
429
461
|
* has on `Microphone`.
|
|
@@ -551,6 +583,46 @@ export interface VadReport {
|
|
|
551
583
|
segments: Segment[];
|
|
552
584
|
}
|
|
553
585
|
|
|
586
|
+
/** Options for `File.save` (also accepted by `AudioWriter`). */
|
|
587
|
+
export interface SaveOptions {
|
|
588
|
+
/**
|
|
589
|
+
* The container format to write. When not given it comes from the path's
|
|
590
|
+
* extension: `.wav`, `.aiff`, `.aif`, `.aifc` or `.flac`. decibri reads a
|
|
591
|
+
* file by its content and writes one by its name; an extension it does not
|
|
592
|
+
* recognise is an error, never a silent default.
|
|
593
|
+
*/
|
|
594
|
+
format?: 'wav' | 'aiff' | 'flac';
|
|
595
|
+
|
|
596
|
+
/**
|
|
597
|
+
* FLAC compression level. Higher levels search harder for a smaller file;
|
|
598
|
+
* every level decodes to identical audio. Applies only to FLAC; ignored
|
|
599
|
+
* for WAV and AIFF.
|
|
600
|
+
* @default 5
|
|
601
|
+
* @range 0 to 8
|
|
602
|
+
*/
|
|
603
|
+
compression?: number;
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
/**
|
|
607
|
+
* What a save did to the samples on their way into the file, resolved by
|
|
608
|
+
* `File.save` and carried by `AudioWriter.report`.
|
|
609
|
+
*/
|
|
610
|
+
export interface SaveReport {
|
|
611
|
+
/**
|
|
612
|
+
* Finite samples outside full scale, clamped to [-1.0, 1.0]. Conditioned
|
|
613
|
+
* audio can exceed full scale (AGC or AEC without a limiter), and 16-bit
|
|
614
|
+
* PCM cannot hold that, so the overshoot clips and this count says how
|
|
615
|
+
* much. The count is a statement about integer encodings: a float encoding
|
|
616
|
+
* would preserve the overshoot instead, and would report zero.
|
|
617
|
+
*/
|
|
618
|
+
clippedSamples: number;
|
|
619
|
+
/**
|
|
620
|
+
* Non-finite samples replaced before writing: NaN with silence, an
|
|
621
|
+
* infinity with full scale. The same replacement on every format.
|
|
622
|
+
*/
|
|
623
|
+
nonFiniteSamples: number;
|
|
624
|
+
}
|
|
625
|
+
|
|
554
626
|
/**
|
|
555
627
|
* Offline audio source: conditions a recording or in-memory samples through
|
|
556
628
|
* the same chain as the live `Microphone`, delivered as a finite Readable
|
|
@@ -559,7 +631,7 @@ export interface VadReport {
|
|
|
559
631
|
* analyze the whole recording for speech with `analyze()` / `analyse()`,
|
|
560
632
|
* which a live stream cannot do.
|
|
561
633
|
*
|
|
562
|
-
* Construction: `new File(path)` reads the
|
|
634
|
+
* Construction: `new File(path)` reads the file synchronously (fine for a
|
|
563
635
|
* script; it blocks the event loop on disk I/O), `await File.open(path)` reads
|
|
564
636
|
* it off the event loop (the recommended form, mirroring `Microphone.open`),
|
|
565
637
|
* and `File.buffer(samples, { inputRate })` wraps a `Float32Array` of samples
|
|
@@ -586,15 +658,16 @@ export interface VadReport {
|
|
|
586
658
|
*/
|
|
587
659
|
export declare class File extends Readable {
|
|
588
660
|
/**
|
|
589
|
-
* Open
|
|
590
|
-
* in servers).
|
|
591
|
-
*
|
|
661
|
+
* Open an audio file synchronously (blocks on disk I/O; prefer `File.open`
|
|
662
|
+
* in servers). Reads WAV, AIFF, AIFF-C and FLAC, identified from the
|
|
663
|
+
* file's own bytes rather than its extension; the input rate and channel
|
|
664
|
+
* count come from the header.
|
|
592
665
|
*/
|
|
593
666
|
constructor(path: string, options?: FileOptions);
|
|
594
667
|
|
|
595
668
|
/**
|
|
596
|
-
* Open
|
|
597
|
-
*
|
|
669
|
+
* Open an audio file without blocking the event loop: the disk read,
|
|
670
|
+
* decode, and chain construction run on the native thread pool. The
|
|
598
671
|
* recommended form, mirroring `Microphone.open`.
|
|
599
672
|
*/
|
|
600
673
|
static open(path: string, options?: FileOptions): Promise<File>;
|
|
@@ -622,7 +695,7 @@ export declare class File extends Readable {
|
|
|
622
695
|
readonly sampleRate: number;
|
|
623
696
|
|
|
624
697
|
/**
|
|
625
|
-
* The source's own rate, taken from the
|
|
698
|
+
* The source's own rate, taken from the file's header or from the
|
|
626
699
|
* `inputRate` passed to `File.buffer`. Differs from `sampleRate` when the
|
|
627
700
|
* source was resampled.
|
|
628
701
|
*/
|
|
@@ -646,6 +719,26 @@ export declare class File extends Readable {
|
|
|
646
719
|
/** The same whole-recording analysis under the international spelling. */
|
|
647
720
|
analyse(): Promise<VadReport>;
|
|
648
721
|
|
|
722
|
+
/**
|
|
723
|
+
* Write the conditioned recording to disk, off the event loop. Runs the
|
|
724
|
+
* recording once through the same conditioning pass iteration delivers,
|
|
725
|
+
* whole, and writes it as 16-bit PCM mono at `sampleRate`. The container
|
|
726
|
+
* comes from the path's extension (`.wav`, `.aiff`, `.aif`, `.aifc` or
|
|
727
|
+
* `.flac`), or from `options.format`: decibri reads a file by its content
|
|
728
|
+
* and writes one by its name. Consumes the source (a `File` is a single
|
|
729
|
+
* pass).
|
|
730
|
+
*
|
|
731
|
+
* Resolves to a `SaveReport`: how many samples were clamped to full scale
|
|
732
|
+
* and how many non-finite samples were replaced (NaN as silence, an
|
|
733
|
+
* infinity as full scale).
|
|
734
|
+
*
|
|
735
|
+
* Requires a File that is not already being streamed: once the stream has
|
|
736
|
+
* been engaged this rejects with a `DecibriError` carrying the code
|
|
737
|
+
* `'FILE_ENGAGED'`. Every failure detected before the pass begins leaves
|
|
738
|
+
* the File usable; a failure during the pass consumes the source.
|
|
739
|
+
*/
|
|
740
|
+
save(path: string, options?: SaveOptions): Promise<SaveReport>;
|
|
741
|
+
|
|
649
742
|
/** Release the source. Idempotent; a closed File reads as ended. */
|
|
650
743
|
close(): void;
|
|
651
744
|
|
|
@@ -666,6 +759,61 @@ export declare class File extends Readable {
|
|
|
666
759
|
once(event: string | symbol, listener: (...args: any[]) => void): this;
|
|
667
760
|
}
|
|
668
761
|
|
|
762
|
+
/** Constructor options for `AudioWriter`: the save options plus the stream's own description. */
|
|
763
|
+
export interface AudioWriterOptions extends SaveOptions, WritableOptions {
|
|
764
|
+
/**
|
|
765
|
+
* The rate of the incoming samples in Hz, written into the file's header.
|
|
766
|
+
* Required: raw audio carries no header to read a rate from.
|
|
767
|
+
* @range 1000 to 384000
|
|
768
|
+
*/
|
|
769
|
+
sampleRate: number;
|
|
770
|
+
|
|
771
|
+
/**
|
|
772
|
+
* Number of channels. Audio is written mono; only `1` is accepted.
|
|
773
|
+
* @default 1
|
|
774
|
+
*/
|
|
775
|
+
channels?: 1;
|
|
776
|
+
|
|
777
|
+
/**
|
|
778
|
+
* Sample encoding of the incoming bytes.
|
|
779
|
+
* - `'int16'`: 16-bit signed integer, little-endian, what a `File` or
|
|
780
|
+
* `Microphone` emits by default
|
|
781
|
+
* - `'float32'`: 32-bit IEEE 754 float, little-endian
|
|
782
|
+
* @default 'int16'
|
|
783
|
+
*/
|
|
784
|
+
dtype?: 'int16' | 'float32';
|
|
785
|
+
}
|
|
786
|
+
|
|
787
|
+
/**
|
|
788
|
+
* A file sink for PCM audio: the Writable to pair with decibri's Readable
|
|
789
|
+
* sources, and with any other stream of PCM bytes (a TTS engine, a decoded
|
|
790
|
+
* network stream). Collects the whole stream, then writes it as one audio
|
|
791
|
+
* file when the stream finishes, exactly as `File.save` writes: the same
|
|
792
|
+
* containers from the same extension rule, the same 16-bit PCM encoding, the
|
|
793
|
+
* same clamp and non-finite handling, the same bytes.
|
|
794
|
+
*
|
|
795
|
+
* `'finish'` fires after the file is on disk, and `report` then carries the
|
|
796
|
+
* `SaveReport` the write produced. A failure destroys the stream with the
|
|
797
|
+
* error.
|
|
798
|
+
*
|
|
799
|
+
* @example
|
|
800
|
+
* const { pipeline } = require('node:stream/promises');
|
|
801
|
+
* const { File, AudioWriter } = require('decibri');
|
|
802
|
+
* await pipeline(
|
|
803
|
+
* new File('noisy.wav', { denoise: 'fastenhancer-t' }),
|
|
804
|
+
* new AudioWriter('clean.flac', { sampleRate: 16000 }),
|
|
805
|
+
* );
|
|
806
|
+
*/
|
|
807
|
+
export declare class AudioWriter extends Writable {
|
|
808
|
+
constructor(path: string, options: AudioWriterOptions);
|
|
809
|
+
|
|
810
|
+
/**
|
|
811
|
+
* The `SaveReport` of the completed write, exactly as `File.save` resolves
|
|
812
|
+
* it. `null` until `'finish'` has fired.
|
|
813
|
+
*/
|
|
814
|
+
readonly report: SaveReport | null;
|
|
815
|
+
}
|
|
816
|
+
|
|
669
817
|
export interface SpeakerInfo {
|
|
670
818
|
/** Device index (pass to constructor as `device`). */
|
|
671
819
|
index: number;
|
|
@@ -694,9 +842,11 @@ export interface SpeakerOptions extends WritableOptions {
|
|
|
694
842
|
sampleRate?: number;
|
|
695
843
|
|
|
696
844
|
/**
|
|
697
|
-
* Number of output channels.
|
|
845
|
+
* Number of output channels. The maximum is the device's: a count the device
|
|
846
|
+
* cannot serve throws a `DecibriError` with code
|
|
847
|
+
* `'SPEAKER_CHANNELS_UNSUPPORTED'` naming the count the device reports.
|
|
698
848
|
* @default 1
|
|
699
|
-
* @range 1
|
|
849
|
+
* @range 1 or more
|
|
700
850
|
*/
|
|
701
851
|
channels?: number;
|
|
702
852
|
|
package/src/decibri.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
'use strict';
|
|
2
2
|
|
|
3
|
-
const { Readable } = require('stream');
|
|
3
|
+
const { Readable, Writable } = require('stream');
|
|
4
4
|
const path = require('path');
|
|
5
5
|
const fs = require('fs');
|
|
6
6
|
const { DecibriBridge, FileHandle } = require('../index.js');
|
|
@@ -151,7 +151,7 @@ class Microphone extends Readable {
|
|
|
151
151
|
// additive. The `channels` option is kept for that forward compatibility.
|
|
152
152
|
const channels = options.channels ?? 1;
|
|
153
153
|
if (channels < 1) {
|
|
154
|
-
throw new RangeError('channels must be
|
|
154
|
+
throw new RangeError('channels must be at least 1');
|
|
155
155
|
}
|
|
156
156
|
if (channels > 1) {
|
|
157
157
|
throw new RangeError('multichannel capture is not supported; channels must be 1 (mono)');
|
|
@@ -345,8 +345,8 @@ class Microphone extends Readable {
|
|
|
345
345
|
// ── Validate AEC ─────────────────────────────────────────────────────────
|
|
346
346
|
|
|
347
347
|
// Echo cancellation: the 'tau' shorthand names the model, or an
|
|
348
|
-
// { model, tailMs, suppression, referenceSampleRate }
|
|
349
|
-
// absence leaves it off. The model name is deliberately NOT checked
|
|
348
|
+
// { model, tailMs, suppression, referenceSampleRate, referenceChannels }
|
|
349
|
+
// object tunes it; absence leaves it off. The model name is deliberately NOT checked
|
|
350
350
|
// against a list here: the canceller owns the accepted set, so the native
|
|
351
351
|
// layer parses it (AecModel::from_str) and an unknown name is rejected by
|
|
352
352
|
// the native constructor with the canceller's own message (a DecibriError
|
|
@@ -360,11 +360,12 @@ class Microphone extends Readable {
|
|
|
360
360
|
let aecTailMs;
|
|
361
361
|
let aecSuppression;
|
|
362
362
|
let aecReferenceSampleRate;
|
|
363
|
+
let aecReferenceChannels;
|
|
363
364
|
if (aec !== undefined) {
|
|
364
365
|
if (typeof aec === 'string') {
|
|
365
366
|
aecModel = aec;
|
|
366
367
|
} else if (aec !== null && typeof aec === 'object' && !Array.isArray(aec)) {
|
|
367
|
-
const { model, tailMs, suppression, referenceSampleRate } = aec;
|
|
368
|
+
const { model, tailMs, suppression, referenceSampleRate, referenceChannels } = aec;
|
|
368
369
|
if (typeof model !== 'string') {
|
|
369
370
|
throw new TypeError(
|
|
370
371
|
`Invalid aec model: ${JSON.stringify(model)}. Expected a model name string such as 'tau'.`
|
|
@@ -397,9 +398,18 @@ class Microphone extends Readable {
|
|
|
397
398
|
}
|
|
398
399
|
aecReferenceSampleRate = referenceSampleRate;
|
|
399
400
|
}
|
|
401
|
+
if (referenceChannels !== undefined) {
|
|
402
|
+
if (typeof referenceChannels !== 'number' || Number.isNaN(referenceChannels)) {
|
|
403
|
+
throw new TypeError('aec referenceChannels must be a number');
|
|
404
|
+
}
|
|
405
|
+
if (referenceChannels < 1) {
|
|
406
|
+
throw new RangeError('aec referenceChannels must be at least 1');
|
|
407
|
+
}
|
|
408
|
+
aecReferenceChannels = referenceChannels;
|
|
409
|
+
}
|
|
400
410
|
} else {
|
|
401
411
|
throw new TypeError(
|
|
402
|
-
`Invalid aec value: ${JSON.stringify(aec)}. Expected a model name such as 'tau', or a config object { model, tailMs, suppression, referenceSampleRate }.`
|
|
412
|
+
`Invalid aec value: ${JSON.stringify(aec)}. Expected a model name such as 'tau', or a config object { model, tailMs, suppression, referenceSampleRate, referenceChannels }.`
|
|
403
413
|
);
|
|
404
414
|
}
|
|
405
415
|
}
|
|
@@ -443,6 +453,7 @@ class Microphone extends Readable {
|
|
|
443
453
|
aecTailMs,
|
|
444
454
|
aecSuppression,
|
|
445
455
|
aecReferenceSampleRate,
|
|
456
|
+
aecReferenceChannels,
|
|
446
457
|
},
|
|
447
458
|
};
|
|
448
459
|
}
|
|
@@ -605,8 +616,15 @@ class Microphone extends Readable {
|
|
|
605
616
|
* Queue far-end reference audio for the echo canceller: the audio being
|
|
606
617
|
* played out, pushed as it is played, in played order. Accepts the same
|
|
607
618
|
* input shapes `Speaker.write` accepts (a `Buffer`, any TypedArray, or a
|
|
608
|
-
* `DataView` of PCM bytes in this microphone's `dtype`),
|
|
609
|
-
*
|
|
619
|
+
* `DataView` of PCM bytes in this microphone's `dtype`), at the declared
|
|
620
|
+
* `referenceSampleRate` (the capture rate when unset), interleaved at the
|
|
621
|
+
* declared `referenceChannels` (mono when unset). With `referenceChannels`
|
|
622
|
+
* above 1, each frame is averaged to one mono sample before the canceller
|
|
623
|
+
* sees it; a multichannel reference pushed without declaring the count
|
|
624
|
+
* cancels nothing and reports no error. The declared count must match this
|
|
625
|
+
* buffer's actual interleaving: a mismatch is not detected and raises no
|
|
626
|
+
* error, and shows up only as `aecMetrics().delaySamples` staying `null`
|
|
627
|
+
* with no fault reported.
|
|
610
628
|
*
|
|
611
629
|
* Never blocks and never throws on a full queue: samples that do not fit
|
|
612
630
|
* are discarded and counted by `aecMetrics().referenceDropped`, and the
|
|
@@ -714,19 +732,20 @@ function version() {
|
|
|
714
732
|
|
|
715
733
|
class File extends Readable {
|
|
716
734
|
/**
|
|
717
|
-
* Open
|
|
735
|
+
* Open an audio file as an offline source, synchronously. Everything a
|
|
718
736
|
* `Microphone` does to live audio, a `File` does to audio you already
|
|
719
737
|
* have: the same conditioning options, the same stream of conditioned
|
|
720
738
|
* chunks, and (with `vad` set) the same per-chunk speech events, plus the
|
|
721
739
|
* whole-file `analyze()` a live stream cannot offer.
|
|
722
740
|
*
|
|
723
|
-
* The bare constructor reads the
|
|
741
|
+
* The bare constructor reads the file inline, blocking the event loop on
|
|
724
742
|
* disk I/O; prefer `await File.open(path, options)` in servers and other
|
|
725
743
|
* latency-sensitive code, exactly as `Microphone.open` is preferred over
|
|
726
744
|
* `new Microphone`. Iteration and analysis are separate single passes:
|
|
727
745
|
* each consumes the source once, so use one `File` per operation.
|
|
728
746
|
*
|
|
729
|
-
* @param {string} filePath Path to a WAV
|
|
747
|
+
* @param {string} filePath Path to a WAV, AIFF, AIFF-C or FLAC file. The
|
|
748
|
+
* container is identified from the bytes, not from the extension.
|
|
730
749
|
* @param {import('./decibri').FileOptions} [options]
|
|
731
750
|
* @param {{ prepared: object, native: object }} [_internal] Internal: a
|
|
732
751
|
* pre-resolved options bundle and an already-constructed native handle,
|
|
@@ -932,8 +951,8 @@ class File extends Readable {
|
|
|
932
951
|
}
|
|
933
952
|
|
|
934
953
|
/**
|
|
935
|
-
* Open
|
|
936
|
-
*
|
|
954
|
+
* Open an audio file without blocking the event loop: the disk read,
|
|
955
|
+
* decode, and chain construction run on the native thread pool. The
|
|
937
956
|
* recommended form in Node, mirroring `Microphone.open`. The synchronous
|
|
938
957
|
* `new File(path)` remains available for scripts.
|
|
939
958
|
*
|
|
@@ -1132,10 +1151,10 @@ class File extends Readable {
|
|
|
1132
1151
|
}
|
|
1133
1152
|
|
|
1134
1153
|
/**
|
|
1135
|
-
* The source's own rate, taken from the
|
|
1136
|
-
* passed to `File.buffer`. Differs from `sampleRate` when the
|
|
1137
|
-
* resampled. Readable for the life of the File, including after
|
|
1138
|
-
* is consumed or closed.
|
|
1154
|
+
* The source's own rate, taken from the file's header or from the
|
|
1155
|
+
* `inputRate` passed to `File.buffer`. Differs from `sampleRate` when the
|
|
1156
|
+
* source was resampled. Readable for the life of the File, including after
|
|
1157
|
+
* the source is consumed or closed.
|
|
1139
1158
|
* @returns {number}
|
|
1140
1159
|
*/
|
|
1141
1160
|
get inputRate() {
|
|
@@ -1190,6 +1209,77 @@ class File extends Readable {
|
|
|
1190
1209
|
return this.analyze();
|
|
1191
1210
|
}
|
|
1192
1211
|
|
|
1212
|
+
/**
|
|
1213
|
+
* Validate the save options and resolve them into the native options
|
|
1214
|
+
* object. Shared by `File.save` and `AudioWriter`, so the two spellings of
|
|
1215
|
+
* a write accept and reject identically.
|
|
1216
|
+
* @internal
|
|
1217
|
+
* @param {import('./decibri').SaveOptions} options
|
|
1218
|
+
*/
|
|
1219
|
+
static _prepareSaveOptions(options) {
|
|
1220
|
+
const format = options.format;
|
|
1221
|
+
if (format !== undefined && format !== 'wav' && format !== 'aiff' && format !== 'flac') {
|
|
1222
|
+
throw new TypeError(
|
|
1223
|
+
`Invalid format value: ${JSON.stringify(format)}. Expected 'wav', 'aiff', or 'flac'.`
|
|
1224
|
+
);
|
|
1225
|
+
}
|
|
1226
|
+
const compression = options.compression;
|
|
1227
|
+
if (compression !== undefined) {
|
|
1228
|
+
if (typeof compression !== 'number' || Number.isNaN(compression)) {
|
|
1229
|
+
throw new TypeError('compression must be a number');
|
|
1230
|
+
}
|
|
1231
|
+
if (compression < 0 || compression > 8) {
|
|
1232
|
+
throw new RangeError('flac compression level must be between 0 and 8');
|
|
1233
|
+
}
|
|
1234
|
+
}
|
|
1235
|
+
return { format, compression };
|
|
1236
|
+
}
|
|
1237
|
+
|
|
1238
|
+
/**
|
|
1239
|
+
* Write the conditioned recording to disk, off the event loop. Runs the
|
|
1240
|
+
* recording once through the same conditioning pass iteration delivers,
|
|
1241
|
+
* whole, and writes it as 16-bit PCM mono at `sampleRate`. Consumes the
|
|
1242
|
+
* source: a save is a single pass, separate from iteration and analysis.
|
|
1243
|
+
*
|
|
1244
|
+
* The container comes from the path's extension (`.wav`, `.aiff`, `.aif`,
|
|
1245
|
+
* `.aifc` or `.flac`), or from `options.format`: decibri reads a file by
|
|
1246
|
+
* its content and writes one by its name. An extension it does not
|
|
1247
|
+
* recognise rejects rather than defaulting. `options.compression` sets the
|
|
1248
|
+
* FLAC compression level (0-8, default 5); it applies only to FLAC.
|
|
1249
|
+
*
|
|
1250
|
+
* Resolves to a `SaveReport`: `clippedSamples` counts finite samples
|
|
1251
|
+
* outside full scale clamped to `[-1.0, 1.0]` (AGC or AEC without a
|
|
1252
|
+
* limiter can overshoot, and 16-bit PCM cannot hold it), and
|
|
1253
|
+
* `nonFiniteSamples` counts NaN samples written as silence and infinite
|
|
1254
|
+
* samples written as full scale.
|
|
1255
|
+
*
|
|
1256
|
+
* Requires a `File` that is not already being streamed: once the stream
|
|
1257
|
+
* has been engaged this rejects with a `DecibriError` carrying the code
|
|
1258
|
+
* `'FILE_ENGAGED'`. Every failure detected before the pass begins leaves
|
|
1259
|
+
* the `File` usable; a failure during the pass consumes the source.
|
|
1260
|
+
*
|
|
1261
|
+
* @param {string} filePath
|
|
1262
|
+
* @param {import('./decibri').SaveOptions} [options]
|
|
1263
|
+
* @returns {Promise<import('./decibri').SaveReport>}
|
|
1264
|
+
*/
|
|
1265
|
+
async save(filePath, options = {}) {
|
|
1266
|
+
// Checked before the native handle is touched: the rejection belongs to
|
|
1267
|
+
// this call, and the source is never taken from a File that keeps
|
|
1268
|
+
// streaming.
|
|
1269
|
+
if (this._engaged) {
|
|
1270
|
+
throw fileEngagedError();
|
|
1271
|
+
}
|
|
1272
|
+
if (typeof filePath !== 'string') {
|
|
1273
|
+
throw new TypeError('path must be a string');
|
|
1274
|
+
}
|
|
1275
|
+
const nativeOptions = File._prepareSaveOptions(options);
|
|
1276
|
+
try {
|
|
1277
|
+
return await this._native.save(filePath, nativeOptions);
|
|
1278
|
+
} catch (err) {
|
|
1279
|
+
throw wrapNativeError(err);
|
|
1280
|
+
}
|
|
1281
|
+
}
|
|
1282
|
+
|
|
1193
1283
|
/**
|
|
1194
1284
|
* Release the source. Idempotent; a closed File reads as ended.
|
|
1195
1285
|
*/
|
|
@@ -1198,10 +1288,131 @@ class File extends Readable {
|
|
|
1198
1288
|
}
|
|
1199
1289
|
}
|
|
1200
1290
|
|
|
1291
|
+
// ─── AudioWriter (Writable): file sink ───────────────────────────────────────
|
|
1292
|
+
|
|
1293
|
+
class AudioWriter extends Writable {
|
|
1294
|
+
/**
|
|
1295
|
+
* A file sink for PCM audio: the Writable to pair with decibri's Readable
|
|
1296
|
+
* sources, and with any other stream of PCM bytes (a TTS engine, a decoded
|
|
1297
|
+
* network stream). Collects the whole stream, then writes it as one audio
|
|
1298
|
+
* file when the stream finishes, exactly as `File.save` writes: the same
|
|
1299
|
+
* containers from the same extension rule, the same 16-bit PCM encoding,
|
|
1300
|
+
* the same clamp and non-finite handling, the same bytes.
|
|
1301
|
+
*
|
|
1302
|
+
* Chunks are raw PCM bytes in `dtype` ('int16' little-endian by default,
|
|
1303
|
+
* matching what a `File` or `Microphone` emits; 'float32' for raw f32
|
|
1304
|
+
* bytes). Audio is written mono: `channels` may only be 1. `sampleRate` is
|
|
1305
|
+
* required, because raw audio carries no header to read one from.
|
|
1306
|
+
*
|
|
1307
|
+
* The file is written when the stream finishes ('finish' fires after the
|
|
1308
|
+
* file is on disk), and `report` then carries the `SaveReport` the write
|
|
1309
|
+
* produced. A failure destroys the stream with the error.
|
|
1310
|
+
*
|
|
1311
|
+
* @param {string} filePath Path whose extension names the container,
|
|
1312
|
+
* unless `format` overrides it.
|
|
1313
|
+
* @param {import('./decibri').AudioWriterOptions} options
|
|
1314
|
+
*/
|
|
1315
|
+
constructor(filePath, options = {}) {
|
|
1316
|
+
super({ highWaterMark: options.highWaterMark });
|
|
1317
|
+
if (typeof filePath !== 'string') {
|
|
1318
|
+
throw new TypeError('path must be a string');
|
|
1319
|
+
}
|
|
1320
|
+
const sampleRate = options.sampleRate;
|
|
1321
|
+
if (typeof sampleRate !== 'number' || Number.isNaN(sampleRate)) {
|
|
1322
|
+
throw new TypeError('sampleRate is required for AudioWriter (raw audio carries no header)');
|
|
1323
|
+
}
|
|
1324
|
+
if (sampleRate < 1000 || sampleRate > 384000) {
|
|
1325
|
+
throw new RangeError('sample rate must be between 1000 and 384000');
|
|
1326
|
+
}
|
|
1327
|
+
const channels = options.channels;
|
|
1328
|
+
if (channels !== undefined && channels !== 1) {
|
|
1329
|
+
throw new RangeError('multichannel write is not supported; channels must be 1 (mono)');
|
|
1330
|
+
}
|
|
1331
|
+
const dtype = options.dtype ?? 'int16';
|
|
1332
|
+
if (dtype !== 'int16' && dtype !== 'float32') {
|
|
1333
|
+
throw new TypeError("dtype must be 'int16' or 'float32'");
|
|
1334
|
+
}
|
|
1335
|
+
// Validated now, so a bad option is a construction-time throw rather
|
|
1336
|
+
// than a deferred 'error' event after the audio has been streamed.
|
|
1337
|
+
this._saveOptions = File._prepareSaveOptions(options);
|
|
1338
|
+
this._filePath = filePath;
|
|
1339
|
+
this._sampleRate = sampleRate;
|
|
1340
|
+
this._dtype = dtype;
|
|
1341
|
+
this._chunks = [];
|
|
1342
|
+
this._report = null;
|
|
1343
|
+
}
|
|
1344
|
+
|
|
1345
|
+
/**
|
|
1346
|
+
* The `SaveReport` of the completed write: `clippedSamples` and
|
|
1347
|
+
* `nonFiniteSamples`, exactly as `File.save` resolves. `null` until
|
|
1348
|
+
* 'finish' has fired.
|
|
1349
|
+
* @returns {import('./decibri').SaveReport | null}
|
|
1350
|
+
*/
|
|
1351
|
+
get report() {
|
|
1352
|
+
return this._report;
|
|
1353
|
+
}
|
|
1354
|
+
|
|
1355
|
+
/** @internal */
|
|
1356
|
+
_write(chunk, encoding, callback) {
|
|
1357
|
+
this._chunks.push(chunk);
|
|
1358
|
+
callback();
|
|
1359
|
+
}
|
|
1360
|
+
|
|
1361
|
+
/** @internal */
|
|
1362
|
+
_final(callback) {
|
|
1363
|
+
const bytes = Buffer.concat(this._chunks);
|
|
1364
|
+
this._chunks = [];
|
|
1365
|
+
const bytesPerSample = this._dtype === 'int16' ? 2 : 4;
|
|
1366
|
+
if (bytes.length % bytesPerSample !== 0) {
|
|
1367
|
+
callback(
|
|
1368
|
+
new RangeError(
|
|
1369
|
+
`audio bytes do not divide into whole ${this._dtype} samples; ` +
|
|
1370
|
+
`${bytes.length % bytesPerSample} byte(s) over`
|
|
1371
|
+
)
|
|
1372
|
+
);
|
|
1373
|
+
return;
|
|
1374
|
+
}
|
|
1375
|
+
const samples = new Float32Array(bytes.length / bytesPerSample);
|
|
1376
|
+
if (this._dtype === 'int16') {
|
|
1377
|
+
// The inverse of the int16 delivery encoding: value / 32768, so bytes
|
|
1378
|
+
// that came from a decibri source re-quantise to the identical file.
|
|
1379
|
+
for (let i = 0; i < samples.length; i++) {
|
|
1380
|
+
samples[i] = bytes.readInt16LE(i * 2) / 32768;
|
|
1381
|
+
}
|
|
1382
|
+
} else {
|
|
1383
|
+
for (let i = 0; i < samples.length; i++) {
|
|
1384
|
+
samples[i] = bytes.readFloatLE(i * 4);
|
|
1385
|
+
}
|
|
1386
|
+
}
|
|
1387
|
+
// The write is File.save on a source at the writer's own rate: the same
|
|
1388
|
+
// encode path, so the two spellings produce the same bytes.
|
|
1389
|
+
let file;
|
|
1390
|
+
try {
|
|
1391
|
+
file = File.buffer(samples, {
|
|
1392
|
+
inputRate: this._sampleRate,
|
|
1393
|
+
sampleRate: this._sampleRate,
|
|
1394
|
+
});
|
|
1395
|
+
} catch (err) {
|
|
1396
|
+
callback(err);
|
|
1397
|
+
return;
|
|
1398
|
+
}
|
|
1399
|
+
file
|
|
1400
|
+
.save(this._filePath, this._saveOptions)
|
|
1401
|
+
.then((report) => {
|
|
1402
|
+
this._report = report;
|
|
1403
|
+
callback();
|
|
1404
|
+
})
|
|
1405
|
+
.catch((err) => {
|
|
1406
|
+
callback(err);
|
|
1407
|
+
});
|
|
1408
|
+
}
|
|
1409
|
+
}
|
|
1410
|
+
|
|
1201
1411
|
module.exports = {
|
|
1202
1412
|
Microphone,
|
|
1203
1413
|
Speaker,
|
|
1204
1414
|
File,
|
|
1415
|
+
AudioWriter,
|
|
1205
1416
|
inputDevices,
|
|
1206
1417
|
outputDevices,
|
|
1207
1418
|
version,
|
package/src/errors.js
CHANGED
|
@@ -75,13 +75,15 @@ class OrtPathError extends OrtError {
|
|
|
75
75
|
// the core ever sees it).
|
|
76
76
|
const RANGE_PREFIXES = [
|
|
77
77
|
'sample rate must be between',
|
|
78
|
-
'channels must be
|
|
78
|
+
'channels must be at least',
|
|
79
79
|
'multichannel capture is not supported',
|
|
80
80
|
'frames per buffer must be between',
|
|
81
81
|
'agc target level must be between',
|
|
82
82
|
'limiter ceiling must be between',
|
|
83
|
+
'flac compression level must be between',
|
|
83
84
|
'aec tailMs must be between',
|
|
84
85
|
'aec referenceSampleRate must be between',
|
|
86
|
+
'aec referenceChannels must be at',
|
|
85
87
|
'Silero VAD only supports',
|
|
86
88
|
'VAD threshold must be between',
|
|
87
89
|
'echo cancellation only supports',
|
|
@@ -93,6 +95,7 @@ const TYPE_PREFIXES = [
|
|
|
93
95
|
"dtype must be 'int16' or 'float32'",
|
|
94
96
|
"format must be 'int16' or 'float32'",
|
|
95
97
|
'aec suppression must be',
|
|
98
|
+
'Invalid format value:',
|
|
96
99
|
];
|
|
97
100
|
|
|
98
101
|
const DEVICE_CODES = [
|
|
@@ -123,6 +126,7 @@ const BASE_CODES = [
|
|
|
123
126
|
['audio stream is already running', 'ALREADY_RUNNING'],
|
|
124
127
|
['Failed to open audio stream', 'STREAM_OPEN_FAILED'],
|
|
125
128
|
['Failed to start audio stream', 'STREAM_START_FAILED'],
|
|
129
|
+
['the output device does not support', 'SPEAKER_CHANNELS_UNSUPPORTED'],
|
|
126
130
|
['Microphone permission denied.', 'PERMISSION_DENIED'],
|
|
127
131
|
['Microphone stream is closed', 'MICROPHONE_STREAM_CLOSED'],
|
|
128
132
|
['Speaker stream is closed', 'SPEAKER_STREAM_CLOSED'],
|
|
@@ -132,7 +136,10 @@ const BASE_CODES = [
|
|
|
132
136
|
['resampler error:', 'RESAMPLE_FAILED'],
|
|
133
137
|
['echo canceller configuration error:', 'AEC_CONFIG_INVALID'],
|
|
134
138
|
['Failed to read audio file', 'FILE_READ_FAILED'],
|
|
135
|
-
['
|
|
139
|
+
['Failed to write audio file', 'FILE_WRITE_FAILED'],
|
|
140
|
+
['unsupported audio format:', 'AUDIO_FORMAT_UNSUPPORTED'],
|
|
141
|
+
['malformed audio file:', 'AUDIO_FILE_MALFORMED'],
|
|
142
|
+
['truncated audio file:', 'AUDIO_FILE_TRUNCATED'],
|
|
136
143
|
['ONNX Runtime was initialized in pid', 'FORK_AFTER_ORT_INIT'],
|
|
137
144
|
['ONNX backend error from', 'ONNX_BACKEND_FAILED'],
|
|
138
145
|
// Authored in the napi layer (bindings/node/src/lib.rs), not in
|