decibri 5.5.0 → 5.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +54 -1
- package/MIGRATION.md +8 -12
- package/README.md +13 -9
- package/examples/decibri.browser.js +93 -12
- package/index.d.ts +94 -15
- package/index.js +52 -52
- package/package.json +7 -5
- package/src/browser/decibri-browser.js +174 -25
- package/src/browser/index.d.ts +58 -7
- package/src/browser/worklet-inline.js +2 -2
- package/src/browser/worklet-processor.js +116 -32
- package/src/decibri.d.ts +183 -22
- package/src/decibri.js +242 -47
- package/src/errors.js +9 -1
package/src/decibri.js
CHANGED
|
@@ -25,18 +25,19 @@ const PACKAGE_VERSION = require('../package.json').version;
|
|
|
25
25
|
* file: bundled ORT dylib filename }
|
|
26
26
|
*
|
|
27
27
|
* Dylib filenames are unversioned across all platforms for consistency. The
|
|
28
|
-
*
|
|
29
|
-
* versioned upstream tarball file (e.g. libonnxruntime.1.
|
|
28
|
+
* publish workflow (.github/workflows/publish-npm.yml) copies Microsoft's
|
|
29
|
+
* versioned upstream tarball file (e.g. libonnxruntime.1.28.1.dylib) into the
|
|
30
30
|
* platform package with the unversioned name listed here.
|
|
31
31
|
*
|
|
32
32
|
* If you add a new platform, update this table AND the matching platform job
|
|
33
|
-
* in
|
|
33
|
+
* in publish-npm.yml.
|
|
34
34
|
*/
|
|
35
35
|
const PLATFORM_DYLIB = {
|
|
36
36
|
'darwin-arm64': { pkg: '@decibri/decibri-darwin-arm64', file: 'libonnxruntime.dylib' },
|
|
37
37
|
'linux-x64': { pkg: '@decibri/decibri-linux-x64-gnu', file: 'libonnxruntime.so' },
|
|
38
38
|
'linux-arm64': { pkg: '@decibri/decibri-linux-arm64-gnu', file: 'libonnxruntime.so' },
|
|
39
39
|
'win32-x64': { pkg: '@decibri/decibri-win32-x64-msvc', file: 'onnxruntime.dll' },
|
|
40
|
+
'win32-arm64': { pkg: '@decibri/decibri-win32-arm64-msvc', file: 'onnxruntime.dll' },
|
|
40
41
|
};
|
|
41
42
|
|
|
42
43
|
/**
|
|
@@ -102,6 +103,7 @@ class Microphone extends Readable {
|
|
|
102
103
|
this._vad = prepared.vadEnabled;
|
|
103
104
|
this._vadThreshold = prepared.vadThreshold;
|
|
104
105
|
this._vadHoldoff = prepared.vadHoldoff;
|
|
106
|
+
this._dtype = prepared.dtype;
|
|
105
107
|
this._vadScore = 0;
|
|
106
108
|
this._isSpeaking = false;
|
|
107
109
|
this._silenceTimer = null;
|
|
@@ -145,16 +147,42 @@ class Microphone extends Readable {
|
|
|
145
147
|
throw new RangeError('sample rate must be between 1000 and 384000');
|
|
146
148
|
}
|
|
147
149
|
|
|
148
|
-
//
|
|
149
|
-
//
|
|
150
|
-
//
|
|
151
|
-
// additive. The `channels` option is kept for that forward compatibility.
|
|
150
|
+
// The number of channels delivered, interleaved frame by frame. Bounded
|
|
151
|
+
// below here; bounded above by the resolved device alone, which answers
|
|
152
|
+
// when the stream starts. No fixed maximum exists on this path.
|
|
152
153
|
const channels = options.channels ?? 1;
|
|
153
154
|
if (channels < 1) {
|
|
154
155
|
throw new RangeError('channels must be at least 1');
|
|
155
156
|
}
|
|
156
|
-
|
|
157
|
-
|
|
157
|
+
|
|
158
|
+
// ── Validate channel map ─────────────────────────────────────────────────
|
|
159
|
+
|
|
160
|
+
// An optional list of 0-based device channel indices, one per delivered
|
|
161
|
+
// channel: delivered channel j carries device channel channelMap[j].
|
|
162
|
+
// Absence delivers the documented average of every opened channel. The
|
|
163
|
+
// checks here are shape-only (an array of integers that fit the channel
|
|
164
|
+
// count's width, with one entry per channel); whether each entry exists on
|
|
165
|
+
// the device is the core's check, made against the resolved device's own
|
|
166
|
+
// report when the stream starts, because only the device can say how many
|
|
167
|
+
// channels it has. No fixed maximum exists on this path.
|
|
168
|
+
const channelMap = options.channelMap;
|
|
169
|
+
if (channelMap !== undefined) {
|
|
170
|
+
if (!Array.isArray(channelMap)) {
|
|
171
|
+
throw new TypeError(
|
|
172
|
+
`Invalid channelMap value: ${JSON.stringify(channelMap)}. Expected an array of 0-based device channel indices, such as [0].`
|
|
173
|
+
);
|
|
174
|
+
}
|
|
175
|
+
for (const entry of channelMap) {
|
|
176
|
+
if (typeof entry !== 'number' || !Number.isInteger(entry)) {
|
|
177
|
+
throw new TypeError('channelMap entries must be integers');
|
|
178
|
+
}
|
|
179
|
+
if (entry < 0 || entry > 65535) {
|
|
180
|
+
throw new RangeError('channelMap entries must be between 0 and 65535');
|
|
181
|
+
}
|
|
182
|
+
}
|
|
183
|
+
if (channelMap.length !== channels) {
|
|
184
|
+
throw new RangeError('channelMap must have exactly one entry per channel');
|
|
185
|
+
}
|
|
158
186
|
}
|
|
159
187
|
|
|
160
188
|
const framesPerBuffer = options.framesPerBuffer ?? 1600;
|
|
@@ -202,7 +230,7 @@ class Microphone extends Readable {
|
|
|
202
230
|
// vad selects the detector and (optionally) its threshold/holdoff policy.
|
|
203
231
|
// It accepts false (disabled, default), the 'silero'/'energy' shorthand
|
|
204
232
|
// (which uses the mode's default threshold and holdoff), or a config object
|
|
205
|
-
// { model, threshold, holdoffMs } to tune the policy. The legacy two-flag
|
|
233
|
+
// { model, threshold, holdoffMs, source } to tune the policy. The legacy two-flag
|
|
206
234
|
// form (vad: true plus vadMode) and the flat vadThreshold/vadHoldoff
|
|
207
235
|
// options are rejected with a migration error. The threshold and holdoff
|
|
208
236
|
// live JS-side (the state machine runs in this wrapper); only the mode is
|
|
@@ -218,6 +246,7 @@ class Microphone extends Readable {
|
|
|
218
246
|
let vadMode;
|
|
219
247
|
let vadThreshold;
|
|
220
248
|
let vadHoldoff;
|
|
249
|
+
let vadSource;
|
|
221
250
|
if (vad === false) {
|
|
222
251
|
vadEnabled = false;
|
|
223
252
|
vadMode = 'energy'; // inert placeholder; ignored while disabled
|
|
@@ -229,10 +258,13 @@ class Microphone extends Readable {
|
|
|
229
258
|
vadEnabled = true;
|
|
230
259
|
vadMode = vad;
|
|
231
260
|
} else if (vad !== null && typeof vad === 'object' && !Array.isArray(vad)) {
|
|
232
|
-
// Config object form: { model, threshold?, holdoffMs? }. model
|
|
233
|
-
// and selects the detector; threshold and holdoffMs override
|
|
234
|
-
// defaults when supplied
|
|
235
|
-
|
|
261
|
+
// Config object form: { model, threshold?, holdoffMs?, source? }. model
|
|
262
|
+
// is required and selects the detector; threshold and holdoffMs override
|
|
263
|
+
// the mode defaults when supplied; source names the 0-based DELIVERED
|
|
264
|
+
// channel the detector reads (the position within the delivered
|
|
265
|
+
// interleaved frames, after any channelMap), absent feeding the frame
|
|
266
|
+
// average of every delivered channel.
|
|
267
|
+
const { model, threshold, holdoffMs, source } = vad;
|
|
236
268
|
if (model !== 'silero' && model !== 'energy') {
|
|
237
269
|
throw new TypeError(
|
|
238
270
|
`Invalid vad model: ${JSON.stringify(model)}. Expected 'silero' or 'energy'.`
|
|
@@ -258,9 +290,26 @@ class Microphone extends Readable {
|
|
|
258
290
|
}
|
|
259
291
|
vadHoldoff = holdoffMs;
|
|
260
292
|
}
|
|
293
|
+
if (source !== undefined) {
|
|
294
|
+
if (typeof source !== 'number' || !Number.isInteger(source)) {
|
|
295
|
+
throw new TypeError('vad source must be an integer');
|
|
296
|
+
}
|
|
297
|
+
if (source < 0 || source > 65535) {
|
|
298
|
+
throw new RangeError('vad source must be between 0 and 65535');
|
|
299
|
+
}
|
|
300
|
+
// The delivered count is the only ceiling, checked here where both
|
|
301
|
+
// sides are in scope; the message is the core's own for the same
|
|
302
|
+
// condition. No fixed maximum exists.
|
|
303
|
+
if (source >= channels) {
|
|
304
|
+
throw new RangeError(
|
|
305
|
+
`the detector source names delivered channel ${source}; the delivered channel count is ${channels}`
|
|
306
|
+
);
|
|
307
|
+
}
|
|
308
|
+
vadSource = source;
|
|
309
|
+
}
|
|
261
310
|
} else {
|
|
262
311
|
throw new TypeError(
|
|
263
|
-
`Invalid vad value: ${JSON.stringify(vad)}. Expected false, 'silero', 'energy', or a config object { model, threshold, holdoffMs }.`
|
|
312
|
+
`Invalid vad value: ${JSON.stringify(vad)}. Expected false, 'silero', 'energy', or a config object { model, threshold, holdoffMs, source }.`
|
|
264
313
|
);
|
|
265
314
|
}
|
|
266
315
|
|
|
@@ -433,6 +482,7 @@ class Microphone extends Readable {
|
|
|
433
482
|
nativeOptions: {
|
|
434
483
|
sampleRate,
|
|
435
484
|
channels,
|
|
485
|
+
channelMap,
|
|
436
486
|
framesPerBuffer,
|
|
437
487
|
format: dtype,
|
|
438
488
|
device: resolvedDevice,
|
|
@@ -441,6 +491,9 @@ class Microphone extends Readable {
|
|
|
441
491
|
// native compute the energy score for a microphone that did not ask for
|
|
442
492
|
// VAD. Absent means VAD off in native.
|
|
443
493
|
vadMode: vadEnabled ? vadMode : undefined,
|
|
494
|
+
// The delivered channel the detector reads, from the vad config
|
|
495
|
+
// object's source key. Absent feeds the frame average.
|
|
496
|
+
detectorSource: vadSource,
|
|
444
497
|
modelPath,
|
|
445
498
|
dcRemoval,
|
|
446
499
|
denoise,
|
|
@@ -614,24 +667,27 @@ class Microphone extends Readable {
|
|
|
614
667
|
|
|
615
668
|
/**
|
|
616
669
|
* Queue far-end reference audio for the echo canceller: the audio being
|
|
617
|
-
* played out, pushed as it is played, in played order. Accepts
|
|
618
|
-
*
|
|
619
|
-
*
|
|
620
|
-
* `
|
|
621
|
-
*
|
|
622
|
-
*
|
|
623
|
-
*
|
|
624
|
-
*
|
|
625
|
-
*
|
|
626
|
-
*
|
|
627
|
-
*
|
|
670
|
+
* played out, pushed as it is played, in played order. Accepts a `Buffer`,
|
|
671
|
+
* `Uint8Array`, or `DataView` of PCM bytes in this microphone's `dtype`,
|
|
672
|
+
* or the typed array carrying that dtype (`Int16Array` for `'int16'`,
|
|
673
|
+
* `Float32Array` for `'float32'`), at the declared `referenceSampleRate`
|
|
674
|
+
* (the capture rate when unset), interleaved at the declared
|
|
675
|
+
* `referenceChannels` (mono when unset). A typed array carrying any other
|
|
676
|
+
* sample dtype throws a `TypeError`, whatever the capture state. With
|
|
677
|
+
* `referenceChannels` above 1, each frame is averaged to one mono sample
|
|
678
|
+
* before the canceller sees it; a multichannel reference pushed without
|
|
679
|
+
* declaring the count cancels nothing and reports no error. The declared
|
|
680
|
+
* count must match this buffer's actual interleaving: a mismatch is not
|
|
681
|
+
* detected and raises no error, and shows up only as
|
|
682
|
+
* `aecMetrics().delaySamples` staying `null` with no fault reported.
|
|
628
683
|
*
|
|
629
684
|
* Never blocks and never throws on a full queue: samples that do not fit
|
|
630
685
|
* are discarded and counted by `aecMetrics().referenceDropped`, and the
|
|
631
686
|
* span they occupied is represented as silence. Silence between played
|
|
632
687
|
* audio need not be pushed; a caller that stops pushing has said nothing is
|
|
633
|
-
* playing. A push while capture is not running
|
|
634
|
-
*
|
|
688
|
+
* playing. A push while capture is not running is discarded and counted by
|
|
689
|
+
* `referenceDropped`, read once capture runs; a push with the `aec` option
|
|
690
|
+
* unset is a no-op.
|
|
635
691
|
*
|
|
636
692
|
* @param {Buffer | NodeJS.ArrayBufferView} data PCM samples in the
|
|
637
693
|
* configured `dtype`.
|
|
@@ -641,8 +697,28 @@ class Microphone extends Readable {
|
|
|
641
697
|
if (Buffer.isBuffer(data)) {
|
|
642
698
|
buf = data;
|
|
643
699
|
} else if (ArrayBuffer.isView(data)) {
|
|
644
|
-
//
|
|
645
|
-
// the
|
|
700
|
+
// A typed array names its own sample dtype, so one carrying a dtype
|
|
701
|
+
// other than the configured one is refused rather than read as raw
|
|
702
|
+
// bytes. Buffer, Uint8Array, and DataView are format-agnostic byte
|
|
703
|
+
// carriers, exactly as bytes are on the Python surface; the accepted
|
|
704
|
+
// view is normalized to the same bytes, no copy, exactly as the stream
|
|
705
|
+
// machinery normalizes a typed-array write to a Speaker.
|
|
706
|
+
if (!(data instanceof Uint8Array) && !(data instanceof DataView)) {
|
|
707
|
+
const expected = this._dtype === 'int16' ? Int16Array : Float32Array;
|
|
708
|
+
if (!(data instanceof expected)) {
|
|
709
|
+
const mismatched = this._dtype === 'int16' ? Float32Array : Int16Array;
|
|
710
|
+
if (data instanceof mismatched) {
|
|
711
|
+
const other = this._dtype === 'int16' ? 'float32' : 'int16';
|
|
712
|
+
throw new TypeError(
|
|
713
|
+
`dtype '${this._dtype}' configured but ${mismatched.name} samples were pushed; ` +
|
|
714
|
+
`convert to ${expected.name} or construct Microphone with dtype: '${other}'`
|
|
715
|
+
);
|
|
716
|
+
}
|
|
717
|
+
throw new TypeError(
|
|
718
|
+
'pushAecReference requires a Buffer, TypedArray, or DataView of PCM samples in the configured dtype'
|
|
719
|
+
);
|
|
720
|
+
}
|
|
721
|
+
}
|
|
646
722
|
buf = Buffer.from(data.buffer, data.byteOffset, data.byteLength);
|
|
647
723
|
} else {
|
|
648
724
|
throw new TypeError(
|
|
@@ -663,6 +739,11 @@ class Microphone extends Readable {
|
|
|
663
739
|
* `referenceDropped` means single pushes are exceeding the reference
|
|
664
740
|
* queue's bound.
|
|
665
741
|
*
|
|
742
|
+
* The top-level engine fields report the first delivered channel's
|
|
743
|
+
* canceller; `channels` carries every delivered channel's report in
|
|
744
|
+
* delivered order, one entry per channel, so the two agree on a
|
|
745
|
+
* single-channel stream.
|
|
746
|
+
*
|
|
666
747
|
* @returns {import('./decibri').AecMetrics | null}
|
|
667
748
|
*/
|
|
668
749
|
aecMetrics() {
|
|
@@ -677,6 +758,14 @@ class Microphone extends Readable {
|
|
|
677
758
|
referenceReanchors: m.referenceReanchors,
|
|
678
759
|
referenceDropped: m.referenceDropped,
|
|
679
760
|
referenceSilence: m.referenceSilence,
|
|
761
|
+
channels: m.channels.map((c) => ({
|
|
762
|
+
delaySamples: c.delaySamples ?? null,
|
|
763
|
+
erleDb: c.erleDb,
|
|
764
|
+
doubleTalk: c.doubleTalk,
|
|
765
|
+
referenceStarved: c.referenceStarved,
|
|
766
|
+
acquisitionParked: c.acquisitionParked,
|
|
767
|
+
referenceReanchors: c.referenceReanchors,
|
|
768
|
+
})),
|
|
680
769
|
};
|
|
681
770
|
}
|
|
682
771
|
|
|
@@ -755,6 +844,15 @@ class File extends Readable {
|
|
|
755
844
|
constructor(filePath, options = {}, _internal = undefined) {
|
|
756
845
|
super({ highWaterMark: options.highWaterMark, objectMode: false });
|
|
757
846
|
|
|
847
|
+
// The open path reads the source's channel count from the file's own
|
|
848
|
+
// header, so a caller-stated interleave has nothing to describe here;
|
|
849
|
+
// it is refused rather than ignored.
|
|
850
|
+
if (!_internal && options.inputChannels !== undefined) {
|
|
851
|
+
throw new TypeError(
|
|
852
|
+
'inputChannels applies only to File.buffer; a file carries its channel count in its own header'
|
|
853
|
+
);
|
|
854
|
+
}
|
|
855
|
+
|
|
758
856
|
const prepared = _internal ? _internal.prepared : File._prepareOptions(options);
|
|
759
857
|
|
|
760
858
|
// ── Store config ───────────────────────────────────────────────────────
|
|
@@ -772,6 +870,9 @@ class File extends Readable {
|
|
|
772
870
|
this._silenceStartPos = null;
|
|
773
871
|
this._position = 0;
|
|
774
872
|
this._sampleRate = prepared.nativeOptions.sampleRate;
|
|
873
|
+
// The delivered channel count: file time advances by frames, so the
|
|
874
|
+
// interleaved sample count divides by it before it divides by the rate.
|
|
875
|
+
this._channels = prepared.channels;
|
|
775
876
|
this._bytesPerSample = prepared.dtype === 'int16' ? 2 : 4;
|
|
776
877
|
this._ended = false;
|
|
777
878
|
// Set the moment the consumer asks the stream for data, which is earlier
|
|
@@ -810,6 +911,42 @@ class File extends Readable {
|
|
|
810
911
|
throw new RangeError('sample rate must be between 1000 and 384000');
|
|
811
912
|
}
|
|
812
913
|
|
|
914
|
+
// The number of channels delivered, interleaved frame by frame. Bounded
|
|
915
|
+
// below here; bounded above by the source's own channel count alone,
|
|
916
|
+
// which the core reads from the header (or takes from inputChannels)
|
|
917
|
+
// when the source is opened. No fixed maximum exists on this path.
|
|
918
|
+
const channels = options.channels ?? 1;
|
|
919
|
+
if (channels < 1) {
|
|
920
|
+
throw new RangeError('channels must be at least 1');
|
|
921
|
+
}
|
|
922
|
+
|
|
923
|
+
// An optional list of 0-based source channel indices, one per delivered
|
|
924
|
+
// channel: delivered channel j carries source channel channelMap[j].
|
|
925
|
+
// Absence delivers the documented average of every source channel. The
|
|
926
|
+
// checks here are shape-only, exactly as the Microphone's: whether each
|
|
927
|
+
// entry exists on the source is the core's check, made against the
|
|
928
|
+
// source's own count, because only the opened source can say how many
|
|
929
|
+
// channels it has. No fixed maximum exists on this path.
|
|
930
|
+
const channelMap = options.channelMap;
|
|
931
|
+
if (channelMap !== undefined) {
|
|
932
|
+
if (!Array.isArray(channelMap)) {
|
|
933
|
+
throw new TypeError(
|
|
934
|
+
`Invalid channelMap value: ${JSON.stringify(channelMap)}. Expected an array of 0-based source channel indices, such as [0].`
|
|
935
|
+
);
|
|
936
|
+
}
|
|
937
|
+
for (const entry of channelMap) {
|
|
938
|
+
if (typeof entry !== 'number' || !Number.isInteger(entry)) {
|
|
939
|
+
throw new TypeError('channelMap entries must be integers');
|
|
940
|
+
}
|
|
941
|
+
if (entry < 0 || entry > 65535) {
|
|
942
|
+
throw new RangeError('channelMap entries must be between 0 and 65535');
|
|
943
|
+
}
|
|
944
|
+
}
|
|
945
|
+
if (channelMap.length !== channels) {
|
|
946
|
+
throw new RangeError('channelMap must have exactly one entry per channel');
|
|
947
|
+
}
|
|
948
|
+
}
|
|
949
|
+
|
|
813
950
|
const dtype = options.dtype ?? 'int16';
|
|
814
951
|
if (dtype !== 'int16' && dtype !== 'float32') {
|
|
815
952
|
throw new TypeError("dtype must be 'int16' or 'float32'");
|
|
@@ -830,6 +967,7 @@ class File extends Readable {
|
|
|
830
967
|
let vadMode;
|
|
831
968
|
let vadThreshold;
|
|
832
969
|
let vadHoldoff;
|
|
970
|
+
let vadSource;
|
|
833
971
|
if (vad === false) {
|
|
834
972
|
vadEnabled = false;
|
|
835
973
|
vadMode = 'energy'; // inert placeholder; ignored while disabled
|
|
@@ -841,7 +979,7 @@ class File extends Readable {
|
|
|
841
979
|
vadEnabled = true;
|
|
842
980
|
vadMode = vad;
|
|
843
981
|
} else if (vad !== null && typeof vad === 'object' && !Array.isArray(vad)) {
|
|
844
|
-
const { model, threshold, holdoffMs } = vad;
|
|
982
|
+
const { model, threshold, holdoffMs, source } = vad;
|
|
845
983
|
if (model !== 'silero' && model !== 'energy') {
|
|
846
984
|
throw new TypeError(
|
|
847
985
|
`Invalid vad model: ${JSON.stringify(model)}. Expected 'silero' or 'energy'.`
|
|
@@ -867,9 +1005,25 @@ class File extends Readable {
|
|
|
867
1005
|
}
|
|
868
1006
|
vadHoldoff = holdoffMs;
|
|
869
1007
|
}
|
|
1008
|
+
// source names the 0-based DELIVERED channel the detector reads,
|
|
1009
|
+
// exactly as on Microphone; the checks and messages are the same.
|
|
1010
|
+
if (source !== undefined) {
|
|
1011
|
+
if (typeof source !== 'number' || !Number.isInteger(source)) {
|
|
1012
|
+
throw new TypeError('vad source must be an integer');
|
|
1013
|
+
}
|
|
1014
|
+
if (source < 0 || source > 65535) {
|
|
1015
|
+
throw new RangeError('vad source must be between 0 and 65535');
|
|
1016
|
+
}
|
|
1017
|
+
if (source >= channels) {
|
|
1018
|
+
throw new RangeError(
|
|
1019
|
+
`the detector source names delivered channel ${source}; the delivered channel count is ${channels}`
|
|
1020
|
+
);
|
|
1021
|
+
}
|
|
1022
|
+
vadSource = source;
|
|
1023
|
+
}
|
|
870
1024
|
} else {
|
|
871
1025
|
throw new TypeError(
|
|
872
|
-
`Invalid vad value: ${JSON.stringify(vad)}. Expected false, 'silero', 'energy', or a config object { model, threshold, holdoffMs }.`
|
|
1026
|
+
`Invalid vad value: ${JSON.stringify(vad)}. Expected false, 'silero', 'energy', or a config object { model, threshold, holdoffMs, source }.`
|
|
873
1027
|
);
|
|
874
1028
|
}
|
|
875
1029
|
|
|
@@ -923,16 +1077,22 @@ class File extends Readable {
|
|
|
923
1077
|
|
|
924
1078
|
return {
|
|
925
1079
|
dtype,
|
|
1080
|
+
channels,
|
|
926
1081
|
vadEnabled,
|
|
927
1082
|
vadMode,
|
|
928
1083
|
vadThreshold: vadThreshold ?? (vadMode === 'silero' ? 0.5 : 0.01),
|
|
929
1084
|
vadHoldoff: vadHoldoff ?? 300,
|
|
930
1085
|
nativeOptions: {
|
|
931
1086
|
sampleRate,
|
|
1087
|
+
channels,
|
|
1088
|
+
channelMap,
|
|
932
1089
|
format: dtype,
|
|
933
1090
|
// Pass the mode to native only when VAD is enabled, exactly as the
|
|
934
1091
|
// Microphone options do; absent means VAD off in native.
|
|
935
1092
|
vadMode: vadEnabled ? vadMode : undefined,
|
|
1093
|
+
// The delivered channel the detector reads, from the vad config
|
|
1094
|
+
// object's source key. Absent feeds the frame average.
|
|
1095
|
+
detectorSource: vadSource,
|
|
936
1096
|
// The whole-file analysis applies threshold and holdoff in the core
|
|
937
1097
|
// (segment merging in file time), so both cross the boundary here,
|
|
938
1098
|
// unlike the live path where the policy is wrapper-only.
|
|
@@ -964,6 +1124,13 @@ class File extends Readable {
|
|
|
964
1124
|
if (typeof filePath !== 'string') {
|
|
965
1125
|
throw new TypeError('path must be a string');
|
|
966
1126
|
}
|
|
1127
|
+
// The same refusal the synchronous constructor makes: a path's channel
|
|
1128
|
+
// count comes from its own header.
|
|
1129
|
+
if (options.inputChannels !== undefined) {
|
|
1130
|
+
throw new TypeError(
|
|
1131
|
+
'inputChannels applies only to File.buffer; a file carries its channel count in its own header'
|
|
1132
|
+
);
|
|
1133
|
+
}
|
|
967
1134
|
const prepared = File._prepareOptions(options);
|
|
968
1135
|
let native;
|
|
969
1136
|
try {
|
|
@@ -976,12 +1143,13 @@ class File extends Readable {
|
|
|
976
1143
|
|
|
977
1144
|
/**
|
|
978
1145
|
* Wrap in-memory samples as an offline source. `samples` must be a
|
|
979
|
-
* `Float32Array` of
|
|
980
|
-
*
|
|
981
|
-
*
|
|
982
|
-
*
|
|
983
|
-
*
|
|
984
|
-
*
|
|
1146
|
+
* `Float32Array` of samples in [-1.0, 1.0], frame-interleaved at
|
|
1147
|
+
* `inputChannels` (1, mono, by default); a raw `Buffer` of PCM bytes is
|
|
1148
|
+
* rejected as ambiguous (encoded bytes, int16 PCM, and f32 samples are
|
|
1149
|
+
* indistinguishable, and decibri's own capture output is a `Buffer`). Raw
|
|
1150
|
+
* samples carry no header, so `inputRate` (their native rate) is
|
|
1151
|
+
* required; `sampleRate` stays the target output rate. No I/O, so
|
|
1152
|
+
* construction is synchronous.
|
|
985
1153
|
*
|
|
986
1154
|
* @param {Float32Array} samples
|
|
987
1155
|
* @param {import('./decibri').FileBufferOptions} [options]
|
|
@@ -1003,10 +1171,23 @@ class File extends Readable {
|
|
|
1003
1171
|
if (inputRate < 1000 || inputRate > 384000) {
|
|
1004
1172
|
throw new RangeError('inputRate must be between 1000 and 384000');
|
|
1005
1173
|
}
|
|
1174
|
+
// The channel counterpart of inputRate: the interleave of the caller's
|
|
1175
|
+
// own samples. Shape-checked here; whether the samples divide into
|
|
1176
|
+
// whole frames at this count is the core's check.
|
|
1177
|
+
const inputChannels = options.inputChannels ?? 1;
|
|
1178
|
+
if (typeof inputChannels !== 'number' || !Number.isInteger(inputChannels)) {
|
|
1179
|
+
throw new TypeError('inputChannels must be an integer');
|
|
1180
|
+
}
|
|
1181
|
+
if (inputChannels < 1 || inputChannels > 65535) {
|
|
1182
|
+
throw new RangeError('inputChannels must be between 1 and 65535');
|
|
1183
|
+
}
|
|
1006
1184
|
const prepared = File._prepareOptions(options);
|
|
1007
1185
|
let native;
|
|
1008
1186
|
try {
|
|
1009
|
-
native = FileHandle.buffer(samples, inputRate,
|
|
1187
|
+
native = FileHandle.buffer(samples, inputRate, {
|
|
1188
|
+
...prepared.nativeOptions,
|
|
1189
|
+
inputChannels,
|
|
1190
|
+
});
|
|
1010
1191
|
} catch (err) {
|
|
1011
1192
|
throw wrapNativeError(err);
|
|
1012
1193
|
}
|
|
@@ -1094,7 +1275,9 @@ class File extends Readable {
|
|
|
1094
1275
|
// before the opt-in conditioning step, exactly as the live pump does.
|
|
1095
1276
|
this._processVadValue(this._native.vadProbability, chunk.length);
|
|
1096
1277
|
} else {
|
|
1097
|
-
|
|
1278
|
+
// Bytes to interleaved samples to frames to seconds of file time.
|
|
1279
|
+
this._position +=
|
|
1280
|
+
chunk.length / this._bytesPerSample / this._channels / this._sampleRate;
|
|
1098
1281
|
}
|
|
1099
1282
|
this.push(chunk);
|
|
1100
1283
|
}
|
|
@@ -1108,7 +1291,9 @@ class File extends Readable {
|
|
|
1108
1291
|
*/
|
|
1109
1292
|
_processVadValue(value, chunkBytes) {
|
|
1110
1293
|
const chunkStart = this._position;
|
|
1111
|
-
|
|
1294
|
+
// Bytes to interleaved samples to frames to seconds of file time.
|
|
1295
|
+
const chunkEnd =
|
|
1296
|
+
chunkStart + chunkBytes / this._bytesPerSample / this._channels / this._sampleRate;
|
|
1112
1297
|
this._position = chunkEnd;
|
|
1113
1298
|
this._vadScore = value;
|
|
1114
1299
|
if (value >= this._vadThreshold) {
|
|
@@ -1301,7 +1486,9 @@ class AudioWriter extends Writable {
|
|
|
1301
1486
|
*
|
|
1302
1487
|
* Chunks are raw PCM bytes in `dtype` ('int16' little-endian by default,
|
|
1303
1488
|
* matching what a `File` or `Microphone` emits; 'float32' for raw f32
|
|
1304
|
-
* bytes)
|
|
1489
|
+
* bytes), frame-interleaved at `channels` (1, mono, by default; the
|
|
1490
|
+
* stream's total sample count must divide into whole frames, and each
|
|
1491
|
+
* container's own channel ceiling applies at the write). `sampleRate` is
|
|
1305
1492
|
* required, because raw audio carries no header to read one from.
|
|
1306
1493
|
*
|
|
1307
1494
|
* The file is written when the stream finishes ('finish' fires after the
|
|
@@ -1324,9 +1511,12 @@ class AudioWriter extends Writable {
|
|
|
1324
1511
|
if (sampleRate < 1000 || sampleRate > 384000) {
|
|
1325
1512
|
throw new RangeError('sample rate must be between 1000 and 384000');
|
|
1326
1513
|
}
|
|
1327
|
-
|
|
1328
|
-
|
|
1329
|
-
|
|
1514
|
+
// Bounded below here; above, each container's own ceiling answers at
|
|
1515
|
+
// the write, with the container layer's own message. No decibri-side
|
|
1516
|
+
// maximum exists on this path.
|
|
1517
|
+
const channels = options.channels ?? 1;
|
|
1518
|
+
if (channels < 1) {
|
|
1519
|
+
throw new RangeError('channels must be at least 1');
|
|
1330
1520
|
}
|
|
1331
1521
|
const dtype = options.dtype ?? 'int16';
|
|
1332
1522
|
if (dtype !== 'int16' && dtype !== 'float32') {
|
|
@@ -1337,6 +1527,7 @@ class AudioWriter extends Writable {
|
|
|
1337
1527
|
this._saveOptions = File._prepareSaveOptions(options);
|
|
1338
1528
|
this._filePath = filePath;
|
|
1339
1529
|
this._sampleRate = sampleRate;
|
|
1530
|
+
this._channels = channels;
|
|
1340
1531
|
this._dtype = dtype;
|
|
1341
1532
|
this._chunks = [];
|
|
1342
1533
|
this._report = null;
|
|
@@ -1384,13 +1575,17 @@ class AudioWriter extends Writable {
|
|
|
1384
1575
|
samples[i] = bytes.readFloatLE(i * 4);
|
|
1385
1576
|
}
|
|
1386
1577
|
}
|
|
1387
|
-
// The write is File.save on a source at the writer's own rate
|
|
1388
|
-
// encode path, so the two spellings produce the
|
|
1578
|
+
// The write is File.save on a source at the writer's own rate and
|
|
1579
|
+
// interleave: the same encode path, so the two spellings produce the
|
|
1580
|
+
// same bytes. A stream that does not divide into whole frames is the
|
|
1581
|
+
// core's refusal, surfaced here when the stream finishes.
|
|
1389
1582
|
let file;
|
|
1390
1583
|
try {
|
|
1391
1584
|
file = File.buffer(samples, {
|
|
1392
1585
|
inputRate: this._sampleRate,
|
|
1586
|
+
inputChannels: this._channels,
|
|
1393
1587
|
sampleRate: this._sampleRate,
|
|
1588
|
+
channels: this._channels,
|
|
1394
1589
|
});
|
|
1395
1590
|
} catch (err) {
|
|
1396
1591
|
callback(err);
|
package/src/errors.js
CHANGED
|
@@ -76,10 +76,12 @@ class OrtPathError extends OrtError {
|
|
|
76
76
|
const RANGE_PREFIXES = [
|
|
77
77
|
'sample rate must be between',
|
|
78
78
|
'channels must be at least',
|
|
79
|
-
'multichannel capture is not supported',
|
|
80
79
|
'frames per buffer must be between',
|
|
81
80
|
'agc target level must be between',
|
|
82
81
|
'limiter ceiling must be between',
|
|
82
|
+
'the detector source names',
|
|
83
|
+
'the channel map has',
|
|
84
|
+
'the requested block size',
|
|
83
85
|
'flac compression level must be between',
|
|
84
86
|
'aec tailMs must be between',
|
|
85
87
|
'aec referenceSampleRate must be between',
|
|
@@ -127,6 +129,12 @@ const BASE_CODES = [
|
|
|
127
129
|
['Failed to open audio stream', 'STREAM_OPEN_FAILED'],
|
|
128
130
|
['Failed to start audio stream', 'STREAM_START_FAILED'],
|
|
129
131
|
['the output device does not support', 'SPEAKER_CHANNELS_UNSUPPORTED'],
|
|
132
|
+
['the input device does not support', 'MICROPHONE_CHANNELS_UNSUPPORTED'],
|
|
133
|
+
['a channel map is required', 'CHANNEL_SELECTION_AMBIGUOUS'],
|
|
134
|
+
['the channel map names', 'CHANNEL_MAP_OUT_OF_RANGE'],
|
|
135
|
+
['the file does not have', 'FILE_CHANNELS_UNSUPPORTED'],
|
|
136
|
+
['delivering', 'FILE_CHANNEL_SELECTION_AMBIGUOUS'],
|
|
137
|
+
['the file channel map names', 'FILE_CHANNEL_MAP_OUT_OF_RANGE'],
|
|
130
138
|
['Microphone permission denied.', 'PERMISSION_DENIED'],
|
|
131
139
|
['Microphone stream is closed', 'MICROPHONE_STREAM_CLOSED'],
|
|
132
140
|
['Speaker stream is closed', 'SPEAKER_STREAM_CLOSED'],
|