decibri 5.4.0 → 5.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -1
- package/MIGRATION.md +8 -12
- package/README.md +22 -11
- package/examples/decibri.browser.js +94 -11
- package/index.d.ts +105 -15
- package/index.js +52 -52
- package/models/README.md +21 -103
- package/models/THIRD-PARTY-NOTICES.md +110 -0
- package/package.json +5 -5
- package/src/browser/decibri-browser.js +177 -15
- package/src/browser/decibri-output-browser.js +6 -0
- package/src/browser/index.d.ts +58 -4
- package/src/browser/worklet-inline.js +2 -2
- package/src/browser/worklet-processor.js +116 -32
- package/src/decibri-output.js +6 -2
- package/src/decibri.d.ts +223 -28
- package/src/decibri.js +254 -42
- package/src/errors.js +12 -2
package/src/decibri.js
CHANGED
|
@@ -102,6 +102,7 @@ class Microphone extends Readable {
|
|
|
102
102
|
this._vad = prepared.vadEnabled;
|
|
103
103
|
this._vadThreshold = prepared.vadThreshold;
|
|
104
104
|
this._vadHoldoff = prepared.vadHoldoff;
|
|
105
|
+
this._dtype = prepared.dtype;
|
|
105
106
|
this._vadScore = 0;
|
|
106
107
|
this._isSpeaking = false;
|
|
107
108
|
this._silenceTimer = null;
|
|
@@ -145,16 +146,42 @@ class Microphone extends Readable {
|
|
|
145
146
|
throw new RangeError('sample rate must be between 1000 and 384000');
|
|
146
147
|
}
|
|
147
148
|
|
|
148
|
-
//
|
|
149
|
-
//
|
|
150
|
-
//
|
|
151
|
-
// additive. The `channels` option is kept for that forward compatibility.
|
|
149
|
+
// The number of channels delivered, interleaved frame by frame. Bounded
|
|
150
|
+
// below here; bounded above by the resolved device alone, which answers
|
|
151
|
+
// when the stream starts. No fixed maximum exists on this path.
|
|
152
152
|
const channels = options.channels ?? 1;
|
|
153
153
|
if (channels < 1) {
|
|
154
|
-
throw new RangeError('channels must be
|
|
154
|
+
throw new RangeError('channels must be at least 1');
|
|
155
155
|
}
|
|
156
|
-
|
|
157
|
-
|
|
156
|
+
|
|
157
|
+
// ── Validate channel map ─────────────────────────────────────────────────
|
|
158
|
+
|
|
159
|
+
// An optional list of 0-based device channel indices, one per delivered
|
|
160
|
+
// channel: delivered channel j carries device channel channelMap[j].
|
|
161
|
+
// Absence delivers the documented average of every opened channel. The
|
|
162
|
+
// checks here are shape-only (an array of integers that fit the channel
|
|
163
|
+
// count's width, with one entry per channel); whether each entry exists on
|
|
164
|
+
// the device is the core's check, made against the resolved device's own
|
|
165
|
+
// report when the stream starts, because only the device can say how many
|
|
166
|
+
// channels it has. No fixed maximum exists on this path.
|
|
167
|
+
const channelMap = options.channelMap;
|
|
168
|
+
if (channelMap !== undefined) {
|
|
169
|
+
if (!Array.isArray(channelMap)) {
|
|
170
|
+
throw new TypeError(
|
|
171
|
+
`Invalid channelMap value: ${JSON.stringify(channelMap)}. Expected an array of 0-based device channel indices, such as [0].`
|
|
172
|
+
);
|
|
173
|
+
}
|
|
174
|
+
for (const entry of channelMap) {
|
|
175
|
+
if (typeof entry !== 'number' || !Number.isInteger(entry)) {
|
|
176
|
+
throw new TypeError('channelMap entries must be integers');
|
|
177
|
+
}
|
|
178
|
+
if (entry < 0 || entry > 65535) {
|
|
179
|
+
throw new RangeError('channelMap entries must be between 0 and 65535');
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
if (channelMap.length !== channels) {
|
|
183
|
+
throw new RangeError('channelMap must have exactly one entry per channel');
|
|
184
|
+
}
|
|
158
185
|
}
|
|
159
186
|
|
|
160
187
|
const framesPerBuffer = options.framesPerBuffer ?? 1600;
|
|
@@ -202,7 +229,7 @@ class Microphone extends Readable {
|
|
|
202
229
|
// vad selects the detector and (optionally) its threshold/holdoff policy.
|
|
203
230
|
// It accepts false (disabled, default), the 'silero'/'energy' shorthand
|
|
204
231
|
// (which uses the mode's default threshold and holdoff), or a config object
|
|
205
|
-
// { model, threshold, holdoffMs } to tune the policy. The legacy two-flag
|
|
232
|
+
// { model, threshold, holdoffMs, source } to tune the policy. The legacy two-flag
|
|
206
233
|
// form (vad: true plus vadMode) and the flat vadThreshold/vadHoldoff
|
|
207
234
|
// options are rejected with a migration error. The threshold and holdoff
|
|
208
235
|
// live JS-side (the state machine runs in this wrapper); only the mode is
|
|
@@ -218,6 +245,7 @@ class Microphone extends Readable {
|
|
|
218
245
|
let vadMode;
|
|
219
246
|
let vadThreshold;
|
|
220
247
|
let vadHoldoff;
|
|
248
|
+
let vadSource;
|
|
221
249
|
if (vad === false) {
|
|
222
250
|
vadEnabled = false;
|
|
223
251
|
vadMode = 'energy'; // inert placeholder; ignored while disabled
|
|
@@ -229,10 +257,13 @@ class Microphone extends Readable {
|
|
|
229
257
|
vadEnabled = true;
|
|
230
258
|
vadMode = vad;
|
|
231
259
|
} else if (vad !== null && typeof vad === 'object' && !Array.isArray(vad)) {
|
|
232
|
-
// Config object form: { model, threshold?, holdoffMs? }. model
|
|
233
|
-
// and selects the detector; threshold and holdoffMs override
|
|
234
|
-
// defaults when supplied
|
|
235
|
-
|
|
260
|
+
// Config object form: { model, threshold?, holdoffMs?, source? }. model
|
|
261
|
+
// is required and selects the detector; threshold and holdoffMs override
|
|
262
|
+
// the mode defaults when supplied; source names the 0-based DELIVERED
|
|
263
|
+
// channel the detector reads (the position within the delivered
|
|
264
|
+
// interleaved frames, after any channelMap), absent feeding the frame
|
|
265
|
+
// average of every delivered channel.
|
|
266
|
+
const { model, threshold, holdoffMs, source } = vad;
|
|
236
267
|
if (model !== 'silero' && model !== 'energy') {
|
|
237
268
|
throw new TypeError(
|
|
238
269
|
`Invalid vad model: ${JSON.stringify(model)}. Expected 'silero' or 'energy'.`
|
|
@@ -258,9 +289,26 @@ class Microphone extends Readable {
|
|
|
258
289
|
}
|
|
259
290
|
vadHoldoff = holdoffMs;
|
|
260
291
|
}
|
|
292
|
+
if (source !== undefined) {
|
|
293
|
+
if (typeof source !== 'number' || !Number.isInteger(source)) {
|
|
294
|
+
throw new TypeError('vad source must be an integer');
|
|
295
|
+
}
|
|
296
|
+
if (source < 0 || source > 65535) {
|
|
297
|
+
throw new RangeError('vad source must be between 0 and 65535');
|
|
298
|
+
}
|
|
299
|
+
// The delivered count is the only ceiling, checked here where both
|
|
300
|
+
// sides are in scope; the message is the core's own for the same
|
|
301
|
+
// condition. No fixed maximum exists.
|
|
302
|
+
if (source >= channels) {
|
|
303
|
+
throw new RangeError(
|
|
304
|
+
`the detector source names delivered channel ${source}; the delivered channel count is ${channels}`
|
|
305
|
+
);
|
|
306
|
+
}
|
|
307
|
+
vadSource = source;
|
|
308
|
+
}
|
|
261
309
|
} else {
|
|
262
310
|
throw new TypeError(
|
|
263
|
-
`Invalid vad value: ${JSON.stringify(vad)}. Expected false, 'silero', 'energy', or a config object { model, threshold, holdoffMs }.`
|
|
311
|
+
`Invalid vad value: ${JSON.stringify(vad)}. Expected false, 'silero', 'energy', or a config object { model, threshold, holdoffMs, source }.`
|
|
264
312
|
);
|
|
265
313
|
}
|
|
266
314
|
|
|
@@ -345,8 +393,8 @@ class Microphone extends Readable {
|
|
|
345
393
|
// ── Validate AEC ─────────────────────────────────────────────────────────
|
|
346
394
|
|
|
347
395
|
// Echo cancellation: the 'tau' shorthand names the model, or an
|
|
348
|
-
// { model, tailMs, suppression, referenceSampleRate }
|
|
349
|
-
// absence leaves it off. The model name is deliberately NOT checked
|
|
396
|
+
// { model, tailMs, suppression, referenceSampleRate, referenceChannels }
|
|
397
|
+
// object tunes it; absence leaves it off. The model name is deliberately NOT checked
|
|
350
398
|
// against a list here: the canceller owns the accepted set, so the native
|
|
351
399
|
// layer parses it (AecModel::from_str) and an unknown name is rejected by
|
|
352
400
|
// the native constructor with the canceller's own message (a DecibriError
|
|
@@ -360,11 +408,12 @@ class Microphone extends Readable {
|
|
|
360
408
|
let aecTailMs;
|
|
361
409
|
let aecSuppression;
|
|
362
410
|
let aecReferenceSampleRate;
|
|
411
|
+
let aecReferenceChannels;
|
|
363
412
|
if (aec !== undefined) {
|
|
364
413
|
if (typeof aec === 'string') {
|
|
365
414
|
aecModel = aec;
|
|
366
415
|
} else if (aec !== null && typeof aec === 'object' && !Array.isArray(aec)) {
|
|
367
|
-
const { model, tailMs, suppression, referenceSampleRate } = aec;
|
|
416
|
+
const { model, tailMs, suppression, referenceSampleRate, referenceChannels } = aec;
|
|
368
417
|
if (typeof model !== 'string') {
|
|
369
418
|
throw new TypeError(
|
|
370
419
|
`Invalid aec model: ${JSON.stringify(model)}. Expected a model name string such as 'tau'.`
|
|
@@ -397,9 +446,18 @@ class Microphone extends Readable {
|
|
|
397
446
|
}
|
|
398
447
|
aecReferenceSampleRate = referenceSampleRate;
|
|
399
448
|
}
|
|
449
|
+
if (referenceChannels !== undefined) {
|
|
450
|
+
if (typeof referenceChannels !== 'number' || Number.isNaN(referenceChannels)) {
|
|
451
|
+
throw new TypeError('aec referenceChannels must be a number');
|
|
452
|
+
}
|
|
453
|
+
if (referenceChannels < 1) {
|
|
454
|
+
throw new RangeError('aec referenceChannels must be at least 1');
|
|
455
|
+
}
|
|
456
|
+
aecReferenceChannels = referenceChannels;
|
|
457
|
+
}
|
|
400
458
|
} else {
|
|
401
459
|
throw new TypeError(
|
|
402
|
-
`Invalid aec value: ${JSON.stringify(aec)}. Expected a model name such as 'tau', or a config object { model, tailMs, suppression, referenceSampleRate }.`
|
|
460
|
+
`Invalid aec value: ${JSON.stringify(aec)}. Expected a model name such as 'tau', or a config object { model, tailMs, suppression, referenceSampleRate, referenceChannels }.`
|
|
403
461
|
);
|
|
404
462
|
}
|
|
405
463
|
}
|
|
@@ -423,6 +481,7 @@ class Microphone extends Readable {
|
|
|
423
481
|
nativeOptions: {
|
|
424
482
|
sampleRate,
|
|
425
483
|
channels,
|
|
484
|
+
channelMap,
|
|
426
485
|
framesPerBuffer,
|
|
427
486
|
format: dtype,
|
|
428
487
|
device: resolvedDevice,
|
|
@@ -431,6 +490,9 @@ class Microphone extends Readable {
|
|
|
431
490
|
// native compute the energy score for a microphone that did not ask for
|
|
432
491
|
// VAD. Absent means VAD off in native.
|
|
433
492
|
vadMode: vadEnabled ? vadMode : undefined,
|
|
493
|
+
// The delivered channel the detector reads, from the vad config
|
|
494
|
+
// object's source key. Absent feeds the frame average.
|
|
495
|
+
detectorSource: vadSource,
|
|
434
496
|
modelPath,
|
|
435
497
|
dcRemoval,
|
|
436
498
|
denoise,
|
|
@@ -443,6 +505,7 @@ class Microphone extends Readable {
|
|
|
443
505
|
aecTailMs,
|
|
444
506
|
aecSuppression,
|
|
445
507
|
aecReferenceSampleRate,
|
|
508
|
+
aecReferenceChannels,
|
|
446
509
|
},
|
|
447
510
|
};
|
|
448
511
|
}
|
|
@@ -603,17 +666,27 @@ class Microphone extends Readable {
|
|
|
603
666
|
|
|
604
667
|
/**
|
|
605
668
|
* Queue far-end reference audio for the echo canceller: the audio being
|
|
606
|
-
* played out, pushed as it is played, in played order. Accepts
|
|
607
|
-
*
|
|
608
|
-
*
|
|
609
|
-
*
|
|
669
|
+
* played out, pushed as it is played, in played order. Accepts a `Buffer`,
|
|
670
|
+
* `Uint8Array`, or `DataView` of PCM bytes in this microphone's `dtype`,
|
|
671
|
+
* or the typed array carrying that dtype (`Int16Array` for `'int16'`,
|
|
672
|
+
* `Float32Array` for `'float32'`), at the declared `referenceSampleRate`
|
|
673
|
+
* (the capture rate when unset), interleaved at the declared
|
|
674
|
+
* `referenceChannels` (mono when unset). A typed array carrying any other
|
|
675
|
+
* sample dtype throws a `TypeError`, whatever the capture state. With
|
|
676
|
+
* `referenceChannels` above 1, each frame is averaged to one mono sample
|
|
677
|
+
* before the canceller sees it; a multichannel reference pushed without
|
|
678
|
+
* declaring the count cancels nothing and reports no error. The declared
|
|
679
|
+
* count must match this buffer's actual interleaving: a mismatch is not
|
|
680
|
+
* detected and raises no error, and shows up only as
|
|
681
|
+
* `aecMetrics().delaySamples` staying `null` with no fault reported.
|
|
610
682
|
*
|
|
611
683
|
* Never blocks and never throws on a full queue: samples that do not fit
|
|
612
684
|
* are discarded and counted by `aecMetrics().referenceDropped`, and the
|
|
613
685
|
* span they occupied is represented as silence. Silence between played
|
|
614
686
|
* audio need not be pushed; a caller that stops pushing has said nothing is
|
|
615
|
-
* playing. A push while capture is not running
|
|
616
|
-
*
|
|
687
|
+
* playing. A push while capture is not running is discarded and counted by
|
|
688
|
+
* `referenceDropped`, read once capture runs; a push with the `aec` option
|
|
689
|
+
* unset is a no-op.
|
|
617
690
|
*
|
|
618
691
|
* @param {Buffer | NodeJS.ArrayBufferView} data PCM samples in the
|
|
619
692
|
* configured `dtype`.
|
|
@@ -623,8 +696,28 @@ class Microphone extends Readable {
|
|
|
623
696
|
if (Buffer.isBuffer(data)) {
|
|
624
697
|
buf = data;
|
|
625
698
|
} else if (ArrayBuffer.isView(data)) {
|
|
626
|
-
//
|
|
627
|
-
// the
|
|
699
|
+
// A typed array names its own sample dtype, so one carrying a dtype
|
|
700
|
+
// other than the configured one is refused rather than read as raw
|
|
701
|
+
// bytes. Buffer, Uint8Array, and DataView are format-agnostic byte
|
|
702
|
+
// carriers, exactly as bytes are on the Python surface; the accepted
|
|
703
|
+
// view is normalized to the same bytes, no copy, exactly as the stream
|
|
704
|
+
// machinery normalizes a typed-array write to a Speaker.
|
|
705
|
+
if (!(data instanceof Uint8Array) && !(data instanceof DataView)) {
|
|
706
|
+
const expected = this._dtype === 'int16' ? Int16Array : Float32Array;
|
|
707
|
+
if (!(data instanceof expected)) {
|
|
708
|
+
const mismatched = this._dtype === 'int16' ? Float32Array : Int16Array;
|
|
709
|
+
if (data instanceof mismatched) {
|
|
710
|
+
const other = this._dtype === 'int16' ? 'float32' : 'int16';
|
|
711
|
+
throw new TypeError(
|
|
712
|
+
`dtype '${this._dtype}' configured but ${mismatched.name} samples were pushed; ` +
|
|
713
|
+
`convert to ${expected.name} or construct Microphone with dtype: '${other}'`
|
|
714
|
+
);
|
|
715
|
+
}
|
|
716
|
+
throw new TypeError(
|
|
717
|
+
'pushAecReference requires a Buffer, TypedArray, or DataView of PCM samples in the configured dtype'
|
|
718
|
+
);
|
|
719
|
+
}
|
|
720
|
+
}
|
|
628
721
|
buf = Buffer.from(data.buffer, data.byteOffset, data.byteLength);
|
|
629
722
|
} else {
|
|
630
723
|
throw new TypeError(
|
|
@@ -645,6 +738,11 @@ class Microphone extends Readable {
|
|
|
645
738
|
* `referenceDropped` means single pushes are exceeding the reference
|
|
646
739
|
* queue's bound.
|
|
647
740
|
*
|
|
741
|
+
* The top-level engine fields report the first delivered channel's
|
|
742
|
+
* canceller; `channels` carries every delivered channel's report in
|
|
743
|
+
* delivered order, one entry per channel, so the two agree on a
|
|
744
|
+
* single-channel stream.
|
|
745
|
+
*
|
|
648
746
|
* @returns {import('./decibri').AecMetrics | null}
|
|
649
747
|
*/
|
|
650
748
|
aecMetrics() {
|
|
@@ -659,6 +757,14 @@ class Microphone extends Readable {
|
|
|
659
757
|
referenceReanchors: m.referenceReanchors,
|
|
660
758
|
referenceDropped: m.referenceDropped,
|
|
661
759
|
referenceSilence: m.referenceSilence,
|
|
760
|
+
channels: m.channels.map((c) => ({
|
|
761
|
+
delaySamples: c.delaySamples ?? null,
|
|
762
|
+
erleDb: c.erleDb,
|
|
763
|
+
doubleTalk: c.doubleTalk,
|
|
764
|
+
referenceStarved: c.referenceStarved,
|
|
765
|
+
acquisitionParked: c.acquisitionParked,
|
|
766
|
+
referenceReanchors: c.referenceReanchors,
|
|
767
|
+
})),
|
|
662
768
|
};
|
|
663
769
|
}
|
|
664
770
|
|
|
@@ -737,6 +843,15 @@ class File extends Readable {
|
|
|
737
843
|
constructor(filePath, options = {}, _internal = undefined) {
|
|
738
844
|
super({ highWaterMark: options.highWaterMark, objectMode: false });
|
|
739
845
|
|
|
846
|
+
// The open path reads the source's channel count from the file's own
|
|
847
|
+
// header, so a caller-stated interleave has nothing to describe here;
|
|
848
|
+
// it is refused rather than ignored.
|
|
849
|
+
if (!_internal && options.inputChannels !== undefined) {
|
|
850
|
+
throw new TypeError(
|
|
851
|
+
'inputChannels applies only to File.buffer; a file carries its channel count in its own header'
|
|
852
|
+
);
|
|
853
|
+
}
|
|
854
|
+
|
|
740
855
|
const prepared = _internal ? _internal.prepared : File._prepareOptions(options);
|
|
741
856
|
|
|
742
857
|
// ── Store config ───────────────────────────────────────────────────────
|
|
@@ -754,6 +869,9 @@ class File extends Readable {
|
|
|
754
869
|
this._silenceStartPos = null;
|
|
755
870
|
this._position = 0;
|
|
756
871
|
this._sampleRate = prepared.nativeOptions.sampleRate;
|
|
872
|
+
// The delivered channel count: file time advances by frames, so the
|
|
873
|
+
// interleaved sample count divides by it before it divides by the rate.
|
|
874
|
+
this._channels = prepared.channels;
|
|
757
875
|
this._bytesPerSample = prepared.dtype === 'int16' ? 2 : 4;
|
|
758
876
|
this._ended = false;
|
|
759
877
|
// Set the moment the consumer asks the stream for data, which is earlier
|
|
@@ -792,6 +910,42 @@ class File extends Readable {
|
|
|
792
910
|
throw new RangeError('sample rate must be between 1000 and 384000');
|
|
793
911
|
}
|
|
794
912
|
|
|
913
|
+
// The number of channels delivered, interleaved frame by frame. Bounded
|
|
914
|
+
// below here; bounded above by the source's own channel count alone,
|
|
915
|
+
// which the core reads from the header (or takes from inputChannels)
|
|
916
|
+
// when the source is opened. No fixed maximum exists on this path.
|
|
917
|
+
const channels = options.channels ?? 1;
|
|
918
|
+
if (channels < 1) {
|
|
919
|
+
throw new RangeError('channels must be at least 1');
|
|
920
|
+
}
|
|
921
|
+
|
|
922
|
+
// An optional list of 0-based source channel indices, one per delivered
|
|
923
|
+
// channel: delivered channel j carries source channel channelMap[j].
|
|
924
|
+
// Absence delivers the documented average of every source channel. The
|
|
925
|
+
// checks here are shape-only, exactly as the Microphone's: whether each
|
|
926
|
+
// entry exists on the source is the core's check, made against the
|
|
927
|
+
// source's own count, because only the opened source can say how many
|
|
928
|
+
// channels it has. No fixed maximum exists on this path.
|
|
929
|
+
const channelMap = options.channelMap;
|
|
930
|
+
if (channelMap !== undefined) {
|
|
931
|
+
if (!Array.isArray(channelMap)) {
|
|
932
|
+
throw new TypeError(
|
|
933
|
+
`Invalid channelMap value: ${JSON.stringify(channelMap)}. Expected an array of 0-based source channel indices, such as [0].`
|
|
934
|
+
);
|
|
935
|
+
}
|
|
936
|
+
for (const entry of channelMap) {
|
|
937
|
+
if (typeof entry !== 'number' || !Number.isInteger(entry)) {
|
|
938
|
+
throw new TypeError('channelMap entries must be integers');
|
|
939
|
+
}
|
|
940
|
+
if (entry < 0 || entry > 65535) {
|
|
941
|
+
throw new RangeError('channelMap entries must be between 0 and 65535');
|
|
942
|
+
}
|
|
943
|
+
}
|
|
944
|
+
if (channelMap.length !== channels) {
|
|
945
|
+
throw new RangeError('channelMap must have exactly one entry per channel');
|
|
946
|
+
}
|
|
947
|
+
}
|
|
948
|
+
|
|
795
949
|
const dtype = options.dtype ?? 'int16';
|
|
796
950
|
if (dtype !== 'int16' && dtype !== 'float32') {
|
|
797
951
|
throw new TypeError("dtype must be 'int16' or 'float32'");
|
|
@@ -812,6 +966,7 @@ class File extends Readable {
|
|
|
812
966
|
let vadMode;
|
|
813
967
|
let vadThreshold;
|
|
814
968
|
let vadHoldoff;
|
|
969
|
+
let vadSource;
|
|
815
970
|
if (vad === false) {
|
|
816
971
|
vadEnabled = false;
|
|
817
972
|
vadMode = 'energy'; // inert placeholder; ignored while disabled
|
|
@@ -823,7 +978,7 @@ class File extends Readable {
|
|
|
823
978
|
vadEnabled = true;
|
|
824
979
|
vadMode = vad;
|
|
825
980
|
} else if (vad !== null && typeof vad === 'object' && !Array.isArray(vad)) {
|
|
826
|
-
const { model, threshold, holdoffMs } = vad;
|
|
981
|
+
const { model, threshold, holdoffMs, source } = vad;
|
|
827
982
|
if (model !== 'silero' && model !== 'energy') {
|
|
828
983
|
throw new TypeError(
|
|
829
984
|
`Invalid vad model: ${JSON.stringify(model)}. Expected 'silero' or 'energy'.`
|
|
@@ -849,9 +1004,25 @@ class File extends Readable {
|
|
|
849
1004
|
}
|
|
850
1005
|
vadHoldoff = holdoffMs;
|
|
851
1006
|
}
|
|
1007
|
+
// source names the 0-based DELIVERED channel the detector reads,
|
|
1008
|
+
// exactly as on Microphone; the checks and messages are the same.
|
|
1009
|
+
if (source !== undefined) {
|
|
1010
|
+
if (typeof source !== 'number' || !Number.isInteger(source)) {
|
|
1011
|
+
throw new TypeError('vad source must be an integer');
|
|
1012
|
+
}
|
|
1013
|
+
if (source < 0 || source > 65535) {
|
|
1014
|
+
throw new RangeError('vad source must be between 0 and 65535');
|
|
1015
|
+
}
|
|
1016
|
+
if (source >= channels) {
|
|
1017
|
+
throw new RangeError(
|
|
1018
|
+
`the detector source names delivered channel ${source}; the delivered channel count is ${channels}`
|
|
1019
|
+
);
|
|
1020
|
+
}
|
|
1021
|
+
vadSource = source;
|
|
1022
|
+
}
|
|
852
1023
|
} else {
|
|
853
1024
|
throw new TypeError(
|
|
854
|
-
`Invalid vad value: ${JSON.stringify(vad)}. Expected false, 'silero', 'energy', or a config object { model, threshold, holdoffMs }.`
|
|
1025
|
+
`Invalid vad value: ${JSON.stringify(vad)}. Expected false, 'silero', 'energy', or a config object { model, threshold, holdoffMs, source }.`
|
|
855
1026
|
);
|
|
856
1027
|
}
|
|
857
1028
|
|
|
@@ -905,16 +1076,22 @@ class File extends Readable {
|
|
|
905
1076
|
|
|
906
1077
|
return {
|
|
907
1078
|
dtype,
|
|
1079
|
+
channels,
|
|
908
1080
|
vadEnabled,
|
|
909
1081
|
vadMode,
|
|
910
1082
|
vadThreshold: vadThreshold ?? (vadMode === 'silero' ? 0.5 : 0.01),
|
|
911
1083
|
vadHoldoff: vadHoldoff ?? 300,
|
|
912
1084
|
nativeOptions: {
|
|
913
1085
|
sampleRate,
|
|
1086
|
+
channels,
|
|
1087
|
+
channelMap,
|
|
914
1088
|
format: dtype,
|
|
915
1089
|
// Pass the mode to native only when VAD is enabled, exactly as the
|
|
916
1090
|
// Microphone options do; absent means VAD off in native.
|
|
917
1091
|
vadMode: vadEnabled ? vadMode : undefined,
|
|
1092
|
+
// The delivered channel the detector reads, from the vad config
|
|
1093
|
+
// object's source key. Absent feeds the frame average.
|
|
1094
|
+
detectorSource: vadSource,
|
|
918
1095
|
// The whole-file analysis applies threshold and holdoff in the core
|
|
919
1096
|
// (segment merging in file time), so both cross the boundary here,
|
|
920
1097
|
// unlike the live path where the policy is wrapper-only.
|
|
@@ -946,6 +1123,13 @@ class File extends Readable {
|
|
|
946
1123
|
if (typeof filePath !== 'string') {
|
|
947
1124
|
throw new TypeError('path must be a string');
|
|
948
1125
|
}
|
|
1126
|
+
// The same refusal the synchronous constructor makes: a path's channel
|
|
1127
|
+
// count comes from its own header.
|
|
1128
|
+
if (options.inputChannels !== undefined) {
|
|
1129
|
+
throw new TypeError(
|
|
1130
|
+
'inputChannels applies only to File.buffer; a file carries its channel count in its own header'
|
|
1131
|
+
);
|
|
1132
|
+
}
|
|
949
1133
|
const prepared = File._prepareOptions(options);
|
|
950
1134
|
let native;
|
|
951
1135
|
try {
|
|
@@ -958,12 +1142,13 @@ class File extends Readable {
|
|
|
958
1142
|
|
|
959
1143
|
/**
|
|
960
1144
|
* Wrap in-memory samples as an offline source. `samples` must be a
|
|
961
|
-
* `Float32Array` of
|
|
962
|
-
*
|
|
963
|
-
*
|
|
964
|
-
*
|
|
965
|
-
*
|
|
966
|
-
*
|
|
1145
|
+
* `Float32Array` of samples in [-1.0, 1.0], frame-interleaved at
|
|
1146
|
+
* `inputChannels` (1, mono, by default); a raw `Buffer` of PCM bytes is
|
|
1147
|
+
* rejected as ambiguous (encoded bytes, int16 PCM, and f32 samples are
|
|
1148
|
+
* indistinguishable, and decibri's own capture output is a `Buffer`). Raw
|
|
1149
|
+
* samples carry no header, so `inputRate` (their native rate) is
|
|
1150
|
+
* required; `sampleRate` stays the target output rate. No I/O, so
|
|
1151
|
+
* construction is synchronous.
|
|
967
1152
|
*
|
|
968
1153
|
* @param {Float32Array} samples
|
|
969
1154
|
* @param {import('./decibri').FileBufferOptions} [options]
|
|
@@ -985,10 +1170,23 @@ class File extends Readable {
|
|
|
985
1170
|
if (inputRate < 1000 || inputRate > 384000) {
|
|
986
1171
|
throw new RangeError('inputRate must be between 1000 and 384000');
|
|
987
1172
|
}
|
|
1173
|
+
// The channel counterpart of inputRate: the interleave of the caller's
|
|
1174
|
+
// own samples. Shape-checked here; whether the samples divide into
|
|
1175
|
+
// whole frames at this count is the core's check.
|
|
1176
|
+
const inputChannels = options.inputChannels ?? 1;
|
|
1177
|
+
if (typeof inputChannels !== 'number' || !Number.isInteger(inputChannels)) {
|
|
1178
|
+
throw new TypeError('inputChannels must be an integer');
|
|
1179
|
+
}
|
|
1180
|
+
if (inputChannels < 1 || inputChannels > 65535) {
|
|
1181
|
+
throw new RangeError('inputChannels must be between 1 and 65535');
|
|
1182
|
+
}
|
|
988
1183
|
const prepared = File._prepareOptions(options);
|
|
989
1184
|
let native;
|
|
990
1185
|
try {
|
|
991
|
-
native = FileHandle.buffer(samples, inputRate,
|
|
1186
|
+
native = FileHandle.buffer(samples, inputRate, {
|
|
1187
|
+
...prepared.nativeOptions,
|
|
1188
|
+
inputChannels,
|
|
1189
|
+
});
|
|
992
1190
|
} catch (err) {
|
|
993
1191
|
throw wrapNativeError(err);
|
|
994
1192
|
}
|
|
@@ -1076,7 +1274,9 @@ class File extends Readable {
|
|
|
1076
1274
|
// before the opt-in conditioning step, exactly as the live pump does.
|
|
1077
1275
|
this._processVadValue(this._native.vadProbability, chunk.length);
|
|
1078
1276
|
} else {
|
|
1079
|
-
|
|
1277
|
+
// Bytes to interleaved samples to frames to seconds of file time.
|
|
1278
|
+
this._position +=
|
|
1279
|
+
chunk.length / this._bytesPerSample / this._channels / this._sampleRate;
|
|
1080
1280
|
}
|
|
1081
1281
|
this.push(chunk);
|
|
1082
1282
|
}
|
|
@@ -1090,7 +1290,9 @@ class File extends Readable {
|
|
|
1090
1290
|
*/
|
|
1091
1291
|
_processVadValue(value, chunkBytes) {
|
|
1092
1292
|
const chunkStart = this._position;
|
|
1093
|
-
|
|
1293
|
+
// Bytes to interleaved samples to frames to seconds of file time.
|
|
1294
|
+
const chunkEnd =
|
|
1295
|
+
chunkStart + chunkBytes / this._bytesPerSample / this._channels / this._sampleRate;
|
|
1094
1296
|
this._position = chunkEnd;
|
|
1095
1297
|
this._vadScore = value;
|
|
1096
1298
|
if (value >= this._vadThreshold) {
|
|
@@ -1283,7 +1485,9 @@ class AudioWriter extends Writable {
|
|
|
1283
1485
|
*
|
|
1284
1486
|
* Chunks are raw PCM bytes in `dtype` ('int16' little-endian by default,
|
|
1285
1487
|
* matching what a `File` or `Microphone` emits; 'float32' for raw f32
|
|
1286
|
-
* bytes)
|
|
1488
|
+
* bytes), frame-interleaved at `channels` (1, mono, by default; the
|
|
1489
|
+
* stream's total sample count must divide into whole frames, and each
|
|
1490
|
+
* container's own channel ceiling applies at the write). `sampleRate` is
|
|
1287
1491
|
* required, because raw audio carries no header to read one from.
|
|
1288
1492
|
*
|
|
1289
1493
|
* The file is written when the stream finishes ('finish' fires after the
|
|
@@ -1306,9 +1510,12 @@ class AudioWriter extends Writable {
|
|
|
1306
1510
|
if (sampleRate < 1000 || sampleRate > 384000) {
|
|
1307
1511
|
throw new RangeError('sample rate must be between 1000 and 384000');
|
|
1308
1512
|
}
|
|
1309
|
-
|
|
1310
|
-
|
|
1311
|
-
|
|
1513
|
+
// Bounded below here; above, each container's own ceiling answers at
|
|
1514
|
+
// the write, with the container layer's own message. No decibri-side
|
|
1515
|
+
// maximum exists on this path.
|
|
1516
|
+
const channels = options.channels ?? 1;
|
|
1517
|
+
if (channels < 1) {
|
|
1518
|
+
throw new RangeError('channels must be at least 1');
|
|
1312
1519
|
}
|
|
1313
1520
|
const dtype = options.dtype ?? 'int16';
|
|
1314
1521
|
if (dtype !== 'int16' && dtype !== 'float32') {
|
|
@@ -1319,6 +1526,7 @@ class AudioWriter extends Writable {
|
|
|
1319
1526
|
this._saveOptions = File._prepareSaveOptions(options);
|
|
1320
1527
|
this._filePath = filePath;
|
|
1321
1528
|
this._sampleRate = sampleRate;
|
|
1529
|
+
this._channels = channels;
|
|
1322
1530
|
this._dtype = dtype;
|
|
1323
1531
|
this._chunks = [];
|
|
1324
1532
|
this._report = null;
|
|
@@ -1366,13 +1574,17 @@ class AudioWriter extends Writable {
|
|
|
1366
1574
|
samples[i] = bytes.readFloatLE(i * 4);
|
|
1367
1575
|
}
|
|
1368
1576
|
}
|
|
1369
|
-
// The write is File.save on a source at the writer's own rate
|
|
1370
|
-
// encode path, so the two spellings produce the
|
|
1577
|
+
// The write is File.save on a source at the writer's own rate and
|
|
1578
|
+
// interleave: the same encode path, so the two spellings produce the
|
|
1579
|
+
// same bytes. A stream that does not divide into whole frames is the
|
|
1580
|
+
// core's refusal, surfaced here when the stream finishes.
|
|
1371
1581
|
let file;
|
|
1372
1582
|
try {
|
|
1373
1583
|
file = File.buffer(samples, {
|
|
1374
1584
|
inputRate: this._sampleRate,
|
|
1585
|
+
inputChannels: this._channels,
|
|
1375
1586
|
sampleRate: this._sampleRate,
|
|
1587
|
+
channels: this._channels,
|
|
1376
1588
|
});
|
|
1377
1589
|
} catch (err) {
|
|
1378
1590
|
callback(err);
|
package/src/errors.js
CHANGED
|
@@ -75,14 +75,17 @@ class OrtPathError extends OrtError {
|
|
|
75
75
|
// the core ever sees it).
|
|
76
76
|
const RANGE_PREFIXES = [
|
|
77
77
|
'sample rate must be between',
|
|
78
|
-
'channels must be
|
|
79
|
-
'multichannel capture is not supported',
|
|
78
|
+
'channels must be at least',
|
|
80
79
|
'frames per buffer must be between',
|
|
81
80
|
'agc target level must be between',
|
|
82
81
|
'limiter ceiling must be between',
|
|
82
|
+
'the detector source names',
|
|
83
|
+
'the channel map has',
|
|
84
|
+
'the requested block size',
|
|
83
85
|
'flac compression level must be between',
|
|
84
86
|
'aec tailMs must be between',
|
|
85
87
|
'aec referenceSampleRate must be between',
|
|
88
|
+
'aec referenceChannels must be at',
|
|
86
89
|
'Silero VAD only supports',
|
|
87
90
|
'VAD threshold must be between',
|
|
88
91
|
'echo cancellation only supports',
|
|
@@ -125,6 +128,13 @@ const BASE_CODES = [
|
|
|
125
128
|
['audio stream is already running', 'ALREADY_RUNNING'],
|
|
126
129
|
['Failed to open audio stream', 'STREAM_OPEN_FAILED'],
|
|
127
130
|
['Failed to start audio stream', 'STREAM_START_FAILED'],
|
|
131
|
+
['the output device does not support', 'SPEAKER_CHANNELS_UNSUPPORTED'],
|
|
132
|
+
['the input device does not support', 'MICROPHONE_CHANNELS_UNSUPPORTED'],
|
|
133
|
+
['a channel map is required', 'CHANNEL_SELECTION_AMBIGUOUS'],
|
|
134
|
+
['the channel map names', 'CHANNEL_MAP_OUT_OF_RANGE'],
|
|
135
|
+
['the file does not have', 'FILE_CHANNELS_UNSUPPORTED'],
|
|
136
|
+
['delivering', 'FILE_CHANNEL_SELECTION_AMBIGUOUS'],
|
|
137
|
+
['the file channel map names', 'FILE_CHANNEL_MAP_OUT_OF_RANGE'],
|
|
128
138
|
['Microphone permission denied.', 'PERMISSION_DENIED'],
|
|
129
139
|
['Microphone stream is closed', 'MICROPHONE_STREAM_CLOSED'],
|
|
130
140
|
['Speaker stream is closed', 'SPEAKER_STREAM_CLOSED'],
|