decibri 5.4.0 → 5.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -1
- package/MIGRATION.md +8 -12
- package/README.md +22 -11
- package/examples/decibri.browser.js +94 -11
- package/index.d.ts +105 -15
- package/index.js +52 -52
- package/models/README.md +21 -103
- package/models/THIRD-PARTY-NOTICES.md +110 -0
- package/package.json +5 -5
- package/src/browser/decibri-browser.js +177 -15
- package/src/browser/decibri-output-browser.js +6 -0
- package/src/browser/index.d.ts +58 -4
- package/src/browser/worklet-inline.js +2 -2
- package/src/browser/worklet-processor.js +116 -32
- package/src/decibri-output.js +6 -2
- package/src/decibri.d.ts +223 -28
- package/src/decibri.js +254 -42
- package/src/errors.js +12 -2
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
# Third-Party Notices
|
|
2
|
+
|
|
3
|
+
This directory redistributes the model files listed below, together with the
|
|
4
|
+
license notices they require. The notices ship inside the published npm and
|
|
5
|
+
PyPI packages alongside the model weights they cover, so the attribution
|
|
6
|
+
travels with the files. `README.md` beside this file documents each model's
|
|
7
|
+
tensor interface.
|
|
8
|
+
|
|
9
|
+
## Silero VAD
|
|
10
|
+
|
|
11
|
+
- **Name:** Silero VAD
|
|
12
|
+
- **Version:** v6.2
|
|
13
|
+
- **License:** MIT
|
|
14
|
+
- **Source:** https://github.com/snakers4/silero-vad (release `v6.2`)
|
|
15
|
+
- **Files covered:** `silero_vad.onnx`
|
|
16
|
+
|
|
17
|
+
### License Notice
|
|
18
|
+
|
|
19
|
+
This model is a third-party artifact, not proprietary to decibri. It is
|
|
20
|
+
distributed under the MIT License, reproduced in full below.
|
|
21
|
+
|
|
22
|
+
```text
|
|
23
|
+
MIT License
|
|
24
|
+
|
|
25
|
+
Copyright (c) 2020-present Silero Team
|
|
26
|
+
|
|
27
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
28
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
29
|
+
in the Software without restriction, including without limitation the rights
|
|
30
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
31
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
32
|
+
furnished to do so, subject to the following conditions:
|
|
33
|
+
|
|
34
|
+
The above copyright notice and this permission notice shall be included in all
|
|
35
|
+
copies or substantial portions of the Software.
|
|
36
|
+
|
|
37
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
38
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
39
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
40
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
41
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
42
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
43
|
+
SOFTWARE.
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
## FastEnhancer
|
|
47
|
+
|
|
48
|
+
- **Name:** FastEnhancer-T (tiny tier), VoiceBank-DEMAND checkpoint, waveform variant
|
|
49
|
+
- **Version:** `onnx-vd-v1.0.0`
|
|
50
|
+
- **License:** MIT
|
|
51
|
+
- **Source:** https://github.com/aask1357/fastenhancer (release `onnx-vd-v1.0.0`)
|
|
52
|
+
- **Files covered:** `fastenhancer_t.onnx`
|
|
53
|
+
|
|
54
|
+
### License Notice
|
|
55
|
+
|
|
56
|
+
This model is a third-party artifact, not proprietary to decibri. The model
|
|
57
|
+
code and weights are distributed under the MIT License, reproduced in full
|
|
58
|
+
below.
|
|
59
|
+
|
|
60
|
+
```text
|
|
61
|
+
MIT License
|
|
62
|
+
|
|
63
|
+
Copyright (c) 2025 AHN Sung Hwan
|
|
64
|
+
|
|
65
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
66
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
67
|
+
in the Software without restriction, including without limitation the rights
|
|
68
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
69
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
70
|
+
furnished to do so, subject to the following conditions:
|
|
71
|
+
|
|
72
|
+
The above copyright notice and this permission notice shall be included in all
|
|
73
|
+
copies or substantial portions of the Software.
|
|
74
|
+
|
|
75
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
76
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
77
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
78
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
79
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
80
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
81
|
+
SOFTWARE.
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
### Training-Data Attribution
|
|
85
|
+
|
|
86
|
+
The bundled checkpoint is trained on the VoiceBank-DEMAND noisy speech dataset,
|
|
87
|
+
which pairs clean speech from the CSTR VCTK Corpus with noise from the DEMAND
|
|
88
|
+
database. Each source dataset requires attribution:
|
|
89
|
+
|
|
90
|
+
- **VoiceBank-DEMAND.** Valentini-Botinhao, Cassia. (2017). Noisy speech
|
|
91
|
+
database for training speech enhancement algorithms and TTS models, 2016
|
|
92
|
+
[sound]. University of Edinburgh, School of Informatics, Centre for Speech
|
|
93
|
+
Technology Research (CSTR). Licensed under Creative Commons Attribution 4.0
|
|
94
|
+
International (CC BY 4.0), https://creativecommons.org/licenses/by/4.0/.
|
|
95
|
+
https://doi.org/10.7488/ds/2117
|
|
96
|
+
|
|
97
|
+
- **CSTR VCTK Corpus (version 0.92).** Yamagishi, Junichi; Veaux, Christophe;
|
|
98
|
+
MacDonald, Kirsten. (2019). CSTR VCTK Corpus: English Multi-speaker Corpus
|
|
99
|
+
for CSTR Voice Cloning Toolkit (version 0.92) [sound]. University of
|
|
100
|
+
Edinburgh, Centre for Speech Technology Research (CSTR). Licensed under the
|
|
101
|
+
Open Data Commons Attribution License (ODC-By) v1.0,
|
|
102
|
+
https://opendatacommons.org/licenses/by/1-0/.
|
|
103
|
+
https://doi.org/10.7488/ds/2645
|
|
104
|
+
|
|
105
|
+
- **DEMAND.** Thiemann, Joachim; Ito, Nobutaka; Vincent, Emmanuel. (2013).
|
|
106
|
+
DEMAND: a collection of multi-channel recordings of acoustic noise in
|
|
107
|
+
diverse environments. Licensed under Creative Commons
|
|
108
|
+
Attribution-ShareAlike 3.0 Unported (CC BY-SA 3.0),
|
|
109
|
+
https://creativecommons.org/licenses/by-sa/3.0/.
|
|
110
|
+
https://doi.org/10.5281/zenodo.1227121
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "decibri",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.6.0",
|
|
4
4
|
"description": "Cross-platform audio capture, playback, and processing for Node.js and browsers",
|
|
5
5
|
"main": "src/decibri.js",
|
|
6
6
|
"types": "src/decibri.d.ts",
|
|
@@ -72,10 +72,10 @@
|
|
|
72
72
|
"MIGRATION.md"
|
|
73
73
|
],
|
|
74
74
|
"optionalDependencies": {
|
|
75
|
-
"@decibri/decibri-win32-x64-msvc": "5.
|
|
76
|
-
"@decibri/decibri-darwin-arm64": "5.
|
|
77
|
-
"@decibri/decibri-linux-x64-gnu": "5.
|
|
78
|
-
"@decibri/decibri-linux-arm64-gnu": "5.
|
|
75
|
+
"@decibri/decibri-win32-x64-msvc": "5.6.0",
|
|
76
|
+
"@decibri/decibri-darwin-arm64": "5.6.0",
|
|
77
|
+
"@decibri/decibri-linux-x64-gnu": "5.6.0",
|
|
78
|
+
"@decibri/decibri-linux-arm64-gnu": "5.6.0"
|
|
79
79
|
},
|
|
80
80
|
"devDependencies": {
|
|
81
81
|
"@napi-rs/cli": "^3.7.0"
|
|
@@ -6,13 +6,15 @@ const { WORKLET_SOURCE } = require('./worklet-inline.js');
|
|
|
6
6
|
// Browser build version. Keep in sync with package.json on each release; the
|
|
7
7
|
// browser bundle cannot read package.json at runtime the way the Node wrapper
|
|
8
8
|
// does, so this is a maintained constant.
|
|
9
|
-
const VERSION = '5.
|
|
9
|
+
const VERSION = '5.6.0';
|
|
10
10
|
|
|
11
11
|
/**
|
|
12
12
|
* Browser microphone capture.
|
|
13
13
|
*
|
|
14
14
|
* Uses getUserMedia + AudioWorklet for real-time audio capture in browsers.
|
|
15
|
-
* Emits 'data' events with Int16Array or Float32Array chunks
|
|
15
|
+
* Emits 'data' events with Int16Array or Float32Array chunks holding
|
|
16
|
+
* framesPerBuffer frames of the delivered channel count, interleaved frame
|
|
17
|
+
* by frame.
|
|
16
18
|
*
|
|
17
19
|
* Ported from decibri-web decibri.ts. Logic identical, types removed.
|
|
18
20
|
*
|
|
@@ -39,10 +41,10 @@ class Microphone extends Emitter {
|
|
|
39
41
|
|
|
40
42
|
// ── VAD state ─────────────────────────────────────────────────────────
|
|
41
43
|
// vad accepts false (disabled, default), the 'energy' shorthand, or a config
|
|
42
|
-
// object { model: 'energy', threshold?, holdoffMs? } to tune the
|
|
43
|
-
// browser runs energy VAD only; Silero needs ONNX Runtime,
|
|
44
|
-
// Node-only. The legacy vad: true form and the flat
|
|
45
|
-
// options are rejected with a migration error.
|
|
44
|
+
// object { model: 'energy', threshold?, holdoffMs?, source? } to tune the
|
|
45
|
+
// policy. The browser runs energy VAD only; Silero needs ONNX Runtime,
|
|
46
|
+
// which is Node-only. The legacy vad: true form and the flat
|
|
47
|
+
// vadThreshold/vadHoldoff options are rejected with a migration error.
|
|
46
48
|
if (options.vadThreshold !== undefined || options.vadHoldoff !== undefined) {
|
|
47
49
|
throw new TypeError(
|
|
48
50
|
"vadThreshold and vadHoldoff are no longer supported. Pass them on the vad config object: vad: { model: 'energy', threshold: 0.01, holdoffMs: 300 }."
|
|
@@ -51,6 +53,7 @@ class Microphone extends Emitter {
|
|
|
51
53
|
const vad = options.vad ?? false;
|
|
52
54
|
let vadThreshold = 0.01;
|
|
53
55
|
let vadHoldoff = 300;
|
|
56
|
+
let vadSource;
|
|
54
57
|
if (vad === false) {
|
|
55
58
|
this._vad = false;
|
|
56
59
|
} else if (vad === true) {
|
|
@@ -82,11 +85,34 @@ class Microphone extends Emitter {
|
|
|
82
85
|
}
|
|
83
86
|
vadHoldoff = vad.holdoffMs;
|
|
84
87
|
}
|
|
88
|
+
// source names the 0-based DELIVERED channel the detector reads, the
|
|
89
|
+
// position within the delivered interleaved frames after any
|
|
90
|
+
// channelMap; absent scores the frame average of every delivered
|
|
91
|
+
// channel. The checks and the messages are the node entry's. The
|
|
92
|
+
// delivered-count check runs only against a valid count; a count below
|
|
93
|
+
// one is reported as its own error below, exactly as on the node entry.
|
|
94
|
+
if (vad.source !== undefined) {
|
|
95
|
+
const source = vad.source;
|
|
96
|
+
if (typeof source !== 'number' || !Number.isInteger(source)) {
|
|
97
|
+
throw new TypeError('vad source must be an integer');
|
|
98
|
+
}
|
|
99
|
+
if (source < 0 || source > 65535) {
|
|
100
|
+
throw new RangeError('vad source must be between 0 and 65535');
|
|
101
|
+
}
|
|
102
|
+
const deliveredChannels = options.channels ?? 1;
|
|
103
|
+
if (deliveredChannels >= 1 && source >= deliveredChannels) {
|
|
104
|
+
throw new RangeError(
|
|
105
|
+
`the detector source names delivered channel ${source}; the delivered channel count is ${deliveredChannels}`
|
|
106
|
+
);
|
|
107
|
+
}
|
|
108
|
+
vadSource = source;
|
|
109
|
+
}
|
|
85
110
|
} else {
|
|
86
|
-
throw new TypeError(`Invalid vad value: ${JSON.stringify(vad)}. Expected false, 'energy', or a config object { model, threshold, holdoffMs }.`);
|
|
111
|
+
throw new TypeError(`Invalid vad value: ${JSON.stringify(vad)}. Expected false, 'energy', or a config object { model, threshold, holdoffMs, source }.`);
|
|
87
112
|
}
|
|
88
113
|
this._vadThreshold = vadThreshold;
|
|
89
114
|
this._vadHoldoff = vadHoldoff;
|
|
115
|
+
this._vadSource = vadSource;
|
|
90
116
|
this._vadScore = 0;
|
|
91
117
|
this._isSpeaking = false;
|
|
92
118
|
this._silenceTimer = null;
|
|
@@ -94,6 +120,7 @@ class Microphone extends Emitter {
|
|
|
94
120
|
// ── Options ───────────────────────────────────────────────────────────
|
|
95
121
|
this._sampleRate = options.sampleRate ?? 16000;
|
|
96
122
|
this._channels = options.channels ?? 1;
|
|
123
|
+
this._channelMap = options.channelMap;
|
|
97
124
|
this._framesPerBuffer = options.framesPerBuffer ?? 1600;
|
|
98
125
|
this._device = options.device;
|
|
99
126
|
this._dtype = options.dtype ?? 'int16';
|
|
@@ -105,8 +132,44 @@ class Microphone extends Emitter {
|
|
|
105
132
|
if (this._sampleRate < 1000 || this._sampleRate > 384000) {
|
|
106
133
|
throw new RangeError('sample rate must be between 1000 and 384000');
|
|
107
134
|
}
|
|
108
|
-
|
|
109
|
-
|
|
135
|
+
// The number of channels delivered, interleaved frame by frame. Bounded
|
|
136
|
+
// below here; bounded above by the granted track alone, which answers
|
|
137
|
+
// when the stream starts. No fixed maximum exists on this path. The
|
|
138
|
+
// classes and the messages are the node entry's, so the same value is
|
|
139
|
+
// rejected the same way in both runtimes.
|
|
140
|
+
if (this._channels < 1) {
|
|
141
|
+
throw new RangeError('channels must be at least 1');
|
|
142
|
+
}
|
|
143
|
+
// An optional list of 0-based device channel indices, one per delivered
|
|
144
|
+
// channel: delivered channel j carries device channel channelMap[j].
|
|
145
|
+
// Entries may repeat and may appear in any order, so a map both selects
|
|
146
|
+
// and permutes. Absence derives the delivered channels from the count
|
|
147
|
+
// alone: 1 delivers the documented average of every granted channel, and
|
|
148
|
+
// the granted count delivers every granted channel in granted order. The
|
|
149
|
+
// checks here are shape-only (an array of integers that fit the channel
|
|
150
|
+
// count's width, with one entry per channel), the node entry's classes
|
|
151
|
+
// and messages; whether each entry exists on the track is checked when
|
|
152
|
+
// the stream starts, against the granted track's own report, because
|
|
153
|
+
// only the grant can say how many channels it carries. No fixed maximum
|
|
154
|
+
// exists on this path.
|
|
155
|
+
const channelMap = this._channelMap;
|
|
156
|
+
if (channelMap !== undefined) {
|
|
157
|
+
if (!Array.isArray(channelMap)) {
|
|
158
|
+
throw new TypeError(
|
|
159
|
+
`Invalid channelMap value: ${JSON.stringify(channelMap)}. Expected an array of 0-based device channel indices, such as [0].`
|
|
160
|
+
);
|
|
161
|
+
}
|
|
162
|
+
for (const entry of channelMap) {
|
|
163
|
+
if (typeof entry !== 'number' || !Number.isInteger(entry)) {
|
|
164
|
+
throw new TypeError('channelMap entries must be integers');
|
|
165
|
+
}
|
|
166
|
+
if (entry < 0 || entry > 65535) {
|
|
167
|
+
throw new RangeError('channelMap entries must be between 0 and 65535');
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
if (channelMap.length !== this._channels) {
|
|
171
|
+
throw new RangeError('channelMap must have exactly one entry per channel');
|
|
172
|
+
}
|
|
110
173
|
}
|
|
111
174
|
if (this._framesPerBuffer < 64 || this._framesPerBuffer > 65536) {
|
|
112
175
|
throw new TypeError(`frames per buffer must be between 64 and 65536, got ${this._framesPerBuffer}`);
|
|
@@ -190,6 +253,9 @@ class Microphone extends Emitter {
|
|
|
190
253
|
/**
|
|
191
254
|
* Most recent VAD score: the normalized RMS of the last chunk in `'energy'`
|
|
192
255
|
* mode, or 0 when VAD is disabled or before the first chunk is processed.
|
|
256
|
+
* A chunk carrying more than one channel is collapsed to the average of
|
|
257
|
+
* its channels, or to the one delivered channel a `vad: { source }` names,
|
|
258
|
+
* before the RMS, so the score reflects one channel's level.
|
|
193
259
|
* @returns {number}
|
|
194
260
|
*/
|
|
195
261
|
get vadScore() {
|
|
@@ -227,8 +293,14 @@ class Microphone extends Emitter {
|
|
|
227
293
|
await this._audioContext.resume();
|
|
228
294
|
|
|
229
295
|
// 2. Request microphone access
|
|
296
|
+
// The channel ask is 32 with ideal semantics, the count the Web Audio
|
|
297
|
+
// specification requires an implementation to support: the browser
|
|
298
|
+
// grants what it can serve and never rejects on this constraint. The
|
|
299
|
+
// granted track's own report, not this ask, is the authority for
|
|
300
|
+
// everything downstream, so a grant above the ask flows through
|
|
301
|
+
// unclamped.
|
|
230
302
|
const audioConstraints = {
|
|
231
|
-
channelCount:
|
|
303
|
+
channelCount: { ideal: 32 },
|
|
232
304
|
echoCancellation: this._echoCancellation,
|
|
233
305
|
noiseSuppression: this._noiseSuppression,
|
|
234
306
|
};
|
|
@@ -246,6 +318,49 @@ class Microphone extends Emitter {
|
|
|
246
318
|
throw error;
|
|
247
319
|
}
|
|
248
320
|
|
|
321
|
+
// The granted track's report is the capture-side authority, as the
|
|
322
|
+
// resolved device's report is on the node path: the map's entries, or
|
|
323
|
+
// the unmapped count's derivation, are checked against it here, before
|
|
324
|
+
// the worklet is built, with the node entry's message for the same
|
|
325
|
+
// condition. Without a map, only two derivations have a single meaning:
|
|
326
|
+
// 1 delivers the average of every granted channel, and the granted count
|
|
327
|
+
// delivers every granted channel in granted order; a count above the
|
|
328
|
+
// grant does not exist to deliver, and a strict subset above one does
|
|
329
|
+
// not say which channels it means, so the map has to name them. A
|
|
330
|
+
// browser that omits channelCount from getSettings() defers the check to
|
|
331
|
+
// the worklet, which sees the true channel count of every block it
|
|
332
|
+
// processes. At one delivered channel with no map there is nothing to
|
|
333
|
+
// check: the average serves any granted count.
|
|
334
|
+
if (this._channelMap !== undefined || this._channels > 1) {
|
|
335
|
+
const track = this._stream.getAudioTracks()[0];
|
|
336
|
+
const settings = track && typeof track.getSettings === 'function' ? track.getSettings() : {};
|
|
337
|
+
const granted = settings.channelCount;
|
|
338
|
+
if (typeof granted === 'number') {
|
|
339
|
+
let message = null;
|
|
340
|
+
if (this._channelMap !== undefined) {
|
|
341
|
+
for (const entry of this._channelMap) {
|
|
342
|
+
if (entry >= granted) {
|
|
343
|
+
message = `the channel map names device channel ${entry}; the device reports ${granted} input channels`;
|
|
344
|
+
break;
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
} else if (this._channels > granted) {
|
|
348
|
+
message = `the input device does not support ${this._channels} delivered channels; it reports ${granted}`;
|
|
349
|
+
} else if (this._channels < granted) {
|
|
350
|
+
message = `a channel map is required to deliver ${this._channels} of the device's ${granted} input channels`;
|
|
351
|
+
}
|
|
352
|
+
if (message !== null) {
|
|
353
|
+
this._stream.getTracks().forEach(t => t.stop());
|
|
354
|
+
this._stream = null;
|
|
355
|
+
await this._audioContext.close();
|
|
356
|
+
this._audioContext = null;
|
|
357
|
+
const error = new Error(message);
|
|
358
|
+
this.emit('error', error);
|
|
359
|
+
throw error;
|
|
360
|
+
}
|
|
361
|
+
}
|
|
362
|
+
}
|
|
363
|
+
|
|
249
364
|
// 3. Load AudioWorklet processor
|
|
250
365
|
let blobUrl = null;
|
|
251
366
|
const workletUrl = this._workletUrl ?? (blobUrl = this._createBlobUrl());
|
|
@@ -273,15 +388,31 @@ class Microphone extends Emitter {
|
|
|
273
388
|
format: this._dtype,
|
|
274
389
|
nativeSampleRate,
|
|
275
390
|
targetSampleRate: this._sampleRate,
|
|
391
|
+
channels: this._channels,
|
|
392
|
+
channelMap: this._channelMap ?? null,
|
|
276
393
|
},
|
|
277
394
|
});
|
|
278
395
|
|
|
279
396
|
// 5. Wire up data from worklet
|
|
280
397
|
this._workletNode.port.onmessage = (event) => {
|
|
281
|
-
const
|
|
398
|
+
const data = event.data;
|
|
399
|
+
|
|
400
|
+
// The port carries raw ArrayBuffer chunks and tagged control objects,
|
|
401
|
+
// as on the output worklet's port. The one control message is the
|
|
402
|
+
// worklet's channel-map failure: surface it and stop, so a map naming
|
|
403
|
+
// a channel the track does not carry is never silent.
|
|
404
|
+
if (!(data instanceof ArrayBuffer)) {
|
|
405
|
+
if (data && data.type === 'error') {
|
|
406
|
+
const error = new Error(data.message);
|
|
407
|
+
this.emit('error', error);
|
|
408
|
+
this.stop();
|
|
409
|
+
}
|
|
410
|
+
return;
|
|
411
|
+
}
|
|
412
|
+
|
|
282
413
|
const chunk = this._dtype === 'int16'
|
|
283
|
-
? new Int16Array(
|
|
284
|
-
: new Float32Array(
|
|
414
|
+
? new Int16Array(data)
|
|
415
|
+
: new Float32Array(data);
|
|
285
416
|
|
|
286
417
|
this.emit('data', chunk);
|
|
287
418
|
|
|
@@ -346,11 +477,42 @@ class Microphone extends Emitter {
|
|
|
346
477
|
}
|
|
347
478
|
|
|
348
479
|
_computeRms(chunk) {
|
|
349
|
-
let sum = 0;
|
|
350
480
|
const n = chunk.length;
|
|
351
481
|
if (n === 0) return 0;
|
|
482
|
+
const channels = this._channels;
|
|
483
|
+
const isFloat = chunk instanceof Float32Array;
|
|
484
|
+
|
|
485
|
+
if (channels > 1) {
|
|
486
|
+
// Score one channel's level, as the node path's detector feed does:
|
|
487
|
+
// collapse each interleaved frame as the configured source directs,
|
|
488
|
+
// the engine's average (single-precision accumulation, f32 quotient)
|
|
489
|
+
// by default or one named delivered channel's sample alone, then take
|
|
490
|
+
// the RMS of the collapsed signal.
|
|
491
|
+
const frames = Math.floor(n / channels);
|
|
492
|
+
if (frames === 0) return 0;
|
|
493
|
+
const source = this._vadSource;
|
|
494
|
+
let sum = 0;
|
|
495
|
+
if (source !== undefined) {
|
|
496
|
+
for (let f = 0; f < frames; f++) {
|
|
497
|
+
const s = isFloat ? chunk[f * channels + source] : chunk[f * channels + source] / 32768;
|
|
498
|
+
sum += s * s;
|
|
499
|
+
}
|
|
500
|
+
return Math.sqrt(sum / frames);
|
|
501
|
+
}
|
|
502
|
+
for (let f = 0; f < frames; f++) {
|
|
503
|
+
let acc = 0;
|
|
504
|
+
for (let c = 0; c < channels; c++) {
|
|
505
|
+
const s = isFloat ? chunk[f * channels + c] : chunk[f * channels + c] / 32768;
|
|
506
|
+
acc = Math.fround(acc + s);
|
|
507
|
+
}
|
|
508
|
+
const mono = Math.fround(acc / channels);
|
|
509
|
+
sum += mono * mono;
|
|
510
|
+
}
|
|
511
|
+
return Math.sqrt(sum / frames);
|
|
512
|
+
}
|
|
352
513
|
|
|
353
|
-
|
|
514
|
+
let sum = 0;
|
|
515
|
+
if (isFloat) {
|
|
354
516
|
for (let i = 0; i < n; i++) sum += chunk[i] * chunk[i];
|
|
355
517
|
} else {
|
|
356
518
|
for (let i = 0; i < n; i++) {
|
|
@@ -104,6 +104,12 @@ class Speaker {
|
|
|
104
104
|
if (this._sampleRate < 1000 || this._sampleRate > 384000) {
|
|
105
105
|
throw new TypeError(`sample rate must be between 1000 and 384000, got ${this._sampleRate}`);
|
|
106
106
|
}
|
|
107
|
+
// The 32 is the Web Audio specification's floor: an implementation is
|
|
108
|
+
// required to support up to 32 channels and the specification says nothing
|
|
109
|
+
// above that, so 32 is what a browser can be relied on to accept. The
|
|
110
|
+
// native surface bounds output channels below only and leaves the maximum
|
|
111
|
+
// to the device; the two differ deliberately, because they are answering to
|
|
112
|
+
// different things.
|
|
107
113
|
if (this._channels < 1 || this._channels > 32) {
|
|
108
114
|
throw new TypeError(`channels must be between 1 and 32, got ${this._channels}`);
|
|
109
115
|
}
|
package/src/browser/index.d.ts
CHANGED
|
@@ -35,6 +35,16 @@ export interface VadOptions {
|
|
|
35
35
|
* @default 300
|
|
36
36
|
*/
|
|
37
37
|
holdoffMs?: number;
|
|
38
|
+
/**
|
|
39
|
+
* The 0-based DELIVERED channel the detector reads: the position within
|
|
40
|
+
* the delivered interleaved frames, after any `channelMap` is applied (a
|
|
41
|
+
* `channelMap` names device channels; `source` names the delivered
|
|
42
|
+
* position). Must be below the delivered channel count, which is its only
|
|
43
|
+
* ceiling; no fixed maximum exists. Affects only the detector score; the
|
|
44
|
+
* delivered audio is untouched.
|
|
45
|
+
* @default the frame average of every delivered channel
|
|
46
|
+
*/
|
|
47
|
+
source?: number;
|
|
38
48
|
}
|
|
39
49
|
|
|
40
50
|
/** Constructor options for the browser `Microphone` class. */
|
|
@@ -47,14 +57,54 @@ export interface MicrophoneOptions {
|
|
|
47
57
|
sampleRate?: number;
|
|
48
58
|
|
|
49
59
|
/**
|
|
50
|
-
* Number of
|
|
60
|
+
* Number of channels the stream delivers, interleaved frame by frame in the
|
|
61
|
+
* emitted chunks. Bounded below at `1` (the default); bounded above by the
|
|
62
|
+
* granted track alone, which reports its own count when the stream starts.
|
|
63
|
+
* No fixed maximum exists.
|
|
64
|
+
*
|
|
65
|
+
* The capture itself asks the browser for every channel it will grant (the
|
|
66
|
+
* Web Audio specification's 32-channel floor, with ideal semantics), and
|
|
67
|
+
* decibri derives the delivered channels from the grant. Without a
|
|
68
|
+
* `channelMap`: `1` delivers the documented average of every granted
|
|
69
|
+
* channel; a count equal to the granted count delivers every granted
|
|
70
|
+
* channel in granted order; a count above the grant fails `start()` with
|
|
71
|
+
* an `Error` reading `the input device does not support N delivered
|
|
72
|
+
* channels; it reports M`; and a count above `1` and below the grant fails
|
|
73
|
+
* it with `a channel map is required to deliver N of the device's M input
|
|
74
|
+
* channels`, because which channels it means has no single answer, so
|
|
75
|
+
* `channelMap` names them. Where the browser does not report the granted
|
|
76
|
+
* channel count, the same `Error` is emitted on `'error'` and the capture
|
|
77
|
+
* stops as soon as the audio graph reports its true channel count.
|
|
51
78
|
* @default 1
|
|
52
|
-
* @range 1–32
|
|
53
79
|
*/
|
|
54
80
|
channels?: number;
|
|
55
81
|
|
|
82
|
+
/**
|
|
83
|
+
* Optional list of 0-based device channel indices selecting which granted
|
|
84
|
+
* channels feed the delivered channels: delivered channel `j` carries device
|
|
85
|
+
* channel `channelMap[j]`. The length must equal `channels`. Entries may
|
|
86
|
+
* repeat and may appear in any order, so a map both selects and permutes,
|
|
87
|
+
* and may name more delivered channels than the grant carries. Absent
|
|
88
|
+
* derives the delivered channels from `channels` as documented there.
|
|
89
|
+
*
|
|
90
|
+
* The same shape as the Node entry's `channelMap`, validated with the same
|
|
91
|
+
* classes and messages. Entries are checked against the granted track's own
|
|
92
|
+
* report: where the browser reports the granted channel count, `start()`
|
|
93
|
+
* rejects with an `Error` whose message names the entry and the granted
|
|
94
|
+
* count; where it does not, the same `Error` is emitted on `'error'` and
|
|
95
|
+
* the capture stops as soon as the audio graph reports its true channel
|
|
96
|
+
* count. The granted report is the only ceiling; no fixed maximum exists.
|
|
97
|
+
*
|
|
98
|
+
* A browser typically grants a single processed channel while
|
|
99
|
+
* `echoCancellation` or `noiseSuppression` is enabled, so a multichannel
|
|
100
|
+
* grant generally requires both disabled.
|
|
101
|
+
* @default undefined (the derivation `channels` documents)
|
|
102
|
+
*/
|
|
103
|
+
channelMap?: number[];
|
|
104
|
+
|
|
56
105
|
/**
|
|
57
106
|
* Frames per audio chunk. Controls chunk size and delivery interval.
|
|
107
|
+
* A chunk holds this many frames of the delivered channel count.
|
|
58
108
|
* @default 1600
|
|
59
109
|
* @range 64–65536
|
|
60
110
|
*/
|
|
@@ -78,8 +128,9 @@ export interface MicrophoneOptions {
|
|
|
78
128
|
* Voice activity detection. One of:
|
|
79
129
|
* - `false`: disabled (default)
|
|
80
130
|
* - `'energy'`: RMS energy threshold
|
|
81
|
-
* - a `VadOptions` config object
|
|
82
|
-
*
|
|
131
|
+
* - a `VadOptions` config object
|
|
132
|
+
* `{ model: 'energy', threshold?, holdoffMs?, source? }` to tune the
|
|
133
|
+
* threshold, holdoff, and detector source
|
|
83
134
|
*
|
|
84
135
|
* The browser runs energy VAD only; Silero is Node-only. The string shorthand
|
|
85
136
|
* uses a 0.01 threshold and a 300 ms holdoff; pass a `VadOptions` object to
|
|
@@ -144,6 +195,9 @@ export declare class Microphone {
|
|
|
144
195
|
/**
|
|
145
196
|
* Most recent VAD score: the normalized RMS of the last chunk in `'energy'`
|
|
146
197
|
* mode, or 0 when VAD is disabled or before the first chunk is processed.
|
|
198
|
+
* A chunk carrying more than one channel is collapsed to the average of its
|
|
199
|
+
* channels before the RMS, or read at the one delivered channel the vad
|
|
200
|
+
* `source` names when it is set, so the score reflects one channel's level.
|
|
147
201
|
*/
|
|
148
202
|
readonly vadScore: number;
|
|
149
203
|
|
|
@@ -7,8 +7,8 @@
|
|
|
7
7
|
* The readable version is in worklet-processor.js (documentation/reference only).
|
|
8
8
|
* If worklet-processor.js logic changes, this string MUST be regenerated.
|
|
9
9
|
*
|
|
10
|
-
*
|
|
10
|
+
* Logic identical to worklet-processor.js.
|
|
11
11
|
*/
|
|
12
|
-
const WORKLET_SOURCE = "var
|
|
12
|
+
const WORKLET_SOURCE = "var e=class extends AudioWorkletProcessor{constructor(e){super();let t=e.processorOptions;this.framesPerBuffer=t.framesPerBuffer,this.format=t.format,this.ratio=t.nativeSampleRate/t.targetSampleRate,this.needsResample=t.nativeSampleRate!==t.targetSampleRate,this.channelMap=t.channelMap??null,this.channels=this.channelMap?this.channelMap.length:t.channels??1,this.channelError=!1,this.position=0,this.samplesPerChunk=this.framesPerBuffer*this.channels,this.buffer=new Float32Array(this.samplesPerChunk),this.bufferIndex=0}process(e,t,n){let r=e[0];if(!r||r.length===0||!r[0]||r[0].length===0)return!0;if(this.channelError)return!1;let i=r.length,a;if(this.channelMap){for(let e=0;e<this.channelMap.length;e++)if(this.channelMap[e]>=i)return this.refuse(`the channel map names device channel `+this.channelMap[e]+`; the device reports `+i+` input channels`);a=[];for(let e=0;e<this.channelMap.length;e++)a.push(r[this.channelMap[e]])}else if(this.channels===1)if(i===1)a=[r[0]];else{let e=r[0].length,t=new Float32Array(e);for(let n=0;n<e;n++){let e=0;for(let t=0;t<i;t++)e=Math.fround(e+r[t][n]);t[n]=e/i}a=[t]}else if(this.channels===i){a=[];for(let e=0;e<i;e++)a.push(r[e])}else if(this.channels>i)return this.refuse(`the input device does not support `+this.channels+` delivered channels; it reports `+i);else return this.refuse(`a channel map is required to deliver `+this.channels+` of the device's `+i+` input channels`);this.needsResample&&(a=this.resample(a));let o=a[0].length;for(let e=0;e<o;e++){for(let t=0;t<this.channels;t++)this.buffer[this.bufferIndex++]=a[t][e];this.bufferIndex>=this.samplesPerChunk&&this.flush()}return!0}refuse(e){return this.channelError=!0,this.port.postMessage({type:`error`,message:e}),!1}resample(e){let t=e[0].length,n=0,r=this.position;for(;r<t-1;)n++,r+=this.ratio;let i=e.map(()=>new Float32Array(n));r=this.position;for(let t=0;t<n;t++){let n=Math.floor(r),a=r-n;for(let r=0;r<e.length;r++)i[r][t]=e[r][n]*(1-a)+e[r][n+1]*a;r+=this.ratio}return this.position=Math.max(0,r-t),i}flush(){let e;if(this.format===`int16`){let t=new Int16Array(this.samplesPerChunk);for(let e=0;e<this.samplesPerChunk;e++)t[e]=Math.max(-32768,Math.min(32767,Math.round(this.buffer[e]*32768)));e=t.buffer}else e=this.buffer.slice(0,this.samplesPerChunk).buffer;this.port.postMessage(e,[e]),this.buffer=new Float32Array(this.samplesPerChunk),this.bufferIndex=0}};registerProcessor(`decibri-processor`,e);";
|
|
13
13
|
|
|
14
14
|
module.exports = { WORKLET_SOURCE };
|