decibri 5.5.0 → 5.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -1
- package/MIGRATION.md +8 -12
- package/README.md +12 -9
- package/examples/decibri.browser.js +93 -12
- package/index.d.ts +94 -15
- package/index.js +52 -52
- package/package.json +5 -5
- package/src/browser/decibri-browser.js +174 -25
- package/src/browser/index.d.ts +58 -7
- package/src/browser/worklet-inline.js +2 -2
- package/src/browser/worklet-processor.js +116 -32
- package/src/decibri.d.ts +183 -22
- package/src/decibri.js +238 -44
- package/src/errors.js +9 -1
package/src/browser/index.d.ts
CHANGED
|
@@ -35,6 +35,16 @@ export interface VadOptions {
|
|
|
35
35
|
* @default 300
|
|
36
36
|
*/
|
|
37
37
|
holdoffMs?: number;
|
|
38
|
+
/**
|
|
39
|
+
* The 0-based DELIVERED channel the detector reads: the position within
|
|
40
|
+
* the delivered interleaved frames, after any `channelMap` is applied (a
|
|
41
|
+
* `channelMap` names device channels; `source` names the delivered
|
|
42
|
+
* position). Must be below the delivered channel count, which is its only
|
|
43
|
+
* ceiling; no fixed maximum exists. Affects only the detector score; the
|
|
44
|
+
* delivered audio is untouched.
|
|
45
|
+
* @default the frame average of every delivered channel
|
|
46
|
+
*/
|
|
47
|
+
source?: number;
|
|
38
48
|
}
|
|
39
49
|
|
|
40
50
|
/** Constructor options for the browser `Microphone` class. */
|
|
@@ -47,17 +57,54 @@ export interface MicrophoneOptions {
|
|
|
47
57
|
sampleRate?: number;
|
|
48
58
|
|
|
49
59
|
/**
|
|
50
|
-
* Number of
|
|
51
|
-
*
|
|
52
|
-
*
|
|
53
|
-
*
|
|
54
|
-
*
|
|
60
|
+
* Number of channels the stream delivers, interleaved frame by frame in the
|
|
61
|
+
* emitted chunks. Bounded below at `1` (the default); bounded above by the
|
|
62
|
+
* granted track alone, which reports its own count when the stream starts.
|
|
63
|
+
* No fixed maximum exists.
|
|
64
|
+
*
|
|
65
|
+
* The capture itself asks the browser for every channel it will grant (the
|
|
66
|
+
* Web Audio specification's 32-channel floor, with ideal semantics), and
|
|
67
|
+
* decibri derives the delivered channels from the grant. Without a
|
|
68
|
+
* `channelMap`: `1` delivers the documented average of every granted
|
|
69
|
+
* channel; a count equal to the granted count delivers every granted
|
|
70
|
+
* channel in granted order; a count above the grant fails `start()` with
|
|
71
|
+
* an `Error` reading `the input device does not support N delivered
|
|
72
|
+
* channels; it reports M`; and a count above `1` and below the grant fails
|
|
73
|
+
* it with `a channel map is required to deliver N of the device's M input
|
|
74
|
+
* channels`, because which channels it means has no single answer, so
|
|
75
|
+
* `channelMap` names them. Where the browser does not report the granted
|
|
76
|
+
* channel count, the same `Error` is emitted on `'error'` and the capture
|
|
77
|
+
* stops as soon as the audio graph reports its true channel count.
|
|
55
78
|
* @default 1
|
|
56
79
|
*/
|
|
57
80
|
channels?: number;
|
|
58
81
|
|
|
82
|
+
/**
|
|
83
|
+
* Optional list of 0-based device channel indices selecting which granted
|
|
84
|
+
* channels feed the delivered channels: delivered channel `j` carries device
|
|
85
|
+
* channel `channelMap[j]`. The length must equal `channels`. Entries may
|
|
86
|
+
* repeat and may appear in any order, so a map both selects and permutes,
|
|
87
|
+
* and may name more delivered channels than the grant carries. Absent
|
|
88
|
+
* derives the delivered channels from `channels` as documented there.
|
|
89
|
+
*
|
|
90
|
+
* The same shape as the Node entry's `channelMap`, validated with the same
|
|
91
|
+
* classes and messages. Entries are checked against the granted track's own
|
|
92
|
+
* report: where the browser reports the granted channel count, `start()`
|
|
93
|
+
* rejects with an `Error` whose message names the entry and the granted
|
|
94
|
+
* count; where it does not, the same `Error` is emitted on `'error'` and
|
|
95
|
+
* the capture stops as soon as the audio graph reports its true channel
|
|
96
|
+
* count. The granted report is the only ceiling; no fixed maximum exists.
|
|
97
|
+
*
|
|
98
|
+
* A browser typically grants a single processed channel while
|
|
99
|
+
* `echoCancellation` or `noiseSuppression` is enabled, so a multichannel
|
|
100
|
+
* grant generally requires both disabled.
|
|
101
|
+
* @default undefined (the derivation `channels` documents)
|
|
102
|
+
*/
|
|
103
|
+
channelMap?: number[];
|
|
104
|
+
|
|
59
105
|
/**
|
|
60
106
|
* Frames per audio chunk. Controls chunk size and delivery interval.
|
|
107
|
+
* A chunk holds this many frames of the delivered channel count.
|
|
61
108
|
* @default 1600
|
|
62
109
|
* @range 64–65536
|
|
63
110
|
*/
|
|
@@ -81,8 +128,9 @@ export interface MicrophoneOptions {
|
|
|
81
128
|
* Voice activity detection. One of:
|
|
82
129
|
* - `false`: disabled (default)
|
|
83
130
|
* - `'energy'`: RMS energy threshold
|
|
84
|
-
* - a `VadOptions` config object
|
|
85
|
-
*
|
|
131
|
+
* - a `VadOptions` config object
|
|
132
|
+
* `{ model: 'energy', threshold?, holdoffMs?, source? }` to tune the
|
|
133
|
+
* threshold, holdoff, and detector source
|
|
86
134
|
*
|
|
87
135
|
* The browser runs energy VAD only; Silero is Node-only. The string shorthand
|
|
88
136
|
* uses a 0.01 threshold and a 300 ms holdoff; pass a `VadOptions` object to
|
|
@@ -147,6 +195,9 @@ export declare class Microphone {
|
|
|
147
195
|
/**
|
|
148
196
|
* Most recent VAD score: the normalized RMS of the last chunk in `'energy'`
|
|
149
197
|
* mode, or 0 when VAD is disabled or before the first chunk is processed.
|
|
198
|
+
* A chunk carrying more than one channel is collapsed to the average of its
|
|
199
|
+
* channels before the RMS, or read at the one delivered channel the vad
|
|
200
|
+
* `source` names when it is set, so the score reflects one channel's level.
|
|
150
201
|
*/
|
|
151
202
|
readonly vadScore: number;
|
|
152
203
|
|
|
@@ -7,8 +7,8 @@
|
|
|
7
7
|
* The readable version is in worklet-processor.js (documentation/reference only).
|
|
8
8
|
* If worklet-processor.js logic changes, this string MUST be regenerated.
|
|
9
9
|
*
|
|
10
|
-
*
|
|
10
|
+
* Logic identical to worklet-processor.js.
|
|
11
11
|
*/
|
|
12
|
-
const WORKLET_SOURCE = "var
|
|
12
|
+
const WORKLET_SOURCE = "var e=class extends AudioWorkletProcessor{constructor(e){super();let t=e.processorOptions;this.framesPerBuffer=t.framesPerBuffer,this.format=t.format,this.ratio=t.nativeSampleRate/t.targetSampleRate,this.needsResample=t.nativeSampleRate!==t.targetSampleRate,this.channelMap=t.channelMap??null,this.channels=this.channelMap?this.channelMap.length:t.channels??1,this.channelError=!1,this.position=0,this.samplesPerChunk=this.framesPerBuffer*this.channels,this.buffer=new Float32Array(this.samplesPerChunk),this.bufferIndex=0}process(e,t,n){let r=e[0];if(!r||r.length===0||!r[0]||r[0].length===0)return!0;if(this.channelError)return!1;let i=r.length,a;if(this.channelMap){for(let e=0;e<this.channelMap.length;e++)if(this.channelMap[e]>=i)return this.refuse(`the channel map names device channel `+this.channelMap[e]+`; the device reports `+i+` input channels`);a=[];for(let e=0;e<this.channelMap.length;e++)a.push(r[this.channelMap[e]])}else if(this.channels===1)if(i===1)a=[r[0]];else{let e=r[0].length,t=new Float32Array(e);for(let n=0;n<e;n++){let e=0;for(let t=0;t<i;t++)e=Math.fround(e+r[t][n]);t[n]=e/i}a=[t]}else if(this.channels===i){a=[];for(let e=0;e<i;e++)a.push(r[e])}else if(this.channels>i)return this.refuse(`the input device does not support `+this.channels+` delivered channels; it reports `+i);else return this.refuse(`a channel map is required to deliver `+this.channels+` of the device's `+i+` input channels`);this.needsResample&&(a=this.resample(a));let o=a[0].length;for(let e=0;e<o;e++){for(let t=0;t<this.channels;t++)this.buffer[this.bufferIndex++]=a[t][e];this.bufferIndex>=this.samplesPerChunk&&this.flush()}return!0}refuse(e){return this.channelError=!0,this.port.postMessage({type:`error`,message:e}),!1}resample(e){let t=e[0].length,n=0,r=this.position;for(;r<t-1;)n++,r+=this.ratio;let i=e.map(()=>new Float32Array(n));r=this.position;for(let t=0;t<n;t++){let n=Math.floor(r),a=r-n;for(let r=0;r<e.length;r++)i[r][t]=e[r][n]*(1-a)+e[r][n+1]*a;r+=this.ratio}return this.position=Math.max(0,r-t),i}flush(){let e;if(this.format===`int16`){let t=new Int16Array(this.samplesPerChunk);for(let e=0;e<this.samplesPerChunk;e++)t[e]=Math.max(-32768,Math.min(32767,Math.round(this.buffer[e]*32768)));e=t.buffer}else e=this.buffer.slice(0,this.samplesPerChunk).buffer;this.port.postMessage(e,[e]),this.buffer=new Float32Array(this.samplesPerChunk),this.bufferIndex=0}};registerProcessor(`decibri-processor`,e);";
|
|
13
13
|
|
|
14
14
|
module.exports = { WORKLET_SOURCE };
|
|
@@ -6,9 +6,11 @@
|
|
|
6
6
|
* If you change logic here, you MUST regenerate worklet-inline.js.
|
|
7
7
|
*
|
|
8
8
|
* Runs in a dedicated audio thread. Receives Float32 samples at the
|
|
9
|
-
* browser's native sample rate,
|
|
10
|
-
*
|
|
11
|
-
*
|
|
9
|
+
* browser's native sample rate, derives the delivered channels from the
|
|
10
|
+
* granted ones (the average of every channel, every channel in granted
|
|
11
|
+
* order, or the channels the map selects), resamples each channel to the
|
|
12
|
+
* target rate via linear interpolation, interleaves frame by frame,
|
|
13
|
+
* optionally converts to Int16, and posts chunks to the main thread.
|
|
12
14
|
*
|
|
13
15
|
* This file cannot import other modules (AudioWorklet restriction).
|
|
14
16
|
*
|
|
@@ -23,35 +25,102 @@ class DecibriProcessor extends AudioWorkletProcessor {
|
|
|
23
25
|
this.format = opts.format;
|
|
24
26
|
this.ratio = opts.nativeSampleRate / opts.targetSampleRate;
|
|
25
27
|
this.needsResample = opts.nativeSampleRate !== opts.targetSampleRate;
|
|
28
|
+
// Optional list of 0-based channel indices into the granted track's
|
|
29
|
+
// channels; delivered channel j carries granted channel channelMap[j].
|
|
30
|
+
// null derives the delivered channels from the count alone: 1 delivers
|
|
31
|
+
// the average of every granted channel, the granted count delivers every
|
|
32
|
+
// granted channel in granted order, and any other count is refused in
|
|
33
|
+
// process(). Where the browser reports the granted channel count, the
|
|
34
|
+
// main thread checked all of this before this worklet was built; the
|
|
35
|
+
// per-block guard in process() is the authority where it does not.
|
|
36
|
+
this.channelMap = opts.channelMap ?? null;
|
|
37
|
+
// The delivered channel count. A map carries one entry per delivered
|
|
38
|
+
// channel, so its length is the count when one is present.
|
|
39
|
+
this.channels = this.channelMap ? this.channelMap.length : (opts.channels ?? 1);
|
|
40
|
+
this.channelError = false;
|
|
26
41
|
this.position = 0;
|
|
27
|
-
|
|
42
|
+
// The accumulation buffer holds framesPerBuffer frames of the delivered
|
|
43
|
+
// count, interleaved frame by frame; bufferIndex counts samples. Chunks
|
|
44
|
+
// are flushed at whole frames only.
|
|
45
|
+
this.samplesPerChunk = this.framesPerBuffer * this.channels;
|
|
46
|
+
this.buffer = new Float32Array(this.samplesPerChunk);
|
|
28
47
|
this.bufferIndex = 0;
|
|
29
48
|
}
|
|
30
49
|
|
|
31
50
|
process(inputs, _outputs, _parameters) {
|
|
32
|
-
const input = inputs[0]
|
|
33
|
-
if (!input || input.length === 0) return true;
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
51
|
+
const input = inputs[0];
|
|
52
|
+
if (!input || input.length === 0 || !input[0] || input[0].length === 0) return true;
|
|
53
|
+
if (this.channelError) return false;
|
|
54
|
+
|
|
55
|
+
const granted = input.length;
|
|
56
|
+
// The delivered channels, planar: one Float32Array per delivered
|
|
57
|
+
// channel, equal lengths. Gathered here, resampled per channel in
|
|
58
|
+
// lockstep, interleaved at accumulation.
|
|
59
|
+
let planar;
|
|
60
|
+
|
|
61
|
+
if (this.channelMap) {
|
|
62
|
+
// The granted track's channel count is the only ceiling. A map entry
|
|
63
|
+
// the block cannot serve is reported once and stops the processor; it
|
|
64
|
+
// is never silently substituted.
|
|
65
|
+
for (let j = 0; j < this.channelMap.length; j++) {
|
|
66
|
+
if (this.channelMap[j] >= granted) {
|
|
67
|
+
return this.refuse('the channel map names device channel ' + this.channelMap[j] +
|
|
68
|
+
'; the device reports ' + granted + ' input channels');
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
// Delivered channel j is granted channel channelMap[j], in map order.
|
|
72
|
+
// Entries may repeat and may appear in any order, so a map both
|
|
73
|
+
// selects and permutes.
|
|
74
|
+
planar = [];
|
|
75
|
+
for (let j = 0; j < this.channelMap.length; j++) {
|
|
76
|
+
planar.push(input[this.channelMap[j]]);
|
|
77
|
+
}
|
|
78
|
+
} else if (this.channels === 1) {
|
|
79
|
+
if (granted === 1) {
|
|
80
|
+
planar = [input[0]];
|
|
81
|
+
} else {
|
|
82
|
+
// The documented average of every granted channel: each frame's
|
|
83
|
+
// arithmetic mean, accumulated at single precision (Math.fround per
|
|
84
|
+
// step) and stored as f32, matching the engine's average sample for
|
|
85
|
+
// sample.
|
|
86
|
+
const frames = input[0].length;
|
|
87
|
+
const mono = new Float32Array(frames);
|
|
88
|
+
for (let i = 0; i < frames; i++) {
|
|
89
|
+
let sum = 0;
|
|
90
|
+
for (let c = 0; c < granted; c++) {
|
|
91
|
+
sum = Math.fround(sum + input[c][i]);
|
|
92
|
+
}
|
|
93
|
+
mono[i] = sum / granted;
|
|
94
|
+
}
|
|
95
|
+
planar = [mono];
|
|
96
|
+
}
|
|
97
|
+
} else if (this.channels === granted) {
|
|
98
|
+
// Every granted channel, in granted order: the unmapped identity.
|
|
99
|
+
planar = [];
|
|
100
|
+
for (let c = 0; c < granted; c++) planar.push(input[c]);
|
|
101
|
+
} else if (this.channels > granted) {
|
|
102
|
+
return this.refuse('the input device does not support ' + this.channels +
|
|
103
|
+
' delivered channels; it reports ' + granted);
|
|
39
104
|
} else {
|
|
40
|
-
|
|
105
|
+
// An unmapped strict subset above one: which of the granted channels
|
|
106
|
+
// it means has no single answer, so the map has to name them.
|
|
107
|
+
return this.refuse('a channel map is required to deliver ' + this.channels +
|
|
108
|
+
" of the device's " + granted + ' input channels');
|
|
41
109
|
}
|
|
42
110
|
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
const remaining = this.framesPerBuffer - this.bufferIndex;
|
|
47
|
-
const available = samples.length - offset;
|
|
48
|
-
const toCopy = Math.min(remaining, available);
|
|
49
|
-
|
|
50
|
-
this.buffer.set(samples.subarray(offset, offset + toCopy), this.bufferIndex);
|
|
51
|
-
this.bufferIndex += toCopy;
|
|
52
|
-
offset += toCopy;
|
|
111
|
+
if (this.needsResample) {
|
|
112
|
+
planar = this.resample(planar);
|
|
113
|
+
}
|
|
53
114
|
|
|
54
|
-
|
|
115
|
+
// Interleave the planar channels into the accumulation buffer frame by
|
|
116
|
+
// frame, flushing at whole chunks, so every posted chunk is a whole
|
|
117
|
+
// number of frames.
|
|
118
|
+
const frames = planar[0].length;
|
|
119
|
+
for (let i = 0; i < frames; i++) {
|
|
120
|
+
for (let c = 0; c < this.channels; c++) {
|
|
121
|
+
this.buffer[this.bufferIndex++] = planar[c][i];
|
|
122
|
+
}
|
|
123
|
+
if (this.bufferIndex >= this.samplesPerChunk) {
|
|
55
124
|
this.flush();
|
|
56
125
|
}
|
|
57
126
|
}
|
|
@@ -59,10 +128,23 @@ class DecibriProcessor extends AudioWorkletProcessor {
|
|
|
59
128
|
return true;
|
|
60
129
|
}
|
|
61
130
|
|
|
62
|
-
|
|
63
|
-
|
|
131
|
+
/**
|
|
132
|
+
* Report a channel configuration no block can serve, once, with the
|
|
133
|
+
* engine's message for the same condition, and stop the processor: the
|
|
134
|
+
* failure is terminal and never silently substituted.
|
|
135
|
+
*/
|
|
136
|
+
refuse(message) {
|
|
137
|
+
this.channelError = true;
|
|
138
|
+
this.port.postMessage({ type: 'error', message });
|
|
139
|
+
return false;
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
resample(planar) {
|
|
143
|
+
const inputLength = planar[0].length;
|
|
64
144
|
|
|
65
|
-
// Calculate how many output
|
|
145
|
+
// Calculate how many output frames we can produce. One position shared
|
|
146
|
+
// by every channel: the channels advance in lockstep, so a delivered
|
|
147
|
+
// frame stays a frame.
|
|
66
148
|
let count = 0;
|
|
67
149
|
let pos = this.position;
|
|
68
150
|
while (pos < inputLength - 1) {
|
|
@@ -70,13 +152,15 @@ class DecibriProcessor extends AudioWorkletProcessor {
|
|
|
70
152
|
pos += this.ratio;
|
|
71
153
|
}
|
|
72
154
|
|
|
73
|
-
const output = new Float32Array(count);
|
|
155
|
+
const output = planar.map(() => new Float32Array(count));
|
|
74
156
|
pos = this.position;
|
|
75
157
|
|
|
76
158
|
for (let i = 0; i < count; i++) {
|
|
77
159
|
const idx = Math.floor(pos);
|
|
78
160
|
const frac = pos - idx;
|
|
79
|
-
|
|
161
|
+
for (let c = 0; c < planar.length; c++) {
|
|
162
|
+
output[c][i] = planar[c][idx] * (1 - frac) + planar[c][idx + 1] * frac;
|
|
163
|
+
}
|
|
80
164
|
pos += this.ratio;
|
|
81
165
|
}
|
|
82
166
|
|
|
@@ -90,19 +174,19 @@ class DecibriProcessor extends AudioWorkletProcessor {
|
|
|
90
174
|
let transferBuffer;
|
|
91
175
|
|
|
92
176
|
if (this.format === 'int16') {
|
|
93
|
-
const int16 = new Int16Array(this.
|
|
94
|
-
for (let i = 0; i < this.
|
|
177
|
+
const int16 = new Int16Array(this.samplesPerChunk);
|
|
178
|
+
for (let i = 0; i < this.samplesPerChunk; i++) {
|
|
95
179
|
int16[i] = Math.max(-32768, Math.min(32767, Math.round(this.buffer[i] * 32768)));
|
|
96
180
|
}
|
|
97
181
|
transferBuffer = int16.buffer;
|
|
98
182
|
} else {
|
|
99
|
-
transferBuffer = this.buffer.slice(0, this.
|
|
183
|
+
transferBuffer = this.buffer.slice(0, this.samplesPerChunk).buffer;
|
|
100
184
|
}
|
|
101
185
|
|
|
102
186
|
this.port.postMessage(transferBuffer, [transferBuffer]);
|
|
103
187
|
|
|
104
188
|
// Reset accumulation buffer
|
|
105
|
-
this.buffer = new Float32Array(this.
|
|
189
|
+
this.buffer = new Float32Array(this.samplesPerChunk);
|
|
106
190
|
this.bufferIndex = 0;
|
|
107
191
|
}
|
|
108
192
|
}
|
package/src/decibri.d.ts
CHANGED
|
@@ -8,10 +8,12 @@ export interface MicrophoneInfo {
|
|
|
8
8
|
name: string;
|
|
9
9
|
/**
|
|
10
10
|
* Stable per-host device ID. Pass via `device: { id: ... }` for selection
|
|
11
|
-
* that survives across enumerations.
|
|
12
|
-
*
|
|
13
|
-
* -
|
|
14
|
-
*
|
|
11
|
+
* that survives across enumerations. The lowercase host name, a colon, then
|
|
12
|
+
* the platform device identifier:
|
|
13
|
+
* - Windows (WASAPI): `wasapi:` then the endpoint ID (e.g.
|
|
14
|
+
* `wasapi:{0.0.1.00000000}.{...}`)
|
|
15
|
+
* - macOS (CoreAudio): `coreaudio:` then the device UID
|
|
16
|
+
* - Linux (ALSA): `alsa:` then the PCM identifier
|
|
15
17
|
* Empty string if cpal cannot produce a stable ID for this device.
|
|
16
18
|
*/
|
|
17
19
|
id: string;
|
|
@@ -59,6 +61,17 @@ export interface VadOptions {
|
|
|
59
61
|
* @default 300
|
|
60
62
|
*/
|
|
61
63
|
holdoffMs?: number;
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* The 0-based DELIVERED channel the detector reads: the position within
|
|
67
|
+
* the delivered interleaved frames, after any `channelMap` is applied (a
|
|
68
|
+
* `channelMap` names device channels; `source` names the delivered
|
|
69
|
+
* position). Must be below the delivered channel count, which is its only
|
|
70
|
+
* ceiling; no fixed maximum exists. Affects only the detector feed; the
|
|
71
|
+
* delivered audio is untouched.
|
|
72
|
+
* @default the frame average of every delivered channel
|
|
73
|
+
*/
|
|
74
|
+
source?: number;
|
|
62
75
|
}
|
|
63
76
|
|
|
64
77
|
/**
|
|
@@ -133,10 +146,56 @@ export interface AecOptions {
|
|
|
133
146
|
referenceChannels?: number;
|
|
134
147
|
}
|
|
135
148
|
|
|
149
|
+
/**
|
|
150
|
+
* One delivered channel's canceller report, one entry of
|
|
151
|
+
* `AecMetrics.channels`. Engine-level fields only: the reference queue's
|
|
152
|
+
* counters (`referenceDropped`, `referenceSilence`) describe the shared queue
|
|
153
|
+
* and stay on `AecMetrics` itself.
|
|
154
|
+
*/
|
|
155
|
+
export interface AecChannelMetrics {
|
|
156
|
+
/**
|
|
157
|
+
* This channel's active delay alignment in samples, or `null` while its
|
|
158
|
+
* estimator is still searching. The offset from the reference frontier as
|
|
159
|
+
* the feeding established it, not a measurement of the room's echo path.
|
|
160
|
+
*/
|
|
161
|
+
delaySamples: number | null;
|
|
162
|
+
/**
|
|
163
|
+
* This channel's smoothed echo-return-loss-enhancement estimate in dB. Not
|
|
164
|
+
* a quality ranking across channels: ERLE rises with echo distance, because
|
|
165
|
+
* a weaker echo is easier to reduce in ratio terms, so a far microphone
|
|
166
|
+
* routinely reports a higher figure than a near one while removing less
|
|
167
|
+
* echo in absolute terms. Compare a channel against its own history, not
|
|
168
|
+
* against its neighbours.
|
|
169
|
+
*/
|
|
170
|
+
erleDb: number;
|
|
171
|
+
/**
|
|
172
|
+
* Whether this channel's double-talk detector currently believes the
|
|
173
|
+
* near-end talker is active; its adaptation is held while true.
|
|
174
|
+
*/
|
|
175
|
+
doubleTalk: boolean;
|
|
176
|
+
/**
|
|
177
|
+
* Near-end samples this channel's canceller could find no far-end sample
|
|
178
|
+
* for while an alignment was active.
|
|
179
|
+
*/
|
|
180
|
+
referenceStarved: number;
|
|
181
|
+
/**
|
|
182
|
+
* Near-end samples this channel processed while no delay alignment was
|
|
183
|
+
* active: the searching span, not a transport failure.
|
|
184
|
+
*/
|
|
185
|
+
acquisitionParked: number;
|
|
186
|
+
/**
|
|
187
|
+
* Times this channel's canceller inferred a capture discontinuity and
|
|
188
|
+
* rebuilt its alignment from the reference frontier.
|
|
189
|
+
*/
|
|
190
|
+
referenceReanchors: number;
|
|
191
|
+
}
|
|
192
|
+
|
|
136
193
|
/**
|
|
137
194
|
* The echo canceller's transport and cancellation metrics, returned by
|
|
138
195
|
* `Microphone.aecMetrics()`. One object carries the canceller's own report and
|
|
139
|
-
* the reference queue's counters.
|
|
196
|
+
* the reference queue's counters. The top-level engine fields report the first
|
|
197
|
+
* delivered channel's canceller; `channels` carries every delivered channel's
|
|
198
|
+
* report, so the two agree on a single-channel stream.
|
|
140
199
|
*/
|
|
141
200
|
export interface AecMetrics {
|
|
142
201
|
/**
|
|
@@ -175,10 +234,10 @@ export interface AecMetrics {
|
|
|
175
234
|
*/
|
|
176
235
|
referenceReanchors: number;
|
|
177
236
|
/**
|
|
178
|
-
* Far-end samples discarded
|
|
179
|
-
* queue's bound,
|
|
180
|
-
*
|
|
181
|
-
* span alone.
|
|
237
|
+
* Far-end samples discarded, at the declared reference rate: a single push
|
|
238
|
+
* exceeded the reference queue's bound, or the push arrived while capture
|
|
239
|
+
* was not running. The span an oversized push occupied is still represented
|
|
240
|
+
* as silence, so that discard costs the cancellation of the span alone.
|
|
182
241
|
*/
|
|
183
242
|
referenceDropped: number;
|
|
184
243
|
/**
|
|
@@ -187,6 +246,15 @@ export interface AecMetrics {
|
|
|
187
246
|
* while nothing is playing, the far end is silence.
|
|
188
247
|
*/
|
|
189
248
|
referenceSilence: number;
|
|
249
|
+
/**
|
|
250
|
+
* Every delivered channel's canceller report, in delivered order, one entry
|
|
251
|
+
* per channel. One canceller engine runs per delivered channel, each fed
|
|
252
|
+
* the same pushed reference and each finding its own channel's echo delay,
|
|
253
|
+
* so the entries differ where the channels' acoustic paths differ. On a
|
|
254
|
+
* single-channel stream this holds one entry agreeing with the top-level
|
|
255
|
+
* fields.
|
|
256
|
+
*/
|
|
257
|
+
channels: AecChannelMetrics[];
|
|
190
258
|
}
|
|
191
259
|
|
|
192
260
|
/** Constructor options for `Microphone`. */
|
|
@@ -199,17 +267,50 @@ export interface MicrophoneOptions extends ReadableOptions {
|
|
|
199
267
|
sampleRate?: number;
|
|
200
268
|
|
|
201
269
|
/**
|
|
202
|
-
* Number of
|
|
203
|
-
*
|
|
204
|
-
*
|
|
205
|
-
*
|
|
206
|
-
*
|
|
270
|
+
* Number of channels the stream delivers, interleaved frame by frame in the
|
|
271
|
+
* emitted chunks. Bounded below at `1` (the default); bounded above by the
|
|
272
|
+
* resolved device alone, which reports its own count when the stream
|
|
273
|
+
* starts. No fixed maximum exists.
|
|
274
|
+
*
|
|
275
|
+
* The device itself is opened at its own native channel count, exactly as
|
|
276
|
+
* it is opened at its native rate, and decibri derives the delivered
|
|
277
|
+
* channels from it. Without a `channelMap`: `1` delivers the documented
|
|
278
|
+
* average of every opened channel; a count equal to the device's own
|
|
279
|
+
* delivers every device channel in device order; a count above the
|
|
280
|
+
* device's own fails `start()` with a `DecibriError` carrying the code
|
|
281
|
+
* `'MICROPHONE_CHANNELS_UNSUPPORTED'`; and a count above `1` and below the
|
|
282
|
+
* device's own fails it with `'CHANNEL_SELECTION_AMBIGUOUS'`, because
|
|
283
|
+
* which channels it means has no single answer, so `channelMap` names
|
|
284
|
+
* them. With `aec` set, one canceller runs per delivered channel, and
|
|
285
|
+
* `aecMetrics().channels` reports each delivered channel's canceller in
|
|
286
|
+
* delivered order.
|
|
207
287
|
* @default 1
|
|
208
288
|
*/
|
|
209
289
|
channels?: number;
|
|
210
290
|
|
|
291
|
+
/**
|
|
292
|
+
* Optional list of 0-based device channel indices selecting which device
|
|
293
|
+
* channels feed the delivered channels: delivered channel `j` carries device
|
|
294
|
+
* channel `channelMap[j]`. The length must equal `channels`. Entries may
|
|
295
|
+
* repeat and may appear in any order, so a map both selects and permutes,
|
|
296
|
+
* and may name more delivered channels than the device has. Absent derives
|
|
297
|
+
* the delivered channels from `channels` as documented there.
|
|
298
|
+
*
|
|
299
|
+
* The same shape as CoreAudio AUHAL's channel map
|
|
300
|
+
* (`kAudioOutputUnitProperty_ChannelMap`: an array of device channel
|
|
301
|
+
* indices, one entry per client channel). NOT miniaudio's `channelMap`,
|
|
302
|
+
* which names a spatial layout. Entries are validated against the resolved
|
|
303
|
+
* device's own report when the stream starts: an entry the device does not
|
|
304
|
+
* have throws a `DecibriError` with code `'CHANNEL_MAP_OUT_OF_RANGE'` naming
|
|
305
|
+
* the entry and the count the device reports. The device's report is the
|
|
306
|
+
* only ceiling; no fixed maximum exists.
|
|
307
|
+
* @default undefined (the derivation `channels` documents)
|
|
308
|
+
*/
|
|
309
|
+
channelMap?: number[];
|
|
310
|
+
|
|
211
311
|
/**
|
|
212
312
|
* Frames per audio callback buffer. Controls chunk size and delivery interval.
|
|
313
|
+
* A chunk holds this many frames of the delivered channel count.
|
|
213
314
|
* At 16 kHz mono, 1600 frames = 100 ms chunks of 3200 bytes (int16).
|
|
214
315
|
* @default 1600
|
|
215
316
|
* @range 64–65536
|
|
@@ -239,8 +340,8 @@ export interface MicrophoneOptions extends ReadableOptions {
|
|
|
239
340
|
* - `false`: disabled (default)
|
|
240
341
|
* - `'silero'`: Silero VAD v6.2 ML model (more accurate, ~1ms inference)
|
|
241
342
|
* - `'energy'`: RMS energy threshold (lightweight)
|
|
242
|
-
* - a `VadOptions` config object `{ model, threshold?, holdoffMs? }`
|
|
243
|
-
* the threshold and
|
|
343
|
+
* - a `VadOptions` config object `{ model, threshold?, holdoffMs?, source? }`
|
|
344
|
+
* to tune the threshold, holdoff, and detector source for the chosen model
|
|
244
345
|
*
|
|
245
346
|
* The string shorthand uses the mode's default threshold (0.5 for `'silero'`,
|
|
246
347
|
* 0.01 for `'energy'`) and a 300 ms holdoff; pass a `VadOptions` object to
|
|
@@ -388,8 +489,10 @@ export declare class Microphone extends Readable {
|
|
|
388
489
|
|
|
389
490
|
/**
|
|
390
491
|
* Number of capture buffers dropped because the consumer could not keep pace.
|
|
391
|
-
* 0 while the consumer keeps up,
|
|
392
|
-
*
|
|
492
|
+
* 0 while the consumer keeps up, before capture starts, and after `stop()`,
|
|
493
|
+
* which releases the stream the counter lives on. Read it before stopping to
|
|
494
|
+
* see a session's total. A rising value means audio is being dropped to
|
|
495
|
+
* bound memory.
|
|
393
496
|
*/
|
|
394
497
|
readonly overrunCount: number;
|
|
395
498
|
|
|
@@ -409,7 +512,11 @@ export declare class Microphone extends Readable {
|
|
|
409
512
|
* Never blocks and never throws on a full queue: samples that do not fit
|
|
410
513
|
* are discarded and counted by `aecMetrics().referenceDropped`. Silence
|
|
411
514
|
* between played audio need not be pushed. A push while capture is not
|
|
412
|
-
* running
|
|
515
|
+
* running is discarded and counted by `referenceDropped`, read once
|
|
516
|
+
* capture runs; a push with the `aec` option unset is a no-op. A typed
|
|
517
|
+
* array carrying a sample dtype other than the configured `dtype` throws
|
|
518
|
+
* a `TypeError`, whatever the capture state; `Buffer`, `Uint8Array`, and
|
|
519
|
+
* `DataView` are format-agnostic byte carriers.
|
|
413
520
|
*/
|
|
414
521
|
pushAecReference(data: Buffer | NodeJS.ArrayBufferView): void;
|
|
415
522
|
|
|
@@ -464,6 +571,41 @@ export interface FileOptions extends ReadableOptions {
|
|
|
464
571
|
*/
|
|
465
572
|
sampleRate?: number;
|
|
466
573
|
|
|
574
|
+
/**
|
|
575
|
+
* Number of channels the File delivers, interleaved frame by frame in the
|
|
576
|
+
* emitted chunks. Bounded below at `1` (the default); bounded above by the
|
|
577
|
+
* source's own channel count alone, read from the container's header (or
|
|
578
|
+
* `inputChannels` for `File.buffer`). No fixed maximum exists. The same
|
|
579
|
+
* meaning the option has on `Microphone`, with the source's count standing
|
|
580
|
+
* where the device's report stands.
|
|
581
|
+
*
|
|
582
|
+
* Without a `channelMap`: `1` delivers the documented average of every
|
|
583
|
+
* source channel; a count equal to the source's own delivers every source
|
|
584
|
+
* channel in source order; a count above the source's own throws a
|
|
585
|
+
* `DecibriError` carrying the code `'FILE_CHANNELS_UNSUPPORTED'`; and a
|
|
586
|
+
* count above `1` and below the source's own throws one with
|
|
587
|
+
* `'FILE_CHANNEL_SELECTION_AMBIGUOUS'`, because which channels it means
|
|
588
|
+
* has no single answer, so `channelMap` names them.
|
|
589
|
+
* @default 1
|
|
590
|
+
*/
|
|
591
|
+
channels?: number;
|
|
592
|
+
|
|
593
|
+
/**
|
|
594
|
+
* Optional list of 0-based source channel indices selecting which source
|
|
595
|
+
* channels feed the delivered channels: delivered channel `j` carries
|
|
596
|
+
* source channel `channelMap[j]`. The length must equal `channels`.
|
|
597
|
+
* Entries may repeat and may appear in any order, so a map both selects
|
|
598
|
+
* and permutes, and may name more delivered channels than the source has.
|
|
599
|
+
* Absent derives the delivered channels from `channels` as documented
|
|
600
|
+
* there. The same shape and semantics as the `Microphone` option, with
|
|
601
|
+
* source channels in the device channels' place. An entry the source does
|
|
602
|
+
* not have throws a `DecibriError` with code
|
|
603
|
+
* `'FILE_CHANNEL_MAP_OUT_OF_RANGE'` naming the entry and the source's own
|
|
604
|
+
* count. The source's count is the only ceiling; no fixed maximum exists.
|
|
605
|
+
* @default undefined (the derivation `channels` documents)
|
|
606
|
+
*/
|
|
607
|
+
channelMap?: number[];
|
|
608
|
+
|
|
467
609
|
/**
|
|
468
610
|
* Sample encoding data type of the delivered chunks.
|
|
469
611
|
* - `'int16'`: 16-bit signed integer, little-endian (2 bytes per sample)
|
|
@@ -527,7 +669,7 @@ export interface FileOptions extends ReadableOptions {
|
|
|
527
669
|
limiter?: number;
|
|
528
670
|
}
|
|
529
671
|
|
|
530
|
-
/** Options for `File.buffer`: `FileOptions` plus the samples'
|
|
672
|
+
/** Options for `File.buffer`: `FileOptions` plus the samples' own shape. */
|
|
531
673
|
export interface FileBufferOptions extends FileOptions {
|
|
532
674
|
/**
|
|
533
675
|
* The native rate of the in-memory samples in Hz. Required: raw samples
|
|
@@ -536,6 +678,18 @@ export interface FileBufferOptions extends FileOptions {
|
|
|
536
678
|
* @range 1000–384000
|
|
537
679
|
*/
|
|
538
680
|
inputRate: number;
|
|
681
|
+
|
|
682
|
+
/**
|
|
683
|
+
* The interleave of the in-memory samples: how many channels each frame
|
|
684
|
+
* carries. Raw samples carry no header to read a count from, so above
|
|
685
|
+
* mono it is stated here, the channel counterpart of `inputRate`. The
|
|
686
|
+
* samples' length must be a whole number of frames at this count. Applies
|
|
687
|
+
* to `File.buffer` alone: a file's count is read from its own header, and
|
|
688
|
+
* the option is refused on the open path.
|
|
689
|
+
* @default 1
|
|
690
|
+
* @range 1–65535
|
|
691
|
+
*/
|
|
692
|
+
inputChannels?: number;
|
|
539
693
|
}
|
|
540
694
|
|
|
541
695
|
/**
|
|
@@ -769,10 +923,17 @@ export interface AudioWriterOptions extends SaveOptions, WritableOptions {
|
|
|
769
923
|
sampleRate: number;
|
|
770
924
|
|
|
771
925
|
/**
|
|
772
|
-
* Number of channels
|
|
926
|
+
* Number of channels the incoming bytes are interleaved at, written into
|
|
927
|
+
* the file's header. The stream's total sample count must be a whole
|
|
928
|
+
* number of frames at this count. Bounded below at `1` (the default);
|
|
929
|
+
* above it, each container's own ceiling applies (a FLAC frame carries at
|
|
930
|
+
* most 8 channels; a WAV `fmt ` chunk's `nBlockAlign` is a 16-bit field,
|
|
931
|
+
* so 16-bit samples allow at most 32767), reported when the stream
|
|
932
|
+
* finishes as the container layer's own refusal. decibri enforces no
|
|
933
|
+
* ceiling of its own.
|
|
773
934
|
* @default 1
|
|
774
935
|
*/
|
|
775
|
-
channels?:
|
|
936
|
+
channels?: number;
|
|
776
937
|
|
|
777
938
|
/**
|
|
778
939
|
* Sample encoding of the incoming bytes.
|