decibri 5.4.0 → 5.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -1
- package/MIGRATION.md +8 -12
- package/README.md +22 -11
- package/examples/decibri.browser.js +94 -11
- package/index.d.ts +105 -15
- package/index.js +52 -52
- package/models/README.md +21 -103
- package/models/THIRD-PARTY-NOTICES.md +110 -0
- package/package.json +5 -5
- package/src/browser/decibri-browser.js +177 -15
- package/src/browser/decibri-output-browser.js +6 -0
- package/src/browser/index.d.ts +58 -4
- package/src/browser/worklet-inline.js +2 -2
- package/src/browser/worklet-processor.js +116 -32
- package/src/decibri-output.js +6 -2
- package/src/decibri.d.ts +223 -28
- package/src/decibri.js +254 -42
- package/src/errors.js +12 -2
|
@@ -6,9 +6,11 @@
|
|
|
6
6
|
* If you change logic here, you MUST regenerate worklet-inline.js.
|
|
7
7
|
*
|
|
8
8
|
* Runs in a dedicated audio thread. Receives Float32 samples at the
|
|
9
|
-
* browser's native sample rate,
|
|
10
|
-
*
|
|
11
|
-
*
|
|
9
|
+
* browser's native sample rate, derives the delivered channels from the
|
|
10
|
+
* granted ones (the average of every channel, every channel in granted
|
|
11
|
+
* order, or the channels the map selects), resamples each channel to the
|
|
12
|
+
* target rate via linear interpolation, interleaves frame by frame,
|
|
13
|
+
* optionally converts to Int16, and posts chunks to the main thread.
|
|
12
14
|
*
|
|
13
15
|
* This file cannot import other modules (AudioWorklet restriction).
|
|
14
16
|
*
|
|
@@ -23,35 +25,102 @@ class DecibriProcessor extends AudioWorkletProcessor {
|
|
|
23
25
|
this.format = opts.format;
|
|
24
26
|
this.ratio = opts.nativeSampleRate / opts.targetSampleRate;
|
|
25
27
|
this.needsResample = opts.nativeSampleRate !== opts.targetSampleRate;
|
|
28
|
+
// Optional list of 0-based channel indices into the granted track's
|
|
29
|
+
// channels; delivered channel j carries granted channel channelMap[j].
|
|
30
|
+
// null derives the delivered channels from the count alone: 1 delivers
|
|
31
|
+
// the average of every granted channel, the granted count delivers every
|
|
32
|
+
// granted channel in granted order, and any other count is refused in
|
|
33
|
+
// process(). Where the browser reports the granted channel count, the
|
|
34
|
+
// main thread checked all of this before this worklet was built; the
|
|
35
|
+
// per-block guard in process() is the authority where it does not.
|
|
36
|
+
this.channelMap = opts.channelMap ?? null;
|
|
37
|
+
// The delivered channel count. A map carries one entry per delivered
|
|
38
|
+
// channel, so its length is the count when one is present.
|
|
39
|
+
this.channels = this.channelMap ? this.channelMap.length : (opts.channels ?? 1);
|
|
40
|
+
this.channelError = false;
|
|
26
41
|
this.position = 0;
|
|
27
|
-
|
|
42
|
+
// The accumulation buffer holds framesPerBuffer frames of the delivered
|
|
43
|
+
// count, interleaved frame by frame; bufferIndex counts samples. Chunks
|
|
44
|
+
// are flushed at whole frames only.
|
|
45
|
+
this.samplesPerChunk = this.framesPerBuffer * this.channels;
|
|
46
|
+
this.buffer = new Float32Array(this.samplesPerChunk);
|
|
28
47
|
this.bufferIndex = 0;
|
|
29
48
|
}
|
|
30
49
|
|
|
31
50
|
process(inputs, _outputs, _parameters) {
|
|
32
|
-
const input = inputs[0]
|
|
33
|
-
if (!input || input.length === 0) return true;
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
51
|
+
const input = inputs[0];
|
|
52
|
+
if (!input || input.length === 0 || !input[0] || input[0].length === 0) return true;
|
|
53
|
+
if (this.channelError) return false;
|
|
54
|
+
|
|
55
|
+
const granted = input.length;
|
|
56
|
+
// The delivered channels, planar: one Float32Array per delivered
|
|
57
|
+
// channel, equal lengths. Gathered here, resampled per channel in
|
|
58
|
+
// lockstep, interleaved at accumulation.
|
|
59
|
+
let planar;
|
|
60
|
+
|
|
61
|
+
if (this.channelMap) {
|
|
62
|
+
// The granted track's channel count is the only ceiling. A map entry
|
|
63
|
+
// the block cannot serve is reported once and stops the processor; it
|
|
64
|
+
// is never silently substituted.
|
|
65
|
+
for (let j = 0; j < this.channelMap.length; j++) {
|
|
66
|
+
if (this.channelMap[j] >= granted) {
|
|
67
|
+
return this.refuse('the channel map names device channel ' + this.channelMap[j] +
|
|
68
|
+
'; the device reports ' + granted + ' input channels');
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
// Delivered channel j is granted channel channelMap[j], in map order.
|
|
72
|
+
// Entries may repeat and may appear in any order, so a map both
|
|
73
|
+
// selects and permutes.
|
|
74
|
+
planar = [];
|
|
75
|
+
for (let j = 0; j < this.channelMap.length; j++) {
|
|
76
|
+
planar.push(input[this.channelMap[j]]);
|
|
77
|
+
}
|
|
78
|
+
} else if (this.channels === 1) {
|
|
79
|
+
if (granted === 1) {
|
|
80
|
+
planar = [input[0]];
|
|
81
|
+
} else {
|
|
82
|
+
// The documented average of every granted channel: each frame's
|
|
83
|
+
// arithmetic mean, accumulated at single precision (Math.fround per
|
|
84
|
+
// step) and stored as f32, matching the engine's average sample for
|
|
85
|
+
// sample.
|
|
86
|
+
const frames = input[0].length;
|
|
87
|
+
const mono = new Float32Array(frames);
|
|
88
|
+
for (let i = 0; i < frames; i++) {
|
|
89
|
+
let sum = 0;
|
|
90
|
+
for (let c = 0; c < granted; c++) {
|
|
91
|
+
sum = Math.fround(sum + input[c][i]);
|
|
92
|
+
}
|
|
93
|
+
mono[i] = sum / granted;
|
|
94
|
+
}
|
|
95
|
+
planar = [mono];
|
|
96
|
+
}
|
|
97
|
+
} else if (this.channels === granted) {
|
|
98
|
+
// Every granted channel, in granted order: the unmapped identity.
|
|
99
|
+
planar = [];
|
|
100
|
+
for (let c = 0; c < granted; c++) planar.push(input[c]);
|
|
101
|
+
} else if (this.channels > granted) {
|
|
102
|
+
return this.refuse('the input device does not support ' + this.channels +
|
|
103
|
+
' delivered channels; it reports ' + granted);
|
|
39
104
|
} else {
|
|
40
|
-
|
|
105
|
+
// An unmapped strict subset above one: which of the granted channels
|
|
106
|
+
// it means has no single answer, so the map has to name them.
|
|
107
|
+
return this.refuse('a channel map is required to deliver ' + this.channels +
|
|
108
|
+
" of the device's " + granted + ' input channels');
|
|
41
109
|
}
|
|
42
110
|
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
const remaining = this.framesPerBuffer - this.bufferIndex;
|
|
47
|
-
const available = samples.length - offset;
|
|
48
|
-
const toCopy = Math.min(remaining, available);
|
|
49
|
-
|
|
50
|
-
this.buffer.set(samples.subarray(offset, offset + toCopy), this.bufferIndex);
|
|
51
|
-
this.bufferIndex += toCopy;
|
|
52
|
-
offset += toCopy;
|
|
111
|
+
if (this.needsResample) {
|
|
112
|
+
planar = this.resample(planar);
|
|
113
|
+
}
|
|
53
114
|
|
|
54
|
-
|
|
115
|
+
// Interleave the planar channels into the accumulation buffer frame by
|
|
116
|
+
// frame, flushing at whole chunks, so every posted chunk is a whole
|
|
117
|
+
// number of frames.
|
|
118
|
+
const frames = planar[0].length;
|
|
119
|
+
for (let i = 0; i < frames; i++) {
|
|
120
|
+
for (let c = 0; c < this.channels; c++) {
|
|
121
|
+
this.buffer[this.bufferIndex++] = planar[c][i];
|
|
122
|
+
}
|
|
123
|
+
if (this.bufferIndex >= this.samplesPerChunk) {
|
|
55
124
|
this.flush();
|
|
56
125
|
}
|
|
57
126
|
}
|
|
@@ -59,10 +128,23 @@ class DecibriProcessor extends AudioWorkletProcessor {
|
|
|
59
128
|
return true;
|
|
60
129
|
}
|
|
61
130
|
|
|
62
|
-
|
|
63
|
-
|
|
131
|
+
/**
|
|
132
|
+
* Report a channel configuration no block can serve, once, with the
|
|
133
|
+
* engine's message for the same condition, and stop the processor: the
|
|
134
|
+
* failure is terminal and never silently substituted.
|
|
135
|
+
*/
|
|
136
|
+
refuse(message) {
|
|
137
|
+
this.channelError = true;
|
|
138
|
+
this.port.postMessage({ type: 'error', message });
|
|
139
|
+
return false;
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
resample(planar) {
|
|
143
|
+
const inputLength = planar[0].length;
|
|
64
144
|
|
|
65
|
-
// Calculate how many output
|
|
145
|
+
// Calculate how many output frames we can produce. One position shared
|
|
146
|
+
// by every channel: the channels advance in lockstep, so a delivered
|
|
147
|
+
// frame stays a frame.
|
|
66
148
|
let count = 0;
|
|
67
149
|
let pos = this.position;
|
|
68
150
|
while (pos < inputLength - 1) {
|
|
@@ -70,13 +152,15 @@ class DecibriProcessor extends AudioWorkletProcessor {
|
|
|
70
152
|
pos += this.ratio;
|
|
71
153
|
}
|
|
72
154
|
|
|
73
|
-
const output = new Float32Array(count);
|
|
155
|
+
const output = planar.map(() => new Float32Array(count));
|
|
74
156
|
pos = this.position;
|
|
75
157
|
|
|
76
158
|
for (let i = 0; i < count; i++) {
|
|
77
159
|
const idx = Math.floor(pos);
|
|
78
160
|
const frac = pos - idx;
|
|
79
|
-
|
|
161
|
+
for (let c = 0; c < planar.length; c++) {
|
|
162
|
+
output[c][i] = planar[c][idx] * (1 - frac) + planar[c][idx + 1] * frac;
|
|
163
|
+
}
|
|
80
164
|
pos += this.ratio;
|
|
81
165
|
}
|
|
82
166
|
|
|
@@ -90,19 +174,19 @@ class DecibriProcessor extends AudioWorkletProcessor {
|
|
|
90
174
|
let transferBuffer;
|
|
91
175
|
|
|
92
176
|
if (this.format === 'int16') {
|
|
93
|
-
const int16 = new Int16Array(this.
|
|
94
|
-
for (let i = 0; i < this.
|
|
177
|
+
const int16 = new Int16Array(this.samplesPerChunk);
|
|
178
|
+
for (let i = 0; i < this.samplesPerChunk; i++) {
|
|
95
179
|
int16[i] = Math.max(-32768, Math.min(32767, Math.round(this.buffer[i] * 32768)));
|
|
96
180
|
}
|
|
97
181
|
transferBuffer = int16.buffer;
|
|
98
182
|
} else {
|
|
99
|
-
transferBuffer = this.buffer.slice(0, this.
|
|
183
|
+
transferBuffer = this.buffer.slice(0, this.samplesPerChunk).buffer;
|
|
100
184
|
}
|
|
101
185
|
|
|
102
186
|
this.port.postMessage(transferBuffer, [transferBuffer]);
|
|
103
187
|
|
|
104
188
|
// Reset accumulation buffer
|
|
105
|
-
this.buffer = new Float32Array(this.
|
|
189
|
+
this.buffer = new Float32Array(this.samplesPerChunk);
|
|
106
190
|
this.bufferIndex = 0;
|
|
107
191
|
}
|
|
108
192
|
}
|
package/src/decibri-output.js
CHANGED
|
@@ -67,9 +67,13 @@ class Speaker extends Writable {
|
|
|
67
67
|
throw new RangeError('sample rate must be between 1000 and 384000');
|
|
68
68
|
}
|
|
69
69
|
|
|
70
|
+
// Bounded below only. How many output channels can be carried is the
|
|
71
|
+
// device's answer, so any count above zero is passed through and a device
|
|
72
|
+
// that refuses it throws a DecibriError with code
|
|
73
|
+
// 'SPEAKER_CHANNELS_UNSUPPORTED' naming the count the device reports.
|
|
70
74
|
const channels = options.channels ?? 1;
|
|
71
|
-
if (channels < 1
|
|
72
|
-
throw new RangeError('channels must be
|
|
75
|
+
if (channels < 1) {
|
|
76
|
+
throw new RangeError('channels must be at least 1');
|
|
73
77
|
}
|
|
74
78
|
|
|
75
79
|
const dtype = options.dtype ?? 'int16';
|
package/src/decibri.d.ts
CHANGED
|
@@ -8,10 +8,12 @@ export interface MicrophoneInfo {
|
|
|
8
8
|
name: string;
|
|
9
9
|
/**
|
|
10
10
|
* Stable per-host device ID. Pass via `device: { id: ... }` for selection
|
|
11
|
-
* that survives across enumerations.
|
|
12
|
-
*
|
|
13
|
-
* -
|
|
14
|
-
*
|
|
11
|
+
* that survives across enumerations. The lowercase host name, a colon, then
|
|
12
|
+
* the platform device identifier:
|
|
13
|
+
* - Windows (WASAPI): `wasapi:` then the endpoint ID (e.g.
|
|
14
|
+
* `wasapi:{0.0.1.00000000}.{...}`)
|
|
15
|
+
* - macOS (CoreAudio): `coreaudio:` then the device UID
|
|
16
|
+
* - Linux (ALSA): `alsa:` then the PCM identifier
|
|
15
17
|
* Empty string if cpal cannot produce a stable ID for this device.
|
|
16
18
|
*/
|
|
17
19
|
id: string;
|
|
@@ -42,7 +44,7 @@ export interface VersionInfo {
|
|
|
42
44
|
export interface VadOptions {
|
|
43
45
|
/**
|
|
44
46
|
* Which detector to run.
|
|
45
|
-
* - `'silero'`: Silero VAD
|
|
47
|
+
* - `'silero'`: Silero VAD v6.2 ML model (more accurate, ~1ms inference)
|
|
46
48
|
* - `'energy'`: RMS energy threshold (lightweight, no model)
|
|
47
49
|
*/
|
|
48
50
|
model: 'silero' | 'energy';
|
|
@@ -59,6 +61,17 @@ export interface VadOptions {
|
|
|
59
61
|
* @default 300
|
|
60
62
|
*/
|
|
61
63
|
holdoffMs?: number;
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* The 0-based DELIVERED channel the detector reads: the position within
|
|
67
|
+
* the delivered interleaved frames, after any `channelMap` is applied (a
|
|
68
|
+
* `channelMap` names device channels; `source` names the delivered
|
|
69
|
+
* position). Must be below the delivered channel count, which is its only
|
|
70
|
+
* ceiling; no fixed maximum exists. Affects only the detector feed; the
|
|
71
|
+
* delivered audio is untouched.
|
|
72
|
+
* @default the frame average of every delivered channel
|
|
73
|
+
*/
|
|
74
|
+
source?: number;
|
|
62
75
|
}
|
|
63
76
|
|
|
64
77
|
/**
|
|
@@ -105,12 +118,84 @@ export interface AecOptions {
|
|
|
105
118
|
* @range 1000 to 384000
|
|
106
119
|
*/
|
|
107
120
|
referenceSampleRate?: number;
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* Number of channels in the far-end reference pushed through
|
|
124
|
+
* `pushAecReference`, frame-interleaved. When it names a count above 1,
|
|
125
|
+
* decibri averages each frame to one mono sample before the canceller sees
|
|
126
|
+
* it: a multichannel reference pushed without declaring the count cancels
|
|
127
|
+
* nothing and reports no error, so the collapse is decibri's rather than
|
|
128
|
+
* the caller's.
|
|
129
|
+
*
|
|
130
|
+
* The declared count must match the buffer actually pushed. The reference
|
|
131
|
+
* arrives as flat PCM whose true channel count is not recoverable from its
|
|
132
|
+
* length, so a mismatch is not detected and raises no error: the frames
|
|
133
|
+
* are misread, nothing is cancelled, and the observable signature is
|
|
134
|
+
* `aecMetrics().delaySamples` staying `null` while the canceller reports
|
|
135
|
+
* no fault.
|
|
136
|
+
*
|
|
137
|
+
* The canceller itself reads one mono reference. Against playback through
|
|
138
|
+
* more than one loudspeaker that is a cancellation ceiling: the echo
|
|
139
|
+
* reaching the microphone is the sum of different room responses driven by
|
|
140
|
+
* different signals, and a single-reference canceller models one response
|
|
141
|
+
* applied to their average, so a placement where those paths differ leaves
|
|
142
|
+
* a residual that no amount of adaptation removes.
|
|
143
|
+
* @default 1 (mono)
|
|
144
|
+
* @range at least 1; no upper bound
|
|
145
|
+
*/
|
|
146
|
+
referenceChannels?: number;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* One delivered channel's canceller report, one entry of
|
|
151
|
+
* `AecMetrics.channels`. Engine-level fields only: the reference queue's
|
|
152
|
+
* counters (`referenceDropped`, `referenceSilence`) describe the shared queue
|
|
153
|
+
* and stay on `AecMetrics` itself.
|
|
154
|
+
*/
|
|
155
|
+
export interface AecChannelMetrics {
|
|
156
|
+
/**
|
|
157
|
+
* This channel's active delay alignment in samples, or `null` while its
|
|
158
|
+
* estimator is still searching. The offset from the reference frontier as
|
|
159
|
+
* the feeding established it, not a measurement of the room's echo path.
|
|
160
|
+
*/
|
|
161
|
+
delaySamples: number | null;
|
|
162
|
+
/**
|
|
163
|
+
* This channel's smoothed echo-return-loss-enhancement estimate in dB. Not
|
|
164
|
+
* a quality ranking across channels: ERLE rises with echo distance, because
|
|
165
|
+
* a weaker echo is easier to reduce in ratio terms, so a far microphone
|
|
166
|
+
* routinely reports a higher figure than a near one while removing less
|
|
167
|
+
* echo in absolute terms. Compare a channel against its own history, not
|
|
168
|
+
* against its neighbours.
|
|
169
|
+
*/
|
|
170
|
+
erleDb: number;
|
|
171
|
+
/**
|
|
172
|
+
* Whether this channel's double-talk detector currently believes the
|
|
173
|
+
* near-end talker is active; its adaptation is held while true.
|
|
174
|
+
*/
|
|
175
|
+
doubleTalk: boolean;
|
|
176
|
+
/**
|
|
177
|
+
* Near-end samples this channel's canceller could find no far-end sample
|
|
178
|
+
* for while an alignment was active.
|
|
179
|
+
*/
|
|
180
|
+
referenceStarved: number;
|
|
181
|
+
/**
|
|
182
|
+
* Near-end samples this channel processed while no delay alignment was
|
|
183
|
+
* active: the searching span, not a transport failure.
|
|
184
|
+
*/
|
|
185
|
+
acquisitionParked: number;
|
|
186
|
+
/**
|
|
187
|
+
* Times this channel's canceller inferred a capture discontinuity and
|
|
188
|
+
* rebuilt its alignment from the reference frontier.
|
|
189
|
+
*/
|
|
190
|
+
referenceReanchors: number;
|
|
108
191
|
}
|
|
109
192
|
|
|
110
193
|
/**
|
|
111
194
|
* The echo canceller's transport and cancellation metrics, returned by
|
|
112
195
|
* `Microphone.aecMetrics()`. One object carries the canceller's own report and
|
|
113
|
-
* the reference queue's counters.
|
|
196
|
+
* the reference queue's counters. The top-level engine fields report the first
|
|
197
|
+
* delivered channel's canceller; `channels` carries every delivered channel's
|
|
198
|
+
* report, so the two agree on a single-channel stream.
|
|
114
199
|
*/
|
|
115
200
|
export interface AecMetrics {
|
|
116
201
|
/**
|
|
@@ -149,10 +234,10 @@ export interface AecMetrics {
|
|
|
149
234
|
*/
|
|
150
235
|
referenceReanchors: number;
|
|
151
236
|
/**
|
|
152
|
-
* Far-end samples discarded
|
|
153
|
-
* queue's bound,
|
|
154
|
-
*
|
|
155
|
-
* span alone.
|
|
237
|
+
* Far-end samples discarded, at the declared reference rate: a single push
|
|
238
|
+
* exceeded the reference queue's bound, or the push arrived while capture
|
|
239
|
+
* was not running. The span an oversized push occupied is still represented
|
|
240
|
+
* as silence, so that discard costs the cancellation of the span alone.
|
|
156
241
|
*/
|
|
157
242
|
referenceDropped: number;
|
|
158
243
|
/**
|
|
@@ -161,6 +246,15 @@ export interface AecMetrics {
|
|
|
161
246
|
* while nothing is playing, the far end is silence.
|
|
162
247
|
*/
|
|
163
248
|
referenceSilence: number;
|
|
249
|
+
/**
|
|
250
|
+
* Every delivered channel's canceller report, in delivered order, one entry
|
|
251
|
+
* per channel. One canceller engine runs per delivered channel, each fed
|
|
252
|
+
* the same pushed reference and each finding its own channel's echo delay,
|
|
253
|
+
* so the entries differ where the channels' acoustic paths differ. On a
|
|
254
|
+
* single-channel stream this holds one entry agreeing with the top-level
|
|
255
|
+
* fields.
|
|
256
|
+
*/
|
|
257
|
+
channels: AecChannelMetrics[];
|
|
164
258
|
}
|
|
165
259
|
|
|
166
260
|
/** Constructor options for `Microphone`. */
|
|
@@ -173,17 +267,50 @@ export interface MicrophoneOptions extends ReadableOptions {
|
|
|
173
267
|
sampleRate?: number;
|
|
174
268
|
|
|
175
269
|
/**
|
|
176
|
-
* Number of
|
|
177
|
-
*
|
|
178
|
-
*
|
|
179
|
-
*
|
|
180
|
-
*
|
|
270
|
+
* Number of channels the stream delivers, interleaved frame by frame in the
|
|
271
|
+
* emitted chunks. Bounded below at `1` (the default); bounded above by the
|
|
272
|
+
* resolved device alone, which reports its own count when the stream
|
|
273
|
+
* starts. No fixed maximum exists.
|
|
274
|
+
*
|
|
275
|
+
* The device itself is opened at its own native channel count, exactly as
|
|
276
|
+
* it is opened at its native rate, and decibri derives the delivered
|
|
277
|
+
* channels from it. Without a `channelMap`: `1` delivers the documented
|
|
278
|
+
* average of every opened channel; a count equal to the device's own
|
|
279
|
+
* delivers every device channel in device order; a count above the
|
|
280
|
+
* device's own fails `start()` with a `DecibriError` carrying the code
|
|
281
|
+
* `'MICROPHONE_CHANNELS_UNSUPPORTED'`; and a count above `1` and below the
|
|
282
|
+
* device's own fails it with `'CHANNEL_SELECTION_AMBIGUOUS'`, because
|
|
283
|
+
* which channels it means has no single answer, so `channelMap` names
|
|
284
|
+
* them. With `aec` set, one canceller runs per delivered channel, and
|
|
285
|
+
* `aecMetrics().channels` reports each delivered channel's canceller in
|
|
286
|
+
* delivered order.
|
|
181
287
|
* @default 1
|
|
182
288
|
*/
|
|
183
289
|
channels?: number;
|
|
184
290
|
|
|
291
|
+
/**
|
|
292
|
+
* Optional list of 0-based device channel indices selecting which device
|
|
293
|
+
* channels feed the delivered channels: delivered channel `j` carries device
|
|
294
|
+
* channel `channelMap[j]`. The length must equal `channels`. Entries may
|
|
295
|
+
* repeat and may appear in any order, so a map both selects and permutes,
|
|
296
|
+
* and may name more delivered channels than the device has. Absent derives
|
|
297
|
+
* the delivered channels from `channels` as documented there.
|
|
298
|
+
*
|
|
299
|
+
* The same shape as CoreAudio AUHAL's channel map
|
|
300
|
+
* (`kAudioOutputUnitProperty_ChannelMap`: an array of device channel
|
|
301
|
+
* indices, one entry per client channel). NOT miniaudio's `channelMap`,
|
|
302
|
+
* which names a spatial layout. Entries are validated against the resolved
|
|
303
|
+
* device's own report when the stream starts: an entry the device does not
|
|
304
|
+
* have throws a `DecibriError` with code `'CHANNEL_MAP_OUT_OF_RANGE'` naming
|
|
305
|
+
* the entry and the count the device reports. The device's report is the
|
|
306
|
+
* only ceiling; no fixed maximum exists.
|
|
307
|
+
* @default undefined (the derivation `channels` documents)
|
|
308
|
+
*/
|
|
309
|
+
channelMap?: number[];
|
|
310
|
+
|
|
185
311
|
/**
|
|
186
312
|
* Frames per audio callback buffer. Controls chunk size and delivery interval.
|
|
313
|
+
* A chunk holds this many frames of the delivered channel count.
|
|
187
314
|
* At 16 kHz mono, 1600 frames = 100 ms chunks of 3200 bytes (int16).
|
|
188
315
|
* @default 1600
|
|
189
316
|
* @range 64–65536
|
|
@@ -211,10 +338,10 @@ export interface MicrophoneOptions extends ReadableOptions {
|
|
|
211
338
|
/**
|
|
212
339
|
* Voice activity detection. One of:
|
|
213
340
|
* - `false`: disabled (default)
|
|
214
|
-
* - `'silero'`: Silero VAD
|
|
341
|
+
* - `'silero'`: Silero VAD v6.2 ML model (more accurate, ~1ms inference)
|
|
215
342
|
* - `'energy'`: RMS energy threshold (lightweight)
|
|
216
|
-
* - a `VadOptions` config object `{ model, threshold?, holdoffMs? }`
|
|
217
|
-
* the threshold and
|
|
343
|
+
* - a `VadOptions` config object `{ model, threshold?, holdoffMs?, source? }`
|
|
344
|
+
* to tune the threshold, holdoff, and detector source for the chosen model
|
|
218
345
|
*
|
|
219
346
|
* The string shorthand uses the mode's default threshold (0.5 for `'silero'`,
|
|
220
347
|
* 0.01 for `'energy'`) and a 300 ms holdoff; pass a `VadOptions` object to
|
|
@@ -362,8 +489,10 @@ export declare class Microphone extends Readable {
|
|
|
362
489
|
|
|
363
490
|
/**
|
|
364
491
|
* Number of capture buffers dropped because the consumer could not keep pace.
|
|
365
|
-
* 0 while the consumer keeps up,
|
|
366
|
-
*
|
|
492
|
+
* 0 while the consumer keeps up, before capture starts, and after `stop()`,
|
|
493
|
+
* which releases the stream the counter lives on. Read it before stopping to
|
|
494
|
+
* see a session's total. A rising value means audio is being dropped to
|
|
495
|
+
* bound memory.
|
|
367
496
|
*/
|
|
368
497
|
readonly overrunCount: number;
|
|
369
498
|
|
|
@@ -371,13 +500,23 @@ export declare class Microphone extends Readable {
|
|
|
371
500
|
* Queue far-end reference audio for the echo canceller: the audio being
|
|
372
501
|
* played out, pushed as it is played, in played order. Accepts the same
|
|
373
502
|
* input shapes `Speaker.write` accepts (a `Buffer`, any TypedArray, or a
|
|
374
|
-
* `DataView` of PCM bytes in this microphone's `dtype`),
|
|
375
|
-
*
|
|
503
|
+
* `DataView` of PCM bytes in this microphone's `dtype`), at the declared
|
|
504
|
+
* `referenceSampleRate` (the capture rate when unset), interleaved at the
|
|
505
|
+
* declared `referenceChannels` (mono when unset). With `referenceChannels`
|
|
506
|
+
* above 1, each frame is averaged to one mono sample before the canceller
|
|
507
|
+
* sees it. The declared count must match this buffer's actual
|
|
508
|
+
* interleaving: a mismatch is not detected and raises no error, and shows
|
|
509
|
+
* up only as `aecMetrics().delaySamples` staying `null` with no fault
|
|
510
|
+
* reported.
|
|
376
511
|
*
|
|
377
512
|
* Never blocks and never throws on a full queue: samples that do not fit
|
|
378
513
|
* are discarded and counted by `aecMetrics().referenceDropped`. Silence
|
|
379
514
|
* between played audio need not be pushed. A push while capture is not
|
|
380
|
-
* running
|
|
515
|
+
* running is discarded and counted by `referenceDropped`, read once
|
|
516
|
+
* capture runs; a push with the `aec` option unset is a no-op. A typed
|
|
517
|
+
* array carrying a sample dtype other than the configured `dtype` throws
|
|
518
|
+
* a `TypeError`, whatever the capture state; `Buffer`, `Uint8Array`, and
|
|
519
|
+
* `DataView` are format-agnostic byte carriers.
|
|
381
520
|
*/
|
|
382
521
|
pushAecReference(data: Buffer | NodeJS.ArrayBufferView): void;
|
|
383
522
|
|
|
@@ -432,6 +571,41 @@ export interface FileOptions extends ReadableOptions {
|
|
|
432
571
|
*/
|
|
433
572
|
sampleRate?: number;
|
|
434
573
|
|
|
574
|
+
/**
|
|
575
|
+
* Number of channels the File delivers, interleaved frame by frame in the
|
|
576
|
+
* emitted chunks. Bounded below at `1` (the default); bounded above by the
|
|
577
|
+
* source's own channel count alone, read from the container's header (or
|
|
578
|
+
* `inputChannels` for `File.buffer`). No fixed maximum exists. The same
|
|
579
|
+
* meaning the option has on `Microphone`, with the source's count standing
|
|
580
|
+
* where the device's report stands.
|
|
581
|
+
*
|
|
582
|
+
* Without a `channelMap`: `1` delivers the documented average of every
|
|
583
|
+
* source channel; a count equal to the source's own delivers every source
|
|
584
|
+
* channel in source order; a count above the source's own throws a
|
|
585
|
+
* `DecibriError` carrying the code `'FILE_CHANNELS_UNSUPPORTED'`; and a
|
|
586
|
+
* count above `1` and below the source's own throws one with
|
|
587
|
+
* `'FILE_CHANNEL_SELECTION_AMBIGUOUS'`, because which channels it means
|
|
588
|
+
* has no single answer, so `channelMap` names them.
|
|
589
|
+
* @default 1
|
|
590
|
+
*/
|
|
591
|
+
channels?: number;
|
|
592
|
+
|
|
593
|
+
/**
|
|
594
|
+
* Optional list of 0-based source channel indices selecting which source
|
|
595
|
+
* channels feed the delivered channels: delivered channel `j` carries
|
|
596
|
+
* source channel `channelMap[j]`. The length must equal `channels`.
|
|
597
|
+
* Entries may repeat and may appear in any order, so a map both selects
|
|
598
|
+
* and permutes, and may name more delivered channels than the source has.
|
|
599
|
+
* Absent derives the delivered channels from `channels` as documented
|
|
600
|
+
* there. The same shape and semantics as the `Microphone` option, with
|
|
601
|
+
* source channels in the device channels' place. An entry the source does
|
|
602
|
+
* not have throws a `DecibriError` with code
|
|
603
|
+
* `'FILE_CHANNEL_MAP_OUT_OF_RANGE'` naming the entry and the source's own
|
|
604
|
+
* count. The source's count is the only ceiling; no fixed maximum exists.
|
|
605
|
+
* @default undefined (the derivation `channels` documents)
|
|
606
|
+
*/
|
|
607
|
+
channelMap?: number[];
|
|
608
|
+
|
|
435
609
|
/**
|
|
436
610
|
* Sample encoding data type of the delivered chunks.
|
|
437
611
|
* - `'int16'`: 16-bit signed integer, little-endian (2 bytes per sample)
|
|
@@ -495,7 +669,7 @@ export interface FileOptions extends ReadableOptions {
|
|
|
495
669
|
limiter?: number;
|
|
496
670
|
}
|
|
497
671
|
|
|
498
|
-
/** Options for `File.buffer`: `FileOptions` plus the samples'
|
|
672
|
+
/** Options for `File.buffer`: `FileOptions` plus the samples' own shape. */
|
|
499
673
|
export interface FileBufferOptions extends FileOptions {
|
|
500
674
|
/**
|
|
501
675
|
* The native rate of the in-memory samples in Hz. Required: raw samples
|
|
@@ -504,6 +678,18 @@ export interface FileBufferOptions extends FileOptions {
|
|
|
504
678
|
* @range 1000–384000
|
|
505
679
|
*/
|
|
506
680
|
inputRate: number;
|
|
681
|
+
|
|
682
|
+
/**
|
|
683
|
+
* The interleave of the in-memory samples: how many channels each frame
|
|
684
|
+
* carries. Raw samples carry no header to read a count from, so above
|
|
685
|
+
* mono it is stated here, the channel counterpart of `inputRate`. The
|
|
686
|
+
* samples' length must be a whole number of frames at this count. Applies
|
|
687
|
+
* to `File.buffer` alone: a file's count is read from its own header, and
|
|
688
|
+
* the option is refused on the open path.
|
|
689
|
+
* @default 1
|
|
690
|
+
* @range 1–65535
|
|
691
|
+
*/
|
|
692
|
+
inputChannels?: number;
|
|
507
693
|
}
|
|
508
694
|
|
|
509
695
|
/**
|
|
@@ -737,10 +923,17 @@ export interface AudioWriterOptions extends SaveOptions, WritableOptions {
|
|
|
737
923
|
sampleRate: number;
|
|
738
924
|
|
|
739
925
|
/**
|
|
740
|
-
* Number of channels
|
|
926
|
+
* Number of channels the incoming bytes are interleaved at, written into
|
|
927
|
+
* the file's header. The stream's total sample count must be a whole
|
|
928
|
+
* number of frames at this count. Bounded below at `1` (the default);
|
|
929
|
+
* above it, each container's own ceiling applies (a FLAC frame carries at
|
|
930
|
+
* most 8 channels; a WAV `fmt ` chunk's `nBlockAlign` is a 16-bit field,
|
|
931
|
+
* so 16-bit samples allow at most 32767), reported when the stream
|
|
932
|
+
* finishes as the container layer's own refusal. decibri enforces no
|
|
933
|
+
* ceiling of its own.
|
|
741
934
|
* @default 1
|
|
742
935
|
*/
|
|
743
|
-
channels?:
|
|
936
|
+
channels?: number;
|
|
744
937
|
|
|
745
938
|
/**
|
|
746
939
|
* Sample encoding of the incoming bytes.
|
|
@@ -810,9 +1003,11 @@ export interface SpeakerOptions extends WritableOptions {
|
|
|
810
1003
|
sampleRate?: number;
|
|
811
1004
|
|
|
812
1005
|
/**
|
|
813
|
-
* Number of output channels.
|
|
1006
|
+
* Number of output channels. The maximum is the device's: a count the device
|
|
1007
|
+
* cannot serve throws a `DecibriError` with code
|
|
1008
|
+
* `'SPEAKER_CHANNELS_UNSUPPORTED'` naming the count the device reports.
|
|
814
1009
|
* @default 1
|
|
815
|
-
* @range 1
|
|
1010
|
+
* @range 1 or more
|
|
816
1011
|
*/
|
|
817
1012
|
channels?: number;
|
|
818
1013
|
|