decibri 3.4.1 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +386 -0
- package/MIGRATION.md +151 -0
- package/README.md +206 -112
- package/examples/wav-capture.js +2 -2
- package/examples/websocket-stream.js +2 -2
- package/index.d.ts +1 -1
- package/index.js +52 -52
- package/package.json +9 -7
- package/src/browser/decibri-browser.js +37 -11
- package/src/browser/index.d.ts +28 -15
- package/src/browser/index.js +2 -2
- package/src/decibri-output.js +24 -29
- package/src/decibri.d.ts +81 -47
- package/src/decibri.js +106 -41
- package/src/errors.js +126 -23
package/README.md
CHANGED
|
@@ -1,182 +1,276 @@
|
|
|
1
|
-
<!-- markdownlint-disable MD033 MD041 MD059 -->
|
|
2
|
-
<div align="center">
|
|
3
|
-
<img width="100" height="100" alt="decibri" src="https://github.com/user-attachments/assets/a1a7aa23-6954-4cac-9490-76354f0865ee" />
|
|
4
|
-
|
|
5
1
|
# decibri
|
|
6
2
|
|
|
7
|
-
Cross-platform audio capture, playback, and voice activity detection for
|
|
8
|
-
|
|
9
|
-
<a href="https://pypi.org/project/decibri/"><img src="https://img.shields.io/pypi/v/decibri" alt="PyPI version"></a>
|
|
10
|
-
<a href="https://www.npmjs.com/package/decibri"><img src="https://img.shields.io/npm/v/decibri" alt="npm version"></a>
|
|
11
|
-
<a href="https://crates.io/crates/decibri"><img src="https://img.shields.io/crates/v/decibri" alt="crates.io version"></a>
|
|
12
|
-
<a href="https://github.com/decibri/decibri/blob/main/LICENSE"><img src="https://img.shields.io/badge/License-Apache_2.0-blue.svg" alt="Apache 2.0 License"></a>
|
|
13
|
-
<a href="https://decibri.com"><img src="https://img.shields.io/badge/Website-decibri.com-blue" alt="decibri.com"></a>
|
|
14
|
-
<a href="https://decibri.com/docs/"><img src="https://img.shields.io/badge/docs-passing-brightgreen" alt="Docs"></a>
|
|
3
|
+
Cross-platform audio capture, playback, and voice activity detection for Node.js and browsers.
|
|
15
4
|
|
|
16
|
-
|
|
5
|
+
## Installation
|
|
17
6
|
|
|
18
|
-
|
|
7
|
+
```bash
|
|
8
|
+
npm install decibri
|
|
9
|
+
```
|
|
19
10
|
|
|
20
|
-
|
|
11
|
+
One package for Node.js and browsers. Node.js gets a prebuilt native addon. Browsers get a JavaScript AudioWorklet implementation. Platform-specific binaries are installed automatically.
|
|
21
12
|
|
|
22
|
-
|
|
13
|
+
Requires Node.js >= 18. TypeScript definitions are bundled.
|
|
23
14
|
|
|
24
|
-
|
|
15
|
+
The package uses named exports:
|
|
25
16
|
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
- Cross-platform: Windows, macOS, Linux, browsers
|
|
30
|
-
- Available in Python, Node.js, and Rust
|
|
31
|
-
- Cloud and local speech-to-text integrations
|
|
17
|
+
```javascript
|
|
18
|
+
const { Microphone, Speaker, inputDevices, outputDevices, version } = require('decibri');
|
|
19
|
+
```
|
|
32
20
|
|
|
33
|
-
|
|
21
|
+
## Quick Start
|
|
34
22
|
|
|
35
|
-
|
|
23
|
+
### Capture audio
|
|
36
24
|
|
|
37
|
-
|
|
25
|
+
```javascript
|
|
26
|
+
const { Microphone } = require('decibri');
|
|
38
27
|
|
|
39
|
-
|
|
40
|
-
|
|
28
|
+
const mic = new Microphone({ sampleRate: 16000, channels: 1 });
|
|
29
|
+
mic.on('data', (chunk) => { /* Buffer of Int16 PCM samples */ });
|
|
30
|
+
setTimeout(() => mic.stop(), 5000);
|
|
41
31
|
```
|
|
42
32
|
|
|
43
|
-
|
|
33
|
+
### Play audio
|
|
44
34
|
|
|
45
|
-
```
|
|
46
|
-
|
|
35
|
+
```javascript
|
|
36
|
+
const { Speaker } = require('decibri');
|
|
37
|
+
|
|
38
|
+
const speaker = new Speaker({ sampleRate: 16000, channels: 1 });
|
|
39
|
+
speaker.write(pcmBuffer);
|
|
40
|
+
speaker.end();
|
|
47
41
|
```
|
|
48
42
|
|
|
49
|
-
|
|
43
|
+
### Browser capture
|
|
50
44
|
|
|
51
|
-
|
|
45
|
+
```javascript
|
|
46
|
+
import { Microphone } from 'decibri'; // browser entry via conditional export
|
|
52
47
|
|
|
53
|
-
|
|
54
|
-
|
|
48
|
+
const mic = new Microphone({ sampleRate: 16000 });
|
|
49
|
+
mic.on('data', (chunk) => { /* Int16Array of PCM samples */ });
|
|
50
|
+
await mic.start(); // requires user gesture in Safari
|
|
55
51
|
```
|
|
56
52
|
|
|
57
|
-
|
|
53
|
+
### Pipe capture to playback (echo)
|
|
58
54
|
|
|
59
|
-
|
|
55
|
+
```javascript
|
|
56
|
+
const { Microphone, Speaker } = require('decibri');
|
|
60
57
|
|
|
61
|
-
|
|
62
|
-
|
|
58
|
+
const mic = new Microphone({ sampleRate: 16000, channels: 1 });
|
|
59
|
+
const speaker = new Speaker({ sampleRate: 16000, channels: 1 });
|
|
60
|
+
mic.pipe(speaker);
|
|
63
61
|
```
|
|
64
62
|
|
|
65
|
-
|
|
63
|
+
## API: Microphone (Capture)
|
|
66
64
|
|
|
67
|
-
|
|
65
|
+
### `new Microphone(options?)`
|
|
68
66
|
|
|
69
|
-
|
|
67
|
+
Creates a Readable stream that captures from the microphone.
|
|
70
68
|
|
|
71
|
-
|
|
72
|
-
|
|
69
|
+
| Option | Type | Default | Description |
|
|
70
|
+
| --- | --- | --- | --- |
|
|
71
|
+
| `sampleRate` | number | 16000 | Samples per second (1000 to 384000) |
|
|
72
|
+
| `channels` | number | 1 | Input channels (1 to 32) |
|
|
73
|
+
| `framesPerBuffer` | number | 1600 | Frames per chunk (64 to 65536). At 16kHz mono, 1600 = 100ms = 3200 bytes |
|
|
74
|
+
| `device` | number, string, or `{ id: string }` | system default | Device index, case-insensitive name substring, or stable per-host ID |
|
|
75
|
+
| `dtype` | `'int16'` \| `'float32'` | `'int16'` | Sample encoding |
|
|
76
|
+
| `vad` | `false` \| `'silero'` \| `'energy'` | `false` | Voice activity detection: disabled, the Silero ML model, or an RMS energy threshold |
|
|
77
|
+
| `vadThreshold` | number | 0.5 / 0.01 | Speech threshold. Default is 0.5 for `'silero'`, 0.01 for `'energy'` |
|
|
78
|
+
| `vadHoldoff` | number | 300 | Silence holdoff in ms |
|
|
79
|
+
| `modelPath` | string | bundled model | Path to the Silero model. Only used when `vad` is `'silero'` |
|
|
73
80
|
|
|
74
|
-
|
|
75
|
-
for chunk in mic:
|
|
76
|
-
print(f"Got {len(chunk)} bytes")
|
|
77
|
-
break
|
|
78
|
-
```
|
|
81
|
+
Standard `ReadableOptions` (e.g. `highWaterMark`) are also accepted.
|
|
79
82
|
|
|
80
|
-
|
|
83
|
+
`vad: true` is not accepted; pass the mode explicitly as `vad: 'silero'` or `vad: 'energy'`.
|
|
81
84
|
|
|
82
|
-
|
|
85
|
+
### Methods
|
|
83
86
|
|
|
84
|
-
|
|
87
|
+
| Method | Description |
|
|
88
|
+
| --- | --- |
|
|
89
|
+
| `mic.stop()` | Stop capture and end stream. Safe to call multiple times |
|
|
90
|
+
| `Microphone.devices()` | List available input devices |
|
|
91
|
+
| `Microphone.version()` | Version info: `{ decibri, audioBackend, binding }` |
|
|
85
92
|
|
|
86
|
-
|
|
87
|
-
const Decibri = require('decibri');
|
|
93
|
+
The module-level `inputDevices()` and `version()` free functions are equivalent to the static methods.
|
|
88
94
|
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
95
|
+
### Properties
|
|
96
|
+
|
|
97
|
+
| Property | Type | Description |
|
|
98
|
+
| --- | --- | --- |
|
|
99
|
+
| `mic.isOpen` | boolean | `true` while capturing |
|
|
100
|
+
| `mic.vadScore` | number | Latest VAD score for the active mode (Silero probability or normalized RMS); 0 when disabled |
|
|
101
|
+
|
|
102
|
+
### Events
|
|
103
|
+
|
|
104
|
+
| Event | Payload | Description |
|
|
105
|
+
| --- | --- | --- |
|
|
106
|
+
| `'data'` | Buffer | Audio chunk (Int16 LE or Float32 LE) |
|
|
107
|
+
| `'backpressure'` | - | Internal buffer full, consumer too slow |
|
|
108
|
+
| `'speech'` | - | VAD: audio crosses threshold |
|
|
109
|
+
| `'silence'` | - | VAD: audio below threshold for `vadHoldoff` ms |
|
|
110
|
+
| `'end'` | - | Stream ended |
|
|
111
|
+
| `'error'` | Error | An error occurred |
|
|
93
112
|
|
|
94
|
-
|
|
113
|
+
## API: Speaker (Playback)
|
|
95
114
|
|
|
96
|
-
|
|
115
|
+
### `new Speaker(options?)`
|
|
97
116
|
|
|
98
|
-
|
|
117
|
+
Creates a Writable stream for speaker playback.
|
|
99
118
|
|
|
100
|
-
|
|
101
|
-
|
|
119
|
+
| Option | Type | Default | Description |
|
|
120
|
+
| --- | --- | --- | --- |
|
|
121
|
+
| `sampleRate` | number | 16000 | Playback sample rate (1000 to 384000) |
|
|
122
|
+
| `channels` | number | 1 | Output channels (1 to 32) |
|
|
123
|
+
| `dtype` | `'int16'` \| `'float32'` | `'int16'` | Sample encoding of incoming data |
|
|
124
|
+
| `device` | number, string, or `{ id: string }` | system default | Output device index, case-insensitive name substring, or stable per-host ID |
|
|
102
125
|
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
126
|
+
Standard `WritableOptions` (e.g. `highWaterMark`) are also accepted.
|
|
127
|
+
|
|
128
|
+
| Method / Property | Description |
|
|
129
|
+
| --- | --- |
|
|
130
|
+
| `speaker.write(chunk)` | Write PCM data for playback |
|
|
131
|
+
| `speaker.end()` | Signal end. Drains remaining audio, then emits `'finish'` |
|
|
132
|
+
| `speaker.stop()` | Immediate stop. Discards remaining audio |
|
|
133
|
+
| `speaker.isPlaying` | `true` while audio is being output |
|
|
134
|
+
| `Speaker.devices()` | List available output devices |
|
|
135
|
+
| `Speaker.version()` | Same as `Microphone.version()` |
|
|
136
|
+
|
|
137
|
+
The module-level `outputDevices()` free function is equivalent to `Speaker.devices()`.
|
|
138
|
+
|
|
139
|
+
## Errors
|
|
140
|
+
|
|
141
|
+
Construction errors come as typed classes you can catch:
|
|
142
|
+
|
|
143
|
+
```javascript
|
|
144
|
+
const { Microphone, DecibriError, DeviceError } = require('decibri');
|
|
145
|
+
|
|
146
|
+
try {
|
|
147
|
+
new Microphone({ device: 'no such device' });
|
|
148
|
+
} catch (err) {
|
|
149
|
+
if (err instanceof DeviceError) {
|
|
150
|
+
console.log(err.code); // e.g. 'MICROPHONE_NOT_FOUND'
|
|
151
|
+
}
|
|
110
152
|
}
|
|
111
153
|
```
|
|
112
154
|
|
|
113
|
-
|
|
155
|
+
`DeviceError`, `OrtError`, and `OrtPathError` extend `DecibriError`, which extends `Error`. Each carries a stable `code` string. Argument validation (bad `sampleRate`, `channels`, `dtype`, or `vad`) throws a built-in `RangeError` or `TypeError`.
|
|
114
156
|
|
|
115
|
-
|
|
157
|
+
## API: Browser
|
|
116
158
|
|
|
117
|
-
|
|
159
|
+
The browser API uses `getUserMedia` and `AudioWorklet`. It differs from the Node.js API because browser audio is fundamentally async.
|
|
160
|
+
|
|
161
|
+
### `new Microphone(options?)` (browser)
|
|
118
162
|
|
|
119
|
-
|
|
163
|
+
Same options as Node.js, plus:
|
|
120
164
|
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
165
|
+
| Option | Type | Default | Description |
|
|
166
|
+
| --- | --- | --- | --- |
|
|
167
|
+
| `device` | string | system default | Device ID from `Microphone.devices()` (not index) |
|
|
168
|
+
| `echoCancellation` | boolean | true | Browser echo cancellation |
|
|
169
|
+
| `noiseSuppression` | boolean | true | Browser noise suppression |
|
|
170
|
+
| `workletUrl` | string | inline blob | Custom worklet URL for strict CSP |
|
|
125
171
|
|
|
126
|
-
|
|
172
|
+
The browser runs energy-mode VAD only, so its `vad` option accepts `false` or `'energy'`. The browser `version()` returns `{ decibri }` only.
|
|
173
|
+
|
|
174
|
+
### Key differences from Node.js
|
|
175
|
+
|
|
176
|
+
| Aspect | Node.js | Browser |
|
|
177
|
+
| --- | --- | --- |
|
|
178
|
+
| Start | Automatic on first read | `await mic.start()` |
|
|
179
|
+
| Base class | Readable stream | Custom Emitter |
|
|
180
|
+
| Data type | Buffer | Int16Array / Float32Array |
|
|
181
|
+
| `devices()` | Sync, returns array | Async, returns Promise |
|
|
182
|
+
| Sample rate | Native device rate | Resampled from native rate |
|
|
183
|
+
| VAD | `'silero'` or `'energy'` | `'energy'` only |
|
|
127
184
|
|
|
128
185
|
## Voice Activity Detection
|
|
129
186
|
|
|
130
|
-
|
|
187
|
+
### Energy mode
|
|
131
188
|
|
|
132
|
-
|
|
133
|
-
- **Silero mode**: ML-based detection using the Silero VAD v5 ONNX model. More accurate, especially in noisy environments. The model (~2.3 MB) is bundled; no downloads or API keys required.
|
|
189
|
+
Lightweight RMS energy threshold. No model required.
|
|
134
190
|
|
|
135
|
-
```
|
|
136
|
-
|
|
191
|
+
```javascript
|
|
192
|
+
const mic = new Microphone({ vad: 'energy', vadThreshold: 0.01 });
|
|
193
|
+
mic.on('speech', () => console.log('speaking'));
|
|
194
|
+
mic.on('silence', () => console.log('silent'));
|
|
195
|
+
```
|
|
137
196
|
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
197
|
+
### Silero mode
|
|
198
|
+
|
|
199
|
+
ML-based detection using the Silero VAD v5 model. More accurate than energy mode, especially in noisy environments.
|
|
200
|
+
|
|
201
|
+
```javascript
|
|
202
|
+
const mic = new Microphone({ vad: 'silero', vadThreshold: 0.5 });
|
|
203
|
+
mic.on('speech', () => console.log('speaking'));
|
|
204
|
+
mic.on('silence', () => console.log('silent'));
|
|
142
205
|
```
|
|
143
206
|
|
|
144
|
-
The
|
|
207
|
+
The Silero model (~2MB) ships inside the npm package. No downloads or API keys required. Silero mode is Node.js only.
|
|
145
208
|
|
|
146
|
-
|
|
209
|
+
## Device Selection
|
|
147
210
|
|
|
148
|
-
|
|
211
|
+
```javascript
|
|
212
|
+
const { Microphone } = require('decibri');
|
|
213
|
+
|
|
214
|
+
// System default
|
|
215
|
+
const mic = new Microphone();
|
|
216
|
+
|
|
217
|
+
// By name (case-insensitive substring match)
|
|
218
|
+
const mic = new Microphone({ device: 'USB' });
|
|
219
|
+
|
|
220
|
+
// By index
|
|
221
|
+
const devices = Microphone.devices();
|
|
222
|
+
const mic = new Microphone({ device: devices[1].index });
|
|
223
|
+
|
|
224
|
+
// By stable per-host ID (survives across enumerations)
|
|
225
|
+
const mic = new Microphone({ device: { id: devices[1].id } });
|
|
226
|
+
|
|
227
|
+
Microphone.devices();
|
|
228
|
+
// [
|
|
229
|
+
// { index: 0, name: 'Microphone', id: '{0.0.1.00000000}.{...}', maxInputChannels: 2, defaultSampleRate: 48000, isDefault: true },
|
|
230
|
+
// { index: 1, name: 'USB Headset', id: '{0.0.1.00000000}.{...}', maxInputChannels: 1, defaultSampleRate: 44100, isDefault: false }
|
|
231
|
+
// ]
|
|
232
|
+
```
|
|
149
233
|
|
|
150
|
-
|
|
234
|
+
## Examples
|
|
151
235
|
|
|
152
|
-
|
|
153
|
-
| :--- | :---: | :---: | :---: | :---: | :---: |
|
|
154
|
-
| OpenAI Realtime | Cloud | ✅ | ✅ | | |
|
|
155
|
-
| Deepgram | Cloud | ✅ | ✅ | | |
|
|
156
|
-
| AssemblyAI | Cloud | ✅ | | | |
|
|
157
|
-
| Mistral Voxtral | Cloud | ✅ | | | |
|
|
158
|
-
| AWS Transcribe | Cloud | ✅ | | | |
|
|
159
|
-
| Google Speech | Cloud | ✅ | ✅ | | |
|
|
160
|
-
| Azure Speech | Cloud | ✅ | ✅ | | |
|
|
161
|
-
| Sherpa-ONNX | Local | ✅ | ✅ | ✅ | ✅ |
|
|
162
|
-
| Whisper.cpp | Local | ✅ | | | |
|
|
236
|
+
Runnable examples ship with the package. After `npm install decibri`, find them under `node_modules/decibri/examples/`. From a clone of the repo they live at `npm/decibri/examples/`.
|
|
163
237
|
|
|
164
|
-
|
|
238
|
+
```bash
|
|
239
|
+
# Capture to WAV file (no extra dependencies)
|
|
240
|
+
node node_modules/decibri/examples/wav-capture.js
|
|
241
|
+
|
|
242
|
+
# Stream to WebSocket (requires: npm install ws)
|
|
243
|
+
node node_modules/decibri/examples/websocket-server.js # terminal 1
|
|
244
|
+
node node_modules/decibri/examples/websocket-stream.js # terminal 2
|
|
245
|
+
```
|
|
165
246
|
|
|
166
|
-
##
|
|
247
|
+
## Migrating from 3.x
|
|
167
248
|
|
|
168
|
-
|
|
249
|
+
decibri 4.0.0 renames the API to a microphone and speaker vocabulary and switches to named exports. See [MIGRATION.md](./MIGRATION.md) for a complete before-and-after guide.
|
|
169
250
|
|
|
170
|
-
|
|
251
|
+
## Platform Support
|
|
171
252
|
|
|
172
|
-
|
|
253
|
+
| Platform | Architecture | Audio Backend |
|
|
254
|
+
| --- | --- | --- |
|
|
255
|
+
| Windows | x64 | WASAPI |
|
|
256
|
+
| macOS | arm64 | CoreAudio |
|
|
257
|
+
| Linux | x64 | ALSA |
|
|
258
|
+
| Linux | arm64 | ALSA |
|
|
259
|
+
| Browser | - | Web Audio API (AudioWorklet) |
|
|
173
260
|
|
|
174
|
-
##
|
|
261
|
+
## How It Works
|
|
175
262
|
|
|
176
|
-
|
|
263
|
+
decibri compiles a Rust audio core to a Node.js native addon and ships prebuilt binaries for each platform, so there is no build step on install. Browser support uses a JavaScript AudioWorklet implementation with the same event-driven API.
|
|
177
264
|
|
|
178
|
-
|
|
265
|
+
On Node.js, audio flows from the OS audio device through frame-exact buffering (which guarantees consistent chunk sizes) and into a standard Readable stream. In the browser, audio is captured and resampled in an AudioWorklet and delivered through the same `'data'` event interface.
|
|
266
|
+
|
|
267
|
+
## Documentation
|
|
268
|
+
|
|
269
|
+
- **Source code & issues**: [github.com/decibri/decibri](https://github.com/decibri/decibri)
|
|
270
|
+
- **Provider integrations** (OpenAI, Deepgram, AssemblyAI): [decibri.com/docs](https://decibri.com/docs)
|
|
179
271
|
|
|
180
272
|
## License
|
|
181
273
|
|
|
182
|
-
Apache-2.0
|
|
274
|
+
Apache-2.0. See [LICENSE](https://github.com/decibri/decibri/blob/main/LICENSE) for details.
|
|
275
|
+
|
|
276
|
+
Copyright (c) 2026 Decibri.
|
package/examples/wav-capture.js
CHANGED
|
@@ -5,14 +5,14 @@
|
|
|
5
5
|
// Usage (from repo clone): node npm/decibri/examples/wav-capture.js
|
|
6
6
|
|
|
7
7
|
const fs = require('fs');
|
|
8
|
-
const
|
|
8
|
+
const { Microphone } = require('decibri');
|
|
9
9
|
|
|
10
10
|
const SAMPLE_RATE = 16000;
|
|
11
11
|
const CHANNELS = 1;
|
|
12
12
|
const BITS_PER_SAMPLE = 16;
|
|
13
13
|
const DURATION_MS = 5000;
|
|
14
14
|
|
|
15
|
-
const mic = new
|
|
15
|
+
const mic = new Microphone({ sampleRate: SAMPLE_RATE, channels: CHANNELS });
|
|
16
16
|
const chunks = [];
|
|
17
17
|
|
|
18
18
|
mic.on('data', (chunk) => chunks.push(chunk));
|
|
@@ -5,11 +5,11 @@
|
|
|
5
5
|
// Usage (from repo clone): node npm/decibri/examples/websocket-stream.js [ws://localhost:8080]
|
|
6
6
|
|
|
7
7
|
const { WebSocket } = require('ws');
|
|
8
|
-
const
|
|
8
|
+
const { Microphone } = require('decibri');
|
|
9
9
|
|
|
10
10
|
const url = process.argv[2] || 'ws://localhost:8080';
|
|
11
11
|
const ws = new WebSocket(url);
|
|
12
|
-
const mic = new
|
|
12
|
+
const mic = new Microphone({ sampleRate: 16000, channels: 1 });
|
|
13
13
|
|
|
14
14
|
ws.on('open', () => {
|
|
15
15
|
console.log(`Connected to ${url}, streaming audio. Ctrl+C to stop.`);
|
package/index.d.ts
CHANGED