decibri 5.2.5 → 5.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/decibri.js CHANGED
@@ -1,6 +1,6 @@
1
1
  'use strict';
2
2
 
3
- const { Readable } = require('stream');
3
+ const { Readable, Writable } = require('stream');
4
4
  const path = require('path');
5
5
  const fs = require('fs');
6
6
  const { DecibriBridge, FileHandle } = require('../index.js');
@@ -342,6 +342,68 @@ class Microphone extends Readable {
342
342
  throw new RangeError('limiter ceiling must be between -3.0 and 0.0');
343
343
  }
344
344
 
345
+ // ── Validate AEC ─────────────────────────────────────────────────────────
346
+
347
+ // Echo cancellation: the 'tau' shorthand names the model, or an
348
+ // { model, tailMs, suppression, referenceSampleRate } object tunes it;
349
+ // absence leaves it off. The model name is deliberately NOT checked
350
+ // against a list here: the canceller owns the accepted set, so the native
351
+ // layer parses it (AecModel::from_str) and an unknown name is rejected by
352
+ // the native constructor with the canceller's own message (a DecibriError
353
+ // with code 'AEC_CONFIG_INVALID'). The three tuning fields are checked
354
+ // here with the same RangeError / TypeError classes the other
355
+ // conditioning options use; the native layer backstops the same checks.
356
+ // The capture-rate window (8000..=48000 with AEC on) is guarded by the
357
+ // core, surfacing as a RangeError from the native constructor.
358
+ const aec = options.aec;
359
+ let aecModel;
360
+ let aecTailMs;
361
+ let aecSuppression;
362
+ let aecReferenceSampleRate;
363
+ if (aec !== undefined) {
364
+ if (typeof aec === 'string') {
365
+ aecModel = aec;
366
+ } else if (aec !== null && typeof aec === 'object' && !Array.isArray(aec)) {
367
+ const { model, tailMs, suppression, referenceSampleRate } = aec;
368
+ if (typeof model !== 'string') {
369
+ throw new TypeError(
370
+ `Invalid aec model: ${JSON.stringify(model)}. Expected a model name string such as 'tau'.`
371
+ );
372
+ }
373
+ aecModel = model;
374
+ if (tailMs !== undefined) {
375
+ if (typeof tailMs !== 'number' || Number.isNaN(tailMs)) {
376
+ throw new TypeError('aec tailMs must be a number');
377
+ }
378
+ if (tailMs < 16 || tailMs > 500) {
379
+ throw new RangeError('aec tailMs must be between 16 and 500');
380
+ }
381
+ aecTailMs = tailMs;
382
+ }
383
+ if (suppression !== undefined) {
384
+ if (suppression !== 'conservative' && suppression !== 'off') {
385
+ throw new TypeError(
386
+ `aec suppression must be 'conservative' or 'off'; got ${JSON.stringify(suppression)}`
387
+ );
388
+ }
389
+ aecSuppression = suppression;
390
+ }
391
+ if (referenceSampleRate !== undefined) {
392
+ if (typeof referenceSampleRate !== 'number' || Number.isNaN(referenceSampleRate)) {
393
+ throw new TypeError('aec referenceSampleRate must be a number');
394
+ }
395
+ if (referenceSampleRate < 1000 || referenceSampleRate > 384000) {
396
+ throw new RangeError('aec referenceSampleRate must be between 1000 and 384000');
397
+ }
398
+ aecReferenceSampleRate = referenceSampleRate;
399
+ }
400
+ } else {
401
+ throw new TypeError(
402
+ `Invalid aec value: ${JSON.stringify(aec)}. Expected a model name such as 'tau', or a config object { model, tailMs, suppression, referenceSampleRate }.`
403
+ );
404
+ }
405
+ }
406
+
345
407
  // Internal plumbing: inject the bundled ORT dylib path into the napi
346
408
  // constructor whenever an ONNX stage loads (Silero VAD or denoise). If
347
409
  // resolution fails (unknown platform, platform package not installed), this
@@ -377,6 +439,10 @@ class Microphone extends Readable {
377
439
  highpass,
378
440
  agc,
379
441
  limiter,
442
+ aec: aecModel,
443
+ aecTailMs,
444
+ aecSuppression,
445
+ aecReferenceSampleRate,
380
446
  },
381
447
  };
382
448
  }
@@ -535,6 +601,67 @@ class Microphone extends Readable {
535
601
  return this._native.overrunCount;
536
602
  }
537
603
 
604
+ /**
605
+ * Queue far-end reference audio for the echo canceller: the audio being
606
+ * played out, pushed as it is played, in played order. Accepts the same
607
+ * input shapes `Speaker.write` accepts (a `Buffer`, any TypedArray, or a
608
+ * `DataView` of PCM bytes in this microphone's `dtype`), mono, at the
609
+ * declared `referenceSampleRate` (the capture rate when unset).
610
+ *
611
+ * Never blocks and never throws on a full queue: samples that do not fit
612
+ * are discarded and counted by `aecMetrics().referenceDropped`, and the
613
+ * span they occupied is represented as silence. Silence between played
614
+ * audio need not be pushed; a caller that stops pushing has said nothing is
615
+ * playing. A push while capture is not running, or with the `aec` option
616
+ * unset, is a no-op.
617
+ *
618
+ * @param {Buffer | NodeJS.ArrayBufferView} data PCM samples in the
619
+ * configured `dtype`.
620
+ */
621
+ pushAecReference(data) {
622
+ let buf;
623
+ if (Buffer.isBuffer(data)) {
624
+ buf = data;
625
+ } else if (ArrayBuffer.isView(data)) {
626
+ // Any TypedArray or DataView: view the same bytes, no copy, exactly as
627
+ // the stream machinery normalizes a typed-array write to a Speaker.
628
+ buf = Buffer.from(data.buffer, data.byteOffset, data.byteLength);
629
+ } else {
630
+ throw new TypeError(
631
+ 'pushAecReference requires a Buffer, TypedArray, or DataView of PCM samples in the configured dtype'
632
+ );
633
+ }
634
+ this._native.pushAecReference(buf);
635
+ }
636
+
637
+ /**
638
+ * The echo canceller's transport and cancellation metrics, merged with the
639
+ * reference queue's counters, or `null` when the `aec` option is unset or
640
+ * capture is not running.
641
+ *
642
+ * `delaySamples` staying `null` while `acquisitionParked` climbs is the
643
+ * signature of a canceller with no usable reference: none pushed, not at
644
+ * the declared rate, or not the signal that produced the echo. A climbing
645
+ * `referenceDropped` means single pushes are exceeding the reference
646
+ * queue's bound.
647
+ *
648
+ * @returns {import('./decibri').AecMetrics | null}
649
+ */
650
+ aecMetrics() {
651
+ const m = this._native.aecMetrics();
652
+ if (m === null || m === undefined) return null;
653
+ return {
654
+ delaySamples: m.delaySamples ?? null,
655
+ erleDb: m.erleDb,
656
+ doubleTalk: m.doubleTalk,
657
+ referenceStarved: m.referenceStarved,
658
+ acquisitionParked: m.acquisitionParked,
659
+ referenceReanchors: m.referenceReanchors,
660
+ referenceDropped: m.referenceDropped,
661
+ referenceSilence: m.referenceSilence,
662
+ };
663
+ }
664
+
538
665
  /**
539
666
  * List all available input devices on the system.
540
667
  * @returns {Array<{index: number, name: string, id: string, maxInputChannels: number, defaultSampleRate: number, isDefault: boolean}>}
@@ -587,19 +714,20 @@ function version() {
587
714
 
588
715
  class File extends Readable {
589
716
  /**
590
- * Open a WAV file as an offline source, synchronously. Everything a
717
+ * Open an audio file as an offline source, synchronously. Everything a
591
718
  * `Microphone` does to live audio, a `File` does to audio you already
592
719
  * have: the same conditioning options, the same stream of conditioned
593
720
  * chunks, and (with `vad` set) the same per-chunk speech events, plus the
594
721
  * whole-file `analyze()` a live stream cannot offer.
595
722
  *
596
- * The bare constructor reads the WAV inline, blocking the event loop on
723
+ * The bare constructor reads the file inline, blocking the event loop on
597
724
  * disk I/O; prefer `await File.open(path, options)` in servers and other
598
725
  * latency-sensitive code, exactly as `Microphone.open` is preferred over
599
726
  * `new Microphone`. Iteration and analysis are separate single passes:
600
727
  * each consumes the source once, so use one `File` per operation.
601
728
  *
602
- * @param {string} filePath Path to a WAV file (16-bit PCM or 32-bit float).
729
+ * @param {string} filePath Path to a WAV, AIFF, AIFF-C or FLAC file. The
730
+ * container is identified from the bytes, not from the extension.
603
731
  * @param {import('./decibri').FileOptions} [options]
604
732
  * @param {{ prepared: object, native: object }} [_internal] Internal: a
605
733
  * pre-resolved options bundle and an already-constructed native handle,
@@ -805,8 +933,8 @@ class File extends Readable {
805
933
  }
806
934
 
807
935
  /**
808
- * Open a WAV file without blocking the event loop: the disk read, WAV
809
- * parse, and chain construction run on the native thread pool. The
936
+ * Open an audio file without blocking the event loop: the disk read,
937
+ * decode, and chain construction run on the native thread pool. The
810
938
  * recommended form in Node, mirroring `Microphone.open`. The synchronous
811
939
  * `new File(path)` remains available for scripts.
812
940
  *
@@ -1005,10 +1133,10 @@ class File extends Readable {
1005
1133
  }
1006
1134
 
1007
1135
  /**
1008
- * The source's own rate, taken from the WAV header or from the `inputRate`
1009
- * passed to `File.buffer`. Differs from `sampleRate` when the source was
1010
- * resampled. Readable for the life of the File, including after the source
1011
- * is consumed or closed.
1136
+ * The source's own rate, taken from the file's header or from the
1137
+ * `inputRate` passed to `File.buffer`. Differs from `sampleRate` when the
1138
+ * source was resampled. Readable for the life of the File, including after
1139
+ * the source is consumed or closed.
1012
1140
  * @returns {number}
1013
1141
  */
1014
1142
  get inputRate() {
@@ -1063,6 +1191,77 @@ class File extends Readable {
1063
1191
  return this.analyze();
1064
1192
  }
1065
1193
 
1194
+ /**
1195
+ * Validate the save options and resolve them into the native options
1196
+ * object. Shared by `File.save` and `AudioWriter`, so the two spellings of
1197
+ * a write accept and reject identically.
1198
+ * @internal
1199
+ * @param {import('./decibri').SaveOptions} options
1200
+ */
1201
+ static _prepareSaveOptions(options) {
1202
+ const format = options.format;
1203
+ if (format !== undefined && format !== 'wav' && format !== 'aiff' && format !== 'flac') {
1204
+ throw new TypeError(
1205
+ `Invalid format value: ${JSON.stringify(format)}. Expected 'wav', 'aiff', or 'flac'.`
1206
+ );
1207
+ }
1208
+ const compression = options.compression;
1209
+ if (compression !== undefined) {
1210
+ if (typeof compression !== 'number' || Number.isNaN(compression)) {
1211
+ throw new TypeError('compression must be a number');
1212
+ }
1213
+ if (compression < 0 || compression > 8) {
1214
+ throw new RangeError('flac compression level must be between 0 and 8');
1215
+ }
1216
+ }
1217
+ return { format, compression };
1218
+ }
1219
+
1220
+ /**
1221
+ * Write the conditioned recording to disk, off the event loop. Runs the
1222
+ * recording once through the same conditioning pass iteration delivers,
1223
+ * whole, and writes it as 16-bit PCM mono at `sampleRate`. Consumes the
1224
+ * source: a save is a single pass, separate from iteration and analysis.
1225
+ *
1226
+ * The container comes from the path's extension (`.wav`, `.aiff`, `.aif`,
1227
+ * `.aifc` or `.flac`), or from `options.format`: decibri reads a file by
1228
+ * its content and writes one by its name. An extension it does not
1229
+ * recognise rejects rather than defaulting. `options.compression` sets the
1230
+ * FLAC compression level (0-8, default 5); it applies only to FLAC.
1231
+ *
1232
+ * Resolves to a `SaveReport`: `clippedSamples` counts finite samples
1233
+ * outside full scale clamped to `[-1.0, 1.0]` (AGC or AEC without a
1234
+ * limiter can overshoot, and 16-bit PCM cannot hold it), and
1235
+ * `nonFiniteSamples` counts NaN samples written as silence and infinite
1236
+ * samples written as full scale.
1237
+ *
1238
+ * Requires a `File` that is not already being streamed: once the stream
1239
+ * has been engaged this rejects with a `DecibriError` carrying the code
1240
+ * `'FILE_ENGAGED'`. Every failure detected before the pass begins leaves
1241
+ * the `File` usable; a failure during the pass consumes the source.
1242
+ *
1243
+ * @param {string} filePath
1244
+ * @param {import('./decibri').SaveOptions} [options]
1245
+ * @returns {Promise<import('./decibri').SaveReport>}
1246
+ */
1247
+ async save(filePath, options = {}) {
1248
+ // Checked before the native handle is touched: the rejection belongs to
1249
+ // this call, and the source is never taken from a File that keeps
1250
+ // streaming.
1251
+ if (this._engaged) {
1252
+ throw fileEngagedError();
1253
+ }
1254
+ if (typeof filePath !== 'string') {
1255
+ throw new TypeError('path must be a string');
1256
+ }
1257
+ const nativeOptions = File._prepareSaveOptions(options);
1258
+ try {
1259
+ return await this._native.save(filePath, nativeOptions);
1260
+ } catch (err) {
1261
+ throw wrapNativeError(err);
1262
+ }
1263
+ }
1264
+
1066
1265
  /**
1067
1266
  * Release the source. Idempotent; a closed File reads as ended.
1068
1267
  */
@@ -1071,10 +1270,131 @@ class File extends Readable {
1071
1270
  }
1072
1271
  }
1073
1272
 
1273
+ // ─── AudioWriter (Writable): file sink ───────────────────────────────────────
1274
+
1275
+ class AudioWriter extends Writable {
1276
+ /**
1277
+ * A file sink for PCM audio: the Writable to pair with decibri's Readable
1278
+ * sources, and with any other stream of PCM bytes (a TTS engine, a decoded
1279
+ * network stream). Collects the whole stream, then writes it as one audio
1280
+ * file when the stream finishes, exactly as `File.save` writes: the same
1281
+ * containers from the same extension rule, the same 16-bit PCM encoding,
1282
+ * the same clamp and non-finite handling, the same bytes.
1283
+ *
1284
+ * Chunks are raw PCM bytes in `dtype` ('int16' little-endian by default,
1285
+ * matching what a `File` or `Microphone` emits; 'float32' for raw f32
1286
+ * bytes). Audio is written mono: `channels` may only be 1. `sampleRate` is
1287
+ * required, because raw audio carries no header to read one from.
1288
+ *
1289
+ * The file is written when the stream finishes ('finish' fires after the
1290
+ * file is on disk), and `report` then carries the `SaveReport` the write
1291
+ * produced. A failure destroys the stream with the error.
1292
+ *
1293
+ * @param {string} filePath Path whose extension names the container,
1294
+ * unless `format` overrides it.
1295
+ * @param {import('./decibri').AudioWriterOptions} options
1296
+ */
1297
+ constructor(filePath, options = {}) {
1298
+ super({ highWaterMark: options.highWaterMark });
1299
+ if (typeof filePath !== 'string') {
1300
+ throw new TypeError('path must be a string');
1301
+ }
1302
+ const sampleRate = options.sampleRate;
1303
+ if (typeof sampleRate !== 'number' || Number.isNaN(sampleRate)) {
1304
+ throw new TypeError('sampleRate is required for AudioWriter (raw audio carries no header)');
1305
+ }
1306
+ if (sampleRate < 1000 || sampleRate > 384000) {
1307
+ throw new RangeError('sample rate must be between 1000 and 384000');
1308
+ }
1309
+ const channels = options.channels;
1310
+ if (channels !== undefined && channels !== 1) {
1311
+ throw new RangeError('multichannel write is not supported; channels must be 1 (mono)');
1312
+ }
1313
+ const dtype = options.dtype ?? 'int16';
1314
+ if (dtype !== 'int16' && dtype !== 'float32') {
1315
+ throw new TypeError("dtype must be 'int16' or 'float32'");
1316
+ }
1317
+ // Validated now, so a bad option is a construction-time throw rather
1318
+ // than a deferred 'error' event after the audio has been streamed.
1319
+ this._saveOptions = File._prepareSaveOptions(options);
1320
+ this._filePath = filePath;
1321
+ this._sampleRate = sampleRate;
1322
+ this._dtype = dtype;
1323
+ this._chunks = [];
1324
+ this._report = null;
1325
+ }
1326
+
1327
+ /**
1328
+ * The `SaveReport` of the completed write: `clippedSamples` and
1329
+ * `nonFiniteSamples`, exactly as `File.save` resolves. `null` until
1330
+ * 'finish' has fired.
1331
+ * @returns {import('./decibri').SaveReport | null}
1332
+ */
1333
+ get report() {
1334
+ return this._report;
1335
+ }
1336
+
1337
+ /** @internal */
1338
+ _write(chunk, encoding, callback) {
1339
+ this._chunks.push(chunk);
1340
+ callback();
1341
+ }
1342
+
1343
+ /** @internal */
1344
+ _final(callback) {
1345
+ const bytes = Buffer.concat(this._chunks);
1346
+ this._chunks = [];
1347
+ const bytesPerSample = this._dtype === 'int16' ? 2 : 4;
1348
+ if (bytes.length % bytesPerSample !== 0) {
1349
+ callback(
1350
+ new RangeError(
1351
+ `audio bytes do not divide into whole ${this._dtype} samples; ` +
1352
+ `${bytes.length % bytesPerSample} byte(s) over`
1353
+ )
1354
+ );
1355
+ return;
1356
+ }
1357
+ const samples = new Float32Array(bytes.length / bytesPerSample);
1358
+ if (this._dtype === 'int16') {
1359
+ // The inverse of the int16 delivery encoding: value / 32768, so bytes
1360
+ // that came from a decibri source re-quantise to the identical file.
1361
+ for (let i = 0; i < samples.length; i++) {
1362
+ samples[i] = bytes.readInt16LE(i * 2) / 32768;
1363
+ }
1364
+ } else {
1365
+ for (let i = 0; i < samples.length; i++) {
1366
+ samples[i] = bytes.readFloatLE(i * 4);
1367
+ }
1368
+ }
1369
+ // The write is File.save on a source at the writer's own rate: the same
1370
+ // encode path, so the two spellings produce the same bytes.
1371
+ let file;
1372
+ try {
1373
+ file = File.buffer(samples, {
1374
+ inputRate: this._sampleRate,
1375
+ sampleRate: this._sampleRate,
1376
+ });
1377
+ } catch (err) {
1378
+ callback(err);
1379
+ return;
1380
+ }
1381
+ file
1382
+ .save(this._filePath, this._saveOptions)
1383
+ .then((report) => {
1384
+ this._report = report;
1385
+ callback();
1386
+ })
1387
+ .catch((err) => {
1388
+ callback(err);
1389
+ });
1390
+ }
1391
+ }
1392
+
1074
1393
  module.exports = {
1075
1394
  Microphone,
1076
1395
  Speaker,
1077
1396
  File,
1397
+ AudioWriter,
1078
1398
  inputDevices,
1079
1399
  outputDevices,
1080
1400
  version,
package/src/errors.js CHANGED
@@ -80,8 +80,12 @@ const RANGE_PREFIXES = [
80
80
  'frames per buffer must be between',
81
81
  'agc target level must be between',
82
82
  'limiter ceiling must be between',
83
+ 'flac compression level must be between',
84
+ 'aec tailMs must be between',
85
+ 'aec referenceSampleRate must be between',
83
86
  'Silero VAD only supports',
84
87
  'VAD threshold must be between',
88
+ 'echo cancellation only supports',
85
89
  'device index out of range',
86
90
  'analysis requires VAD',
87
91
  ];
@@ -89,6 +93,8 @@ const RANGE_PREFIXES = [
89
93
  const TYPE_PREFIXES = [
90
94
  "dtype must be 'int16' or 'float32'",
91
95
  "format must be 'int16' or 'float32'",
96
+ 'aec suppression must be',
97
+ 'Invalid format value:',
92
98
  ];
93
99
 
94
100
  const DEVICE_CODES = [
@@ -126,8 +132,12 @@ const BASE_CODES = [
126
132
  ['the requested sample rate conversion is not supported', 'RESAMPLE_CONFIG_INVALID'],
127
133
  ['the resample chain was fed after it was flushed', 'RESAMPLE_AFTER_FLUSH'],
128
134
  ['resampler error:', 'RESAMPLE_FAILED'],
135
+ ['echo canceller configuration error:', 'AEC_CONFIG_INVALID'],
129
136
  ['Failed to read audio file', 'FILE_READ_FAILED'],
130
- ['invalid WAV file:', 'WAV_INVALID'],
137
+ ['Failed to write audio file', 'FILE_WRITE_FAILED'],
138
+ ['unsupported audio format:', 'AUDIO_FORMAT_UNSUPPORTED'],
139
+ ['malformed audio file:', 'AUDIO_FILE_MALFORMED'],
140
+ ['truncated audio file:', 'AUDIO_FILE_TRUNCATED'],
131
141
  ['ONNX Runtime was initialized in pid', 'FORK_AFTER_ORT_INIT'],
132
142
  ['ONNX backend error from', 'ONNX_BACKEND_FAILED'],
133
143
  // Authored in the napi layer (bindings/node/src/lib.rs), not in