decibri 5.3.0 → 5.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -67,9 +67,13 @@ class Speaker extends Writable {
67
67
  throw new RangeError('sample rate must be between 1000 and 384000');
68
68
  }
69
69
 
70
+ // Bounded below only. How many output channels can be carried is the
71
+ // device's answer, so any count above zero is passed through and a device
72
+ // that refuses it throws a DecibriError with code
73
+ // 'SPEAKER_CHANNELS_UNSUPPORTED' naming the count the device reports.
70
74
  const channels = options.channels ?? 1;
71
- if (channels < 1 || channels > 32) {
72
- throw new RangeError('channels must be between 1 and 32');
75
+ if (channels < 1) {
76
+ throw new RangeError('channels must be at least 1');
73
77
  }
74
78
 
75
79
  const dtype = options.dtype ?? 'int16';
package/src/decibri.d.ts CHANGED
@@ -42,7 +42,7 @@ export interface VersionInfo {
42
42
  export interface VadOptions {
43
43
  /**
44
44
  * Which detector to run.
45
- * - `'silero'`: Silero VAD v5 ML model (more accurate, ~1ms inference)
45
+ * - `'silero'`: Silero VAD v6.2 ML model (more accurate, ~1ms inference)
46
46
  * - `'energy'`: RMS energy threshold (lightweight, no model)
47
47
  */
48
48
  model: 'silero' | 'energy';
@@ -105,6 +105,32 @@ export interface AecOptions {
105
105
  * @range 1000 to 384000
106
106
  */
107
107
  referenceSampleRate?: number;
108
+
109
+ /**
110
+ * Number of channels in the far-end reference pushed through
111
+ * `pushAecReference`, frame-interleaved. When it names a count above 1,
112
+ * decibri averages each frame to one mono sample before the canceller sees
113
+ * it: a multichannel reference pushed without declaring the count cancels
114
+ * nothing and reports no error, so the collapse is decibri's rather than
115
+ * the caller's.
116
+ *
117
+ * The declared count must match the buffer actually pushed. The reference
118
+ * arrives as flat PCM whose true channel count is not recoverable from its
119
+ * length, so a mismatch is not detected and raises no error: the frames
120
+ * are misread, nothing is cancelled, and the observable signature is
121
+ * `aecMetrics().delaySamples` staying `null` while the canceller reports
122
+ * no fault.
123
+ *
124
+ * The canceller itself reads one mono reference. Against playback through
125
+ * more than one loudspeaker that is a cancellation ceiling: the echo
126
+ * reaching the microphone is the sum of different room responses driven by
127
+ * different signals, and a single-reference canceller models one response
128
+ * applied to their average, so a placement where those paths differ leaves
129
+ * a residual that no amount of adaptation removes.
130
+ * @default 1 (mono)
131
+ * @range at least 1; no upper bound
132
+ */
133
+ referenceChannels?: number;
108
134
  }
109
135
 
110
136
  /**
@@ -211,7 +237,7 @@ export interface MicrophoneOptions extends ReadableOptions {
211
237
  /**
212
238
  * Voice activity detection. One of:
213
239
  * - `false`: disabled (default)
214
- * - `'silero'`: Silero VAD v5 ML model (more accurate, ~1ms inference)
240
+ * - `'silero'`: Silero VAD v6.2 ML model (more accurate, ~1ms inference)
215
241
  * - `'energy'`: RMS energy threshold (lightweight)
216
242
  * - a `VadOptions` config object `{ model, threshold?, holdoffMs? }` to tune
217
243
  * the threshold and holdoff for the chosen model
@@ -371,8 +397,14 @@ export declare class Microphone extends Readable {
371
397
  * Queue far-end reference audio for the echo canceller: the audio being
372
398
  * played out, pushed as it is played, in played order. Accepts the same
373
399
  * input shapes `Speaker.write` accepts (a `Buffer`, any TypedArray, or a
374
- * `DataView` of PCM bytes in this microphone's `dtype`), mono, at the
375
- * declared `referenceSampleRate` (the capture rate when unset).
400
+ * `DataView` of PCM bytes in this microphone's `dtype`), at the declared
401
+ * `referenceSampleRate` (the capture rate when unset), interleaved at the
402
+ * declared `referenceChannels` (mono when unset). With `referenceChannels`
403
+ * above 1, each frame is averaged to one mono sample before the canceller
404
+ * sees it. The declared count must match this buffer's actual
405
+ * interleaving: a mismatch is not detected and raises no error, and shows
406
+ * up only as `aecMetrics().delaySamples` staying `null` with no fault
407
+ * reported.
376
408
  *
377
409
  * Never blocks and never throws on a full queue: samples that do not fit
378
410
  * are discarded and counted by `aecMetrics().referenceDropped`. Silence
@@ -423,7 +455,7 @@ export declare class Microphone extends Readable {
423
455
  export interface FileOptions extends ReadableOptions {
424
456
  /**
425
457
  * Target output rate in Hz: the rate every delivered chunk carries. The
426
- * source's input rate (from the WAV header, or `inputRate` for
458
+ * source's input rate (from the file's header, or `inputRate` for
427
459
  * `File.buffer`) is resampled to this rate, so a 44.1 kHz recording comes
428
460
  * out at 16 kHz unless you set `sampleRate`. The same meaning the option
429
461
  * has on `Microphone`.
@@ -551,6 +583,46 @@ export interface VadReport {
551
583
  segments: Segment[];
552
584
  }
553
585
 
586
+ /** Options for `File.save` (also accepted by `AudioWriter`). */
587
+ export interface SaveOptions {
588
+ /**
589
+ * The container format to write. When not given it comes from the path's
590
+ * extension: `.wav`, `.aiff`, `.aif`, `.aifc` or `.flac`. decibri reads a
591
+ * file by its content and writes one by its name; an extension it does not
592
+ * recognise is an error, never a silent default.
593
+ */
594
+ format?: 'wav' | 'aiff' | 'flac';
595
+
596
+ /**
597
+ * FLAC compression level. Higher levels search harder for a smaller file;
598
+ * every level decodes to identical audio. Applies only to FLAC; ignored
599
+ * for WAV and AIFF.
600
+ * @default 5
601
+ * @range 0 to 8
602
+ */
603
+ compression?: number;
604
+ }
605
+
606
+ /**
607
+ * What a save did to the samples on their way into the file, resolved by
608
+ * `File.save` and carried by `AudioWriter.report`.
609
+ */
610
+ export interface SaveReport {
611
+ /**
612
+ * Finite samples outside full scale, clamped to [-1.0, 1.0]. Conditioned
613
+ * audio can exceed full scale (AGC or AEC without a limiter), and 16-bit
614
+ * PCM cannot hold that, so the overshoot clips and this count says how
615
+ * much. The count is a statement about integer encodings: a float encoding
616
+ * would preserve the overshoot instead, and would report zero.
617
+ */
618
+ clippedSamples: number;
619
+ /**
620
+ * Non-finite samples replaced before writing: NaN with silence, an
621
+ * infinity with full scale. The same replacement on every format.
622
+ */
623
+ nonFiniteSamples: number;
624
+ }
625
+
554
626
  /**
555
627
  * Offline audio source: conditions a recording or in-memory samples through
556
628
  * the same chain as the live `Microphone`, delivered as a finite Readable
@@ -559,7 +631,7 @@ export interface VadReport {
559
631
  * analyze the whole recording for speech with `analyze()` / `analyse()`,
560
632
  * which a live stream cannot do.
561
633
  *
562
- * Construction: `new File(path)` reads the WAV synchronously (fine for a
634
+ * Construction: `new File(path)` reads the file synchronously (fine for a
563
635
  * script; it blocks the event loop on disk I/O), `await File.open(path)` reads
564
636
  * it off the event loop (the recommended form, mirroring `Microphone.open`),
565
637
  * and `File.buffer(samples, { inputRate })` wraps a `Float32Array` of samples
@@ -586,15 +658,16 @@ export interface VadReport {
586
658
  */
587
659
  export declare class File extends Readable {
588
660
  /**
589
- * Open a WAV file synchronously (blocks on disk I/O; prefer `File.open`
590
- * in servers). Supports 16-bit PCM and 32-bit float WAV files; the input
591
- * rate and channel count come from the header.
661
+ * Open an audio file synchronously (blocks on disk I/O; prefer `File.open`
662
+ * in servers). Reads WAV, AIFF, AIFF-C and FLAC, identified from the
663
+ * file's own bytes rather than its extension; the input rate and channel
664
+ * count come from the header.
592
665
  */
593
666
  constructor(path: string, options?: FileOptions);
594
667
 
595
668
  /**
596
- * Open a WAV file without blocking the event loop: the disk read, WAV
597
- * parse, and chain construction run on the native thread pool. The
669
+ * Open an audio file without blocking the event loop: the disk read,
670
+ * decode, and chain construction run on the native thread pool. The
598
671
  * recommended form, mirroring `Microphone.open`.
599
672
  */
600
673
  static open(path: string, options?: FileOptions): Promise<File>;
@@ -622,7 +695,7 @@ export declare class File extends Readable {
622
695
  readonly sampleRate: number;
623
696
 
624
697
  /**
625
- * The source's own rate, taken from the WAV header or from the
698
+ * The source's own rate, taken from the file's header or from the
626
699
  * `inputRate` passed to `File.buffer`. Differs from `sampleRate` when the
627
700
  * source was resampled.
628
701
  */
@@ -646,6 +719,26 @@ export declare class File extends Readable {
646
719
  /** The same whole-recording analysis under the international spelling. */
647
720
  analyse(): Promise<VadReport>;
648
721
 
722
+ /**
723
+ * Write the conditioned recording to disk, off the event loop. Runs the
724
+ * recording once through the same conditioning pass iteration delivers,
725
+ * whole, and writes it as 16-bit PCM mono at `sampleRate`. The container
726
+ * comes from the path's extension (`.wav`, `.aiff`, `.aif`, `.aifc` or
727
+ * `.flac`), or from `options.format`: decibri reads a file by its content
728
+ * and writes one by its name. Consumes the source (a `File` is a single
729
+ * pass).
730
+ *
731
+ * Resolves to a `SaveReport`: how many samples were clamped to full scale
732
+ * and how many non-finite samples were replaced (NaN as silence, an
733
+ * infinity as full scale).
734
+ *
735
+ * Requires a File that is not already being streamed: once the stream has
736
+ * been engaged this rejects with a `DecibriError` carrying the code
737
+ * `'FILE_ENGAGED'`. Every failure detected before the pass begins leaves
738
+ * the File usable; a failure during the pass consumes the source.
739
+ */
740
+ save(path: string, options?: SaveOptions): Promise<SaveReport>;
741
+
649
742
  /** Release the source. Idempotent; a closed File reads as ended. */
650
743
  close(): void;
651
744
 
@@ -666,6 +759,61 @@ export declare class File extends Readable {
666
759
  once(event: string | symbol, listener: (...args: any[]) => void): this;
667
760
  }
668
761
 
762
+ /** Constructor options for `AudioWriter`: the save options plus the stream's own description. */
763
+ export interface AudioWriterOptions extends SaveOptions, WritableOptions {
764
+ /**
765
+ * The rate of the incoming samples in Hz, written into the file's header.
766
+ * Required: raw audio carries no header to read a rate from.
767
+ * @range 1000 to 384000
768
+ */
769
+ sampleRate: number;
770
+
771
+ /**
772
+ * Number of channels. Audio is written mono; only `1` is accepted.
773
+ * @default 1
774
+ */
775
+ channels?: 1;
776
+
777
+ /**
778
+ * Sample encoding of the incoming bytes.
779
+ * - `'int16'`: 16-bit signed integer, little-endian, what a `File` or
780
+ * `Microphone` emits by default
781
+ * - `'float32'`: 32-bit IEEE 754 float, little-endian
782
+ * @default 'int16'
783
+ */
784
+ dtype?: 'int16' | 'float32';
785
+ }
786
+
787
+ /**
788
+ * A file sink for PCM audio: the Writable to pair with decibri's Readable
789
+ * sources, and with any other stream of PCM bytes (a TTS engine, a decoded
790
+ * network stream). Collects the whole stream, then writes it as one audio
791
+ * file when the stream finishes, exactly as `File.save` writes: the same
792
+ * containers from the same extension rule, the same 16-bit PCM encoding, the
793
+ * same clamp and non-finite handling, the same bytes.
794
+ *
795
+ * `'finish'` fires after the file is on disk, and `report` then carries the
796
+ * `SaveReport` the write produced. A failure destroys the stream with the
797
+ * error.
798
+ *
799
+ * @example
800
+ * const { pipeline } = require('node:stream/promises');
801
+ * const { File, AudioWriter } = require('decibri');
802
+ * await pipeline(
803
+ * new File('noisy.wav', { denoise: 'fastenhancer-t' }),
804
+ * new AudioWriter('clean.flac', { sampleRate: 16000 }),
805
+ * );
806
+ */
807
+ export declare class AudioWriter extends Writable {
808
+ constructor(path: string, options: AudioWriterOptions);
809
+
810
+ /**
811
+ * The `SaveReport` of the completed write, exactly as `File.save` resolves
812
+ * it. `null` until `'finish'` has fired.
813
+ */
814
+ readonly report: SaveReport | null;
815
+ }
816
+
669
817
  export interface SpeakerInfo {
670
818
  /** Device index (pass to constructor as `device`). */
671
819
  index: number;
@@ -694,9 +842,11 @@ export interface SpeakerOptions extends WritableOptions {
694
842
  sampleRate?: number;
695
843
 
696
844
  /**
697
- * Number of output channels.
845
+ * Number of output channels. The maximum is the device's: a count the device
846
+ * cannot serve throws a `DecibriError` with code
847
+ * `'SPEAKER_CHANNELS_UNSUPPORTED'` naming the count the device reports.
698
848
  * @default 1
699
- * @range 1–32
849
+ * @range 1 or more
700
850
  */
701
851
  channels?: number;
702
852
 
package/src/decibri.js CHANGED
@@ -1,6 +1,6 @@
1
1
  'use strict';
2
2
 
3
- const { Readable } = require('stream');
3
+ const { Readable, Writable } = require('stream');
4
4
  const path = require('path');
5
5
  const fs = require('fs');
6
6
  const { DecibriBridge, FileHandle } = require('../index.js');
@@ -151,7 +151,7 @@ class Microphone extends Readable {
151
151
  // additive. The `channels` option is kept for that forward compatibility.
152
152
  const channels = options.channels ?? 1;
153
153
  if (channels < 1) {
154
- throw new RangeError('channels must be between 1 and 32');
154
+ throw new RangeError('channels must be at least 1');
155
155
  }
156
156
  if (channels > 1) {
157
157
  throw new RangeError('multichannel capture is not supported; channels must be 1 (mono)');
@@ -345,8 +345,8 @@ class Microphone extends Readable {
345
345
  // ── Validate AEC ─────────────────────────────────────────────────────────
346
346
 
347
347
  // Echo cancellation: the 'tau' shorthand names the model, or an
348
- // { model, tailMs, suppression, referenceSampleRate } object tunes it;
349
- // absence leaves it off. The model name is deliberately NOT checked
348
+ // { model, tailMs, suppression, referenceSampleRate, referenceChannels }
349
+ // object tunes it; absence leaves it off. The model name is deliberately NOT checked
350
350
  // against a list here: the canceller owns the accepted set, so the native
351
351
  // layer parses it (AecModel::from_str) and an unknown name is rejected by
352
352
  // the native constructor with the canceller's own message (a DecibriError
@@ -360,11 +360,12 @@ class Microphone extends Readable {
360
360
  let aecTailMs;
361
361
  let aecSuppression;
362
362
  let aecReferenceSampleRate;
363
+ let aecReferenceChannels;
363
364
  if (aec !== undefined) {
364
365
  if (typeof aec === 'string') {
365
366
  aecModel = aec;
366
367
  } else if (aec !== null && typeof aec === 'object' && !Array.isArray(aec)) {
367
- const { model, tailMs, suppression, referenceSampleRate } = aec;
368
+ const { model, tailMs, suppression, referenceSampleRate, referenceChannels } = aec;
368
369
  if (typeof model !== 'string') {
369
370
  throw new TypeError(
370
371
  `Invalid aec model: ${JSON.stringify(model)}. Expected a model name string such as 'tau'.`
@@ -397,9 +398,18 @@ class Microphone extends Readable {
397
398
  }
398
399
  aecReferenceSampleRate = referenceSampleRate;
399
400
  }
401
+ if (referenceChannels !== undefined) {
402
+ if (typeof referenceChannels !== 'number' || Number.isNaN(referenceChannels)) {
403
+ throw new TypeError('aec referenceChannels must be a number');
404
+ }
405
+ if (referenceChannels < 1) {
406
+ throw new RangeError('aec referenceChannels must be at least 1');
407
+ }
408
+ aecReferenceChannels = referenceChannels;
409
+ }
400
410
  } else {
401
411
  throw new TypeError(
402
- `Invalid aec value: ${JSON.stringify(aec)}. Expected a model name such as 'tau', or a config object { model, tailMs, suppression, referenceSampleRate }.`
412
+ `Invalid aec value: ${JSON.stringify(aec)}. Expected a model name such as 'tau', or a config object { model, tailMs, suppression, referenceSampleRate, referenceChannels }.`
403
413
  );
404
414
  }
405
415
  }
@@ -443,6 +453,7 @@ class Microphone extends Readable {
443
453
  aecTailMs,
444
454
  aecSuppression,
445
455
  aecReferenceSampleRate,
456
+ aecReferenceChannels,
446
457
  },
447
458
  };
448
459
  }
@@ -605,8 +616,15 @@ class Microphone extends Readable {
605
616
  * Queue far-end reference audio for the echo canceller: the audio being
606
617
  * played out, pushed as it is played, in played order. Accepts the same
607
618
  * input shapes `Speaker.write` accepts (a `Buffer`, any TypedArray, or a
608
- * `DataView` of PCM bytes in this microphone's `dtype`), mono, at the
609
- * declared `referenceSampleRate` (the capture rate when unset).
619
+ * `DataView` of PCM bytes in this microphone's `dtype`), at the declared
620
+ * `referenceSampleRate` (the capture rate when unset), interleaved at the
621
+ * declared `referenceChannels` (mono when unset). With `referenceChannels`
622
+ * above 1, each frame is averaged to one mono sample before the canceller
623
+ * sees it; a multichannel reference pushed without declaring the count
624
+ * cancels nothing and reports no error. The declared count must match this
625
+ * buffer's actual interleaving: a mismatch is not detected and raises no
626
+ * error, and shows up only as `aecMetrics().delaySamples` staying `null`
627
+ * with no fault reported.
610
628
  *
611
629
  * Never blocks and never throws on a full queue: samples that do not fit
612
630
  * are discarded and counted by `aecMetrics().referenceDropped`, and the
@@ -714,19 +732,20 @@ function version() {
714
732
 
715
733
  class File extends Readable {
716
734
  /**
717
- * Open a WAV file as an offline source, synchronously. Everything a
735
+ * Open an audio file as an offline source, synchronously. Everything a
718
736
  * `Microphone` does to live audio, a `File` does to audio you already
719
737
  * have: the same conditioning options, the same stream of conditioned
720
738
  * chunks, and (with `vad` set) the same per-chunk speech events, plus the
721
739
  * whole-file `analyze()` a live stream cannot offer.
722
740
  *
723
- * The bare constructor reads the WAV inline, blocking the event loop on
741
+ * The bare constructor reads the file inline, blocking the event loop on
724
742
  * disk I/O; prefer `await File.open(path, options)` in servers and other
725
743
  * latency-sensitive code, exactly as `Microphone.open` is preferred over
726
744
  * `new Microphone`. Iteration and analysis are separate single passes:
727
745
  * each consumes the source once, so use one `File` per operation.
728
746
  *
729
- * @param {string} filePath Path to a WAV file (16-bit PCM or 32-bit float).
747
+ * @param {string} filePath Path to a WAV, AIFF, AIFF-C or FLAC file. The
748
+ * container is identified from the bytes, not from the extension.
730
749
  * @param {import('./decibri').FileOptions} [options]
731
750
  * @param {{ prepared: object, native: object }} [_internal] Internal: a
732
751
  * pre-resolved options bundle and an already-constructed native handle,
@@ -932,8 +951,8 @@ class File extends Readable {
932
951
  }
933
952
 
934
953
  /**
935
- * Open a WAV file without blocking the event loop: the disk read, WAV
936
- * parse, and chain construction run on the native thread pool. The
954
+ * Open an audio file without blocking the event loop: the disk read,
955
+ * decode, and chain construction run on the native thread pool. The
937
956
  * recommended form in Node, mirroring `Microphone.open`. The synchronous
938
957
  * `new File(path)` remains available for scripts.
939
958
  *
@@ -1132,10 +1151,10 @@ class File extends Readable {
1132
1151
  }
1133
1152
 
1134
1153
  /**
1135
- * The source's own rate, taken from the WAV header or from the `inputRate`
1136
- * passed to `File.buffer`. Differs from `sampleRate` when the source was
1137
- * resampled. Readable for the life of the File, including after the source
1138
- * is consumed or closed.
1154
+ * The source's own rate, taken from the file's header or from the
1155
+ * `inputRate` passed to `File.buffer`. Differs from `sampleRate` when the
1156
+ * source was resampled. Readable for the life of the File, including after
1157
+ * the source is consumed or closed.
1139
1158
  * @returns {number}
1140
1159
  */
1141
1160
  get inputRate() {
@@ -1190,6 +1209,77 @@ class File extends Readable {
1190
1209
  return this.analyze();
1191
1210
  }
1192
1211
 
1212
+ /**
1213
+ * Validate the save options and resolve them into the native options
1214
+ * object. Shared by `File.save` and `AudioWriter`, so the two spellings of
1215
+ * a write accept and reject identically.
1216
+ * @internal
1217
+ * @param {import('./decibri').SaveOptions} options
1218
+ */
1219
+ static _prepareSaveOptions(options) {
1220
+ const format = options.format;
1221
+ if (format !== undefined && format !== 'wav' && format !== 'aiff' && format !== 'flac') {
1222
+ throw new TypeError(
1223
+ `Invalid format value: ${JSON.stringify(format)}. Expected 'wav', 'aiff', or 'flac'.`
1224
+ );
1225
+ }
1226
+ const compression = options.compression;
1227
+ if (compression !== undefined) {
1228
+ if (typeof compression !== 'number' || Number.isNaN(compression)) {
1229
+ throw new TypeError('compression must be a number');
1230
+ }
1231
+ if (compression < 0 || compression > 8) {
1232
+ throw new RangeError('flac compression level must be between 0 and 8');
1233
+ }
1234
+ }
1235
+ return { format, compression };
1236
+ }
1237
+
1238
+ /**
1239
+ * Write the conditioned recording to disk, off the event loop. Runs the
1240
+ * recording once through the same conditioning pass iteration delivers,
1241
+ * whole, and writes it as 16-bit PCM mono at `sampleRate`. Consumes the
1242
+ * source: a save is a single pass, separate from iteration and analysis.
1243
+ *
1244
+ * The container comes from the path's extension (`.wav`, `.aiff`, `.aif`,
1245
+ * `.aifc` or `.flac`), or from `options.format`: decibri reads a file by
1246
+ * its content and writes one by its name. An extension it does not
1247
+ * recognise rejects rather than defaulting. `options.compression` sets the
1248
+ * FLAC compression level (0-8, default 5); it applies only to FLAC.
1249
+ *
1250
+ * Resolves to a `SaveReport`: `clippedSamples` counts finite samples
1251
+ * outside full scale clamped to `[-1.0, 1.0]` (AGC or AEC without a
1252
+ * limiter can overshoot, and 16-bit PCM cannot hold it), and
1253
+ * `nonFiniteSamples` counts NaN samples written as silence and infinite
1254
+ * samples written as full scale.
1255
+ *
1256
+ * Requires a `File` that is not already being streamed: once the stream
1257
+ * has been engaged this rejects with a `DecibriError` carrying the code
1258
+ * `'FILE_ENGAGED'`. Every failure detected before the pass begins leaves
1259
+ * the `File` usable; a failure during the pass consumes the source.
1260
+ *
1261
+ * @param {string} filePath
1262
+ * @param {import('./decibri').SaveOptions} [options]
1263
+ * @returns {Promise<import('./decibri').SaveReport>}
1264
+ */
1265
+ async save(filePath, options = {}) {
1266
+ // Checked before the native handle is touched: the rejection belongs to
1267
+ // this call, and the source is never taken from a File that keeps
1268
+ // streaming.
1269
+ if (this._engaged) {
1270
+ throw fileEngagedError();
1271
+ }
1272
+ if (typeof filePath !== 'string') {
1273
+ throw new TypeError('path must be a string');
1274
+ }
1275
+ const nativeOptions = File._prepareSaveOptions(options);
1276
+ try {
1277
+ return await this._native.save(filePath, nativeOptions);
1278
+ } catch (err) {
1279
+ throw wrapNativeError(err);
1280
+ }
1281
+ }
1282
+
1193
1283
  /**
1194
1284
  * Release the source. Idempotent; a closed File reads as ended.
1195
1285
  */
@@ -1198,10 +1288,131 @@ class File extends Readable {
1198
1288
  }
1199
1289
  }
1200
1290
 
1291
+ // ─── AudioWriter (Writable): file sink ───────────────────────────────────────
1292
+
1293
+ class AudioWriter extends Writable {
1294
+ /**
1295
+ * A file sink for PCM audio: the Writable to pair with decibri's Readable
1296
+ * sources, and with any other stream of PCM bytes (a TTS engine, a decoded
1297
+ * network stream). Collects the whole stream, then writes it as one audio
1298
+ * file when the stream finishes, exactly as `File.save` writes: the same
1299
+ * containers from the same extension rule, the same 16-bit PCM encoding,
1300
+ * the same clamp and non-finite handling, the same bytes.
1301
+ *
1302
+ * Chunks are raw PCM bytes in `dtype` ('int16' little-endian by default,
1303
+ * matching what a `File` or `Microphone` emits; 'float32' for raw f32
1304
+ * bytes). Audio is written mono: `channels` may only be 1. `sampleRate` is
1305
+ * required, because raw audio carries no header to read one from.
1306
+ *
1307
+ * The file is written when the stream finishes ('finish' fires after the
1308
+ * file is on disk), and `report` then carries the `SaveReport` the write
1309
+ * produced. A failure destroys the stream with the error.
1310
+ *
1311
+ * @param {string} filePath Path whose extension names the container,
1312
+ * unless `format` overrides it.
1313
+ * @param {import('./decibri').AudioWriterOptions} options
1314
+ */
1315
+ constructor(filePath, options = {}) {
1316
+ super({ highWaterMark: options.highWaterMark });
1317
+ if (typeof filePath !== 'string') {
1318
+ throw new TypeError('path must be a string');
1319
+ }
1320
+ const sampleRate = options.sampleRate;
1321
+ if (typeof sampleRate !== 'number' || Number.isNaN(sampleRate)) {
1322
+ throw new TypeError('sampleRate is required for AudioWriter (raw audio carries no header)');
1323
+ }
1324
+ if (sampleRate < 1000 || sampleRate > 384000) {
1325
+ throw new RangeError('sample rate must be between 1000 and 384000');
1326
+ }
1327
+ const channels = options.channels;
1328
+ if (channels !== undefined && channels !== 1) {
1329
+ throw new RangeError('multichannel write is not supported; channels must be 1 (mono)');
1330
+ }
1331
+ const dtype = options.dtype ?? 'int16';
1332
+ if (dtype !== 'int16' && dtype !== 'float32') {
1333
+ throw new TypeError("dtype must be 'int16' or 'float32'");
1334
+ }
1335
+ // Validated now, so a bad option is a construction-time throw rather
1336
+ // than a deferred 'error' event after the audio has been streamed.
1337
+ this._saveOptions = File._prepareSaveOptions(options);
1338
+ this._filePath = filePath;
1339
+ this._sampleRate = sampleRate;
1340
+ this._dtype = dtype;
1341
+ this._chunks = [];
1342
+ this._report = null;
1343
+ }
1344
+
1345
+ /**
1346
+ * The `SaveReport` of the completed write: `clippedSamples` and
1347
+ * `nonFiniteSamples`, exactly as `File.save` resolves. `null` until
1348
+ * 'finish' has fired.
1349
+ * @returns {import('./decibri').SaveReport | null}
1350
+ */
1351
+ get report() {
1352
+ return this._report;
1353
+ }
1354
+
1355
+ /** @internal */
1356
+ _write(chunk, encoding, callback) {
1357
+ this._chunks.push(chunk);
1358
+ callback();
1359
+ }
1360
+
1361
+ /** @internal */
1362
+ _final(callback) {
1363
+ const bytes = Buffer.concat(this._chunks);
1364
+ this._chunks = [];
1365
+ const bytesPerSample = this._dtype === 'int16' ? 2 : 4;
1366
+ if (bytes.length % bytesPerSample !== 0) {
1367
+ callback(
1368
+ new RangeError(
1369
+ `audio bytes do not divide into whole ${this._dtype} samples; ` +
1370
+ `${bytes.length % bytesPerSample} byte(s) over`
1371
+ )
1372
+ );
1373
+ return;
1374
+ }
1375
+ const samples = new Float32Array(bytes.length / bytesPerSample);
1376
+ if (this._dtype === 'int16') {
1377
+ // The inverse of the int16 delivery encoding: value / 32768, so bytes
1378
+ // that came from a decibri source re-quantise to the identical file.
1379
+ for (let i = 0; i < samples.length; i++) {
1380
+ samples[i] = bytes.readInt16LE(i * 2) / 32768;
1381
+ }
1382
+ } else {
1383
+ for (let i = 0; i < samples.length; i++) {
1384
+ samples[i] = bytes.readFloatLE(i * 4);
1385
+ }
1386
+ }
1387
+ // The write is File.save on a source at the writer's own rate: the same
1388
+ // encode path, so the two spellings produce the same bytes.
1389
+ let file;
1390
+ try {
1391
+ file = File.buffer(samples, {
1392
+ inputRate: this._sampleRate,
1393
+ sampleRate: this._sampleRate,
1394
+ });
1395
+ } catch (err) {
1396
+ callback(err);
1397
+ return;
1398
+ }
1399
+ file
1400
+ .save(this._filePath, this._saveOptions)
1401
+ .then((report) => {
1402
+ this._report = report;
1403
+ callback();
1404
+ })
1405
+ .catch((err) => {
1406
+ callback(err);
1407
+ });
1408
+ }
1409
+ }
1410
+
1201
1411
  module.exports = {
1202
1412
  Microphone,
1203
1413
  Speaker,
1204
1414
  File,
1415
+ AudioWriter,
1205
1416
  inputDevices,
1206
1417
  outputDevices,
1207
1418
  version,
package/src/errors.js CHANGED
@@ -75,13 +75,15 @@ class OrtPathError extends OrtError {
75
75
  // the core ever sees it).
76
76
  const RANGE_PREFIXES = [
77
77
  'sample rate must be between',
78
- 'channels must be between',
78
+ 'channels must be at least',
79
79
  'multichannel capture is not supported',
80
80
  'frames per buffer must be between',
81
81
  'agc target level must be between',
82
82
  'limiter ceiling must be between',
83
+ 'flac compression level must be between',
83
84
  'aec tailMs must be between',
84
85
  'aec referenceSampleRate must be between',
86
+ 'aec referenceChannels must be at',
85
87
  'Silero VAD only supports',
86
88
  'VAD threshold must be between',
87
89
  'echo cancellation only supports',
@@ -93,6 +95,7 @@ const TYPE_PREFIXES = [
93
95
  "dtype must be 'int16' or 'float32'",
94
96
  "format must be 'int16' or 'float32'",
95
97
  'aec suppression must be',
98
+ 'Invalid format value:',
96
99
  ];
97
100
 
98
101
  const DEVICE_CODES = [
@@ -123,6 +126,7 @@ const BASE_CODES = [
123
126
  ['audio stream is already running', 'ALREADY_RUNNING'],
124
127
  ['Failed to open audio stream', 'STREAM_OPEN_FAILED'],
125
128
  ['Failed to start audio stream', 'STREAM_START_FAILED'],
129
+ ['the output device does not support', 'SPEAKER_CHANNELS_UNSUPPORTED'],
126
130
  ['Microphone permission denied.', 'PERMISSION_DENIED'],
127
131
  ['Microphone stream is closed', 'MICROPHONE_STREAM_CLOSED'],
128
132
  ['Speaker stream is closed', 'SPEAKER_STREAM_CLOSED'],
@@ -132,7 +136,10 @@ const BASE_CODES = [
132
136
  ['resampler error:', 'RESAMPLE_FAILED'],
133
137
  ['echo canceller configuration error:', 'AEC_CONFIG_INVALID'],
134
138
  ['Failed to read audio file', 'FILE_READ_FAILED'],
135
- ['invalid WAV file:', 'WAV_INVALID'],
139
+ ['Failed to write audio file', 'FILE_WRITE_FAILED'],
140
+ ['unsupported audio format:', 'AUDIO_FORMAT_UNSUPPORTED'],
141
+ ['malformed audio file:', 'AUDIO_FILE_MALFORMED'],
142
+ ['truncated audio file:', 'AUDIO_FILE_TRUNCATED'],
136
143
  ['ONNX Runtime was initialized in pid', 'FORK_AFTER_ORT_INIT'],
137
144
  ['ONNX backend error from', 'ONNX_BACKEND_FAILED'],
138
145
  // Authored in the napi layer (bindings/node/src/lib.rs), not in