decibri 5.2.5 → 5.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/decibri.d.ts CHANGED
@@ -61,6 +61,108 @@ export interface VadOptions {
61
61
  holdoffMs?: number;
62
62
  }
63
63
 
64
+ /**
65
+ * Acoustic echo cancellation config object, passed on the `aec` option to tune
66
+ * the canceller. The bare `aec: 'tau'` shorthand selects the model with its
67
+ * defaults; pass this object to override them.
68
+ *
69
+ * The canceller's behaviour while a lost delay alignment is being reacquired
70
+ * is fixed: decibri applies the canceller's graded output transition and does
71
+ * not expose a setting for it.
72
+ */
73
+ export interface AecOptions {
74
+ /**
75
+ * Which echo canceller model to run. The accepted set is owned by the
76
+ * canceller and grows the way `denoise` grows; today it is `'tau'`, a
77
+ * classical adaptive canceller with no model file. An unknown name is
78
+ * rejected with the canceller's own message naming the accepted set.
79
+ */
80
+ model: 'tau';
81
+
82
+ /**
83
+ * Adaptive filter tail length in milliseconds: how much echo delay spread
84
+ * the canceller can model.
85
+ * @default the canceller's own default (200)
86
+ * @range 16 to 500
87
+ */
88
+ tailMs?: number;
89
+
90
+ /**
91
+ * Residual echo suppression policy. `'conservative'` attenuates the residual
92
+ * echo the linear canceller leaves behind while keeping the near-end voice
93
+ * intact; `'off'` delivers the linear canceller output as-is.
94
+ * @default 'conservative'
95
+ */
96
+ suppression?: 'conservative' | 'off';
97
+
98
+ /**
99
+ * Sample rate in Hz of the far-end reference pushed through
100
+ * `pushAecReference`. When it names a rate other than `sampleRate`, decibri
101
+ * converts the reference before the canceller sees it: a reference at an
102
+ * undeclared different rate cancels nothing and reports no error, so the
103
+ * conversion is decibri's rather than the caller's.
104
+ * @default the capture `sampleRate`
105
+ * @range 1000 to 384000
106
+ */
107
+ referenceSampleRate?: number;
108
+ }
109
+
110
+ /**
111
+ * The echo canceller's transport and cancellation metrics, returned by
112
+ * `Microphone.aecMetrics()`. One object carries the canceller's own report and
113
+ * the reference queue's counters.
114
+ */
115
+ export interface AecMetrics {
116
+ /**
117
+ * The active delay alignment in samples, or `null` while the estimator is
118
+ * still searching. Staying `null` while `acquisitionParked` climbs is the
119
+ * signature of a canceller with no usable reference: none pushed, not at the
120
+ * declared rate, or not the signal that produced the echo.
121
+ */
122
+ delaySamples: number | null;
123
+ /**
124
+ * Smoothed echo-return-loss-enhancement estimate in dB: how much echo the
125
+ * canceller is currently removing. 0 before the filter has converged.
126
+ */
127
+ erleDb: number;
128
+ /**
129
+ * Whether the double-talk detector currently believes the near-end talker is
130
+ * active; adaptation is held while true.
131
+ */
132
+ doubleTalk: boolean;
133
+ /**
134
+ * Near-end samples the canceller could find no far-end sample for while an
135
+ * alignment was active. decibri keeps the far-end stream level with the
136
+ * capture, so this stays 0 for a caller who simply stops pushing; a non-zero
137
+ * count means the caller ran further ahead of the capture than the
138
+ * canceller's far-end history reaches.
139
+ */
140
+ referenceStarved: number;
141
+ /**
142
+ * Near-end samples processed while no delay alignment was active: the
143
+ * searching span, not a transport failure.
144
+ */
145
+ acquisitionParked: number;
146
+ /**
147
+ * Times the canceller inferred a capture discontinuity and rebuilt its
148
+ * alignment from the reference frontier.
149
+ */
150
+ referenceReanchors: number;
151
+ /**
152
+ * Far-end samples discarded because a single push exceeded the reference
153
+ * queue's bound, at the declared reference rate. The span they occupied is
154
+ * still represented as silence, so a discard costs the cancellation of that
155
+ * span alone.
156
+ */
157
+ referenceDropped: number;
158
+ /**
159
+ * Far-end samples decibri supplied as silence because the caller had pushed
160
+ * none for them, at the capture rate. An accounting figure, not a fault:
161
+ * while nothing is playing, the far end is silence.
162
+ */
163
+ referenceSilence: number;
164
+ }
165
+
64
166
  /** Constructor options for `Microphone`. */
65
167
  export interface MicrophoneOptions extends ReadableOptions {
66
168
  /**
@@ -186,6 +288,26 @@ export interface MicrophoneOptions extends ReadableOptions {
186
288
  * @default undefined
187
289
  */
188
290
  limiter?: number;
291
+
292
+ /**
293
+ * Acoustic echo cancellation applied to the captured audio, removing the
294
+ * echo of far-end audio the caller pushes through `pushAecReference`. The
295
+ * `'tau'` shorthand names the model; an `AecOptions` object tunes it. Omit
296
+ * to leave echo cancellation off (the default), which keeps the capture
297
+ * path unchanged.
298
+ *
299
+ * Runs before the detector tap: with it on, `vadScore` and the `speech` /
300
+ * `silence` events read the echo-removed signal, so playback stops
301
+ * triggering detection. Requires `sampleRate` in 8000 to 48000, narrower
302
+ * than the range the option otherwise accepts. With no reference pushed,
303
+ * the captured audio passes through unchanged.
304
+ *
305
+ * Native capture only: the browser entry does not take this option, because
306
+ * browser capture already carries the platform's own echo cancellation
307
+ * through its `echoCancellation` constraint, on by default.
308
+ * @default undefined
309
+ */
310
+ aec?: 'tau' | AecOptions;
189
311
  }
190
312
 
191
313
  /**
@@ -245,6 +367,27 @@ export declare class Microphone extends Readable {
245
367
  */
246
368
  readonly overrunCount: number;
247
369
 
370
+ /**
371
+ * Queue far-end reference audio for the echo canceller: the audio being
372
+ * played out, pushed as it is played, in played order. Accepts the same
373
+ * input shapes `Speaker.write` accepts (a `Buffer`, any TypedArray, or a
374
+ * `DataView` of PCM bytes in this microphone's `dtype`), mono, at the
375
+ * declared `referenceSampleRate` (the capture rate when unset).
376
+ *
377
+ * Never blocks and never throws on a full queue: samples that do not fit
378
+ * are discarded and counted by `aecMetrics().referenceDropped`. Silence
379
+ * between played audio need not be pushed. A push while capture is not
380
+ * running, or with the `aec` option unset, is a no-op.
381
+ */
382
+ pushAecReference(data: Buffer | NodeJS.ArrayBufferView): void;
383
+
384
+ /**
385
+ * The echo canceller's transport and cancellation metrics, merged with the
386
+ * reference queue's counters, or `null` when the `aec` option is unset or
387
+ * capture is not running.
388
+ */
389
+ aecMetrics(): AecMetrics | null;
390
+
248
391
  /** List all available audio input devices. */
249
392
  static devices(): MicrophoneInfo[];
250
393
 
@@ -280,7 +423,7 @@ export declare class Microphone extends Readable {
280
423
  export interface FileOptions extends ReadableOptions {
281
424
  /**
282
425
  * Target output rate in Hz: the rate every delivered chunk carries. The
283
- * source's input rate (from the WAV header, or `inputRate` for
426
+ * source's input rate (from the file's header, or `inputRate` for
284
427
  * `File.buffer`) is resampled to this rate, so a 44.1 kHz recording comes
285
428
  * out at 16 kHz unless you set `sampleRate`. The same meaning the option
286
429
  * has on `Microphone`.
@@ -408,6 +551,46 @@ export interface VadReport {
408
551
  segments: Segment[];
409
552
  }
410
553
 
554
+ /** Options for `File.save` (also accepted by `AudioWriter`). */
555
+ export interface SaveOptions {
556
+ /**
557
+ * The container format to write. When not given it comes from the path's
558
+ * extension: `.wav`, `.aiff`, `.aif`, `.aifc` or `.flac`. decibri reads a
559
+ * file by its content and writes one by its name; an extension it does not
560
+ * recognise is an error, never a silent default.
561
+ */
562
+ format?: 'wav' | 'aiff' | 'flac';
563
+
564
+ /**
565
+ * FLAC compression level. Higher levels search harder for a smaller file;
566
+ * every level decodes to identical audio. Applies only to FLAC; ignored
567
+ * for WAV and AIFF.
568
+ * @default 5
569
+ * @range 0 to 8
570
+ */
571
+ compression?: number;
572
+ }
573
+
574
+ /**
575
+ * What a save did to the samples on their way into the file, resolved by
576
+ * `File.save` and carried by `AudioWriter.report`.
577
+ */
578
+ export interface SaveReport {
579
+ /**
580
+ * Finite samples outside full scale, clamped to [-1.0, 1.0]. Conditioned
581
+ * audio can exceed full scale (AGC or AEC without a limiter), and 16-bit
582
+ * PCM cannot hold that, so the overshoot clips and this count says how
583
+ * much. The count is a statement about integer encodings: a float encoding
584
+ * would preserve the overshoot instead, and would report zero.
585
+ */
586
+ clippedSamples: number;
587
+ /**
588
+ * Non-finite samples replaced before writing: NaN with silence, an
589
+ * infinity with full scale. The same replacement on every format.
590
+ */
591
+ nonFiniteSamples: number;
592
+ }
593
+
411
594
  /**
412
595
  * Offline audio source: conditions a recording or in-memory samples through
413
596
  * the same chain as the live `Microphone`, delivered as a finite Readable
@@ -416,7 +599,7 @@ export interface VadReport {
416
599
  * analyze the whole recording for speech with `analyze()` / `analyse()`,
417
600
  * which a live stream cannot do.
418
601
  *
419
- * Construction: `new File(path)` reads the WAV synchronously (fine for a
602
+ * Construction: `new File(path)` reads the file synchronously (fine for a
420
603
  * script; it blocks the event loop on disk I/O), `await File.open(path)` reads
421
604
  * it off the event loop (the recommended form, mirroring `Microphone.open`),
422
605
  * and `File.buffer(samples, { inputRate })` wraps a `Float32Array` of samples
@@ -443,15 +626,16 @@ export interface VadReport {
443
626
  */
444
627
  export declare class File extends Readable {
445
628
  /**
446
- * Open a WAV file synchronously (blocks on disk I/O; prefer `File.open`
447
- * in servers). Supports 16-bit PCM and 32-bit float WAV files; the input
448
- * rate and channel count come from the header.
629
+ * Open an audio file synchronously (blocks on disk I/O; prefer `File.open`
630
+ * in servers). Reads WAV, AIFF, AIFF-C and FLAC, identified from the
631
+ * file's own bytes rather than its extension; the input rate and channel
632
+ * count come from the header.
449
633
  */
450
634
  constructor(path: string, options?: FileOptions);
451
635
 
452
636
  /**
453
- * Open a WAV file without blocking the event loop: the disk read, WAV
454
- * parse, and chain construction run on the native thread pool. The
637
+ * Open an audio file without blocking the event loop: the disk read,
638
+ * decode, and chain construction run on the native thread pool. The
455
639
  * recommended form, mirroring `Microphone.open`.
456
640
  */
457
641
  static open(path: string, options?: FileOptions): Promise<File>;
@@ -479,7 +663,7 @@ export declare class File extends Readable {
479
663
  readonly sampleRate: number;
480
664
 
481
665
  /**
482
- * The source's own rate, taken from the WAV header or from the
666
+ * The source's own rate, taken from the file's header or from the
483
667
  * `inputRate` passed to `File.buffer`. Differs from `sampleRate` when the
484
668
  * source was resampled.
485
669
  */
@@ -503,6 +687,26 @@ export declare class File extends Readable {
503
687
  /** The same whole-recording analysis under the international spelling. */
504
688
  analyse(): Promise<VadReport>;
505
689
 
690
+ /**
691
+ * Write the conditioned recording to disk, off the event loop. Runs the
692
+ * recording once through the same conditioning pass iteration delivers,
693
+ * whole, and writes it as 16-bit PCM mono at `sampleRate`. The container
694
+ * comes from the path's extension (`.wav`, `.aiff`, `.aif`, `.aifc` or
695
+ * `.flac`), or from `options.format`: decibri reads a file by its content
696
+ * and writes one by its name. Consumes the source (a `File` is a single
697
+ * pass).
698
+ *
699
+ * Resolves to a `SaveReport`: how many samples were clamped to full scale
700
+ * and how many non-finite samples were replaced (NaN as silence, an
701
+ * infinity as full scale).
702
+ *
703
+ * Requires a File that is not already being streamed: once the stream has
704
+ * been engaged this rejects with a `DecibriError` carrying the code
705
+ * `'FILE_ENGAGED'`. Every failure detected before the pass begins leaves
706
+ * the File usable; a failure during the pass consumes the source.
707
+ */
708
+ save(path: string, options?: SaveOptions): Promise<SaveReport>;
709
+
506
710
  /** Release the source. Idempotent; a closed File reads as ended. */
507
711
  close(): void;
508
712
 
@@ -523,6 +727,61 @@ export declare class File extends Readable {
523
727
  once(event: string | symbol, listener: (...args: any[]) => void): this;
524
728
  }
525
729
 
730
+ /** Constructor options for `AudioWriter`: the save options plus the stream's own description. */
731
+ export interface AudioWriterOptions extends SaveOptions, WritableOptions {
732
+ /**
733
+ * The rate of the incoming samples in Hz, written into the file's header.
734
+ * Required: raw audio carries no header to read a rate from.
735
+ * @range 1000 to 384000
736
+ */
737
+ sampleRate: number;
738
+
739
+ /**
740
+ * Number of channels. Audio is written mono; only `1` is accepted.
741
+ * @default 1
742
+ */
743
+ channels?: 1;
744
+
745
+ /**
746
+ * Sample encoding of the incoming bytes.
747
+ * - `'int16'`: 16-bit signed integer, little-endian, what a `File` or
748
+ * `Microphone` emits by default
749
+ * - `'float32'`: 32-bit IEEE 754 float, little-endian
750
+ * @default 'int16'
751
+ */
752
+ dtype?: 'int16' | 'float32';
753
+ }
754
+
755
+ /**
756
+ * A file sink for PCM audio: the Writable to pair with decibri's Readable
757
+ * sources, and with any other stream of PCM bytes (a TTS engine, a decoded
758
+ * network stream). Collects the whole stream, then writes it as one audio
759
+ * file when the stream finishes, exactly as `File.save` writes: the same
760
+ * containers from the same extension rule, the same 16-bit PCM encoding, the
761
+ * same clamp and non-finite handling, the same bytes.
762
+ *
763
+ * `'finish'` fires after the file is on disk, and `report` then carries the
764
+ * `SaveReport` the write produced. A failure destroys the stream with the
765
+ * error.
766
+ *
767
+ * @example
768
+ * const { pipeline } = require('node:stream/promises');
769
+ * const { File, AudioWriter } = require('decibri');
770
+ * await pipeline(
771
+ * new File('noisy.wav', { denoise: 'fastenhancer-t' }),
772
+ * new AudioWriter('clean.flac', { sampleRate: 16000 }),
773
+ * );
774
+ */
775
+ export declare class AudioWriter extends Writable {
776
+ constructor(path: string, options: AudioWriterOptions);
777
+
778
+ /**
779
+ * The `SaveReport` of the completed write, exactly as `File.save` resolves
780
+ * it. `null` until `'finish'` has fired.
781
+ */
782
+ readonly report: SaveReport | null;
783
+ }
784
+
526
785
  export interface SpeakerInfo {
527
786
  /** Device index (pass to constructor as `device`). */
528
787
  index: number;