livekit-plugins-denoise 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,222 @@
1
+ """DeepFilterNet3 noise suppression for unknown telephony environments."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import threading
6
+ import time
7
+ from typing import Any
8
+
9
+ import numpy as np
10
+
11
+ from ._resample import StreamResampler
12
+ from .log import logger
13
+
14
+ _PRIME_LIMIT = 64 # pre-roll blocks before we give up and warn
15
+
16
+ # How long to sit out after a failed load. Retrying per call would stall every
17
+ # call on the same slow failure; never retrying would let one flaky download at
18
+ # worker start disable neural suppression for the life of the process.
19
+ _RETRY_AFTER_SECONDS = 60.0
20
+
21
+ _model_lock = threading.Lock()
22
+ _shared_model: Any | None = None
23
+ _failed_at = 0.0
24
+
25
+
26
+ def _get_model() -> Any:
27
+ """Load one ONNX session per process. Thread-safe to share; streams are not."""
28
+
29
+ global _shared_model, _failed_at
30
+
31
+ # Loading downloads the weights on first use, so hold the lock across the
32
+ # whole thing: concurrent calls should wait, not each start a download.
33
+ with _model_lock:
34
+ if _shared_model is not None:
35
+ return _shared_model
36
+
37
+ waited = time.monotonic() - _failed_at
38
+ if _failed_at and waited < _RETRY_AFTER_SECONDS:
39
+ raise RuntimeError(
40
+ f"DeepFilterNet3 unavailable, retrying in "
41
+ f"{_RETRY_AFTER_SECONDS - waited:.0f}s"
42
+ )
43
+
44
+ try:
45
+ from deepfilter_stream import DeepFilterModel
46
+
47
+ # One ONNX thread per session so concurrent SIP calls do not thrash.
48
+ _shared_model = DeepFilterModel(intra_op_num_threads=1)
49
+ except Exception:
50
+ _failed_at = time.monotonic()
51
+ logger.exception("DeepFilterNet3 failed to load; falling back to WebRTC NS")
52
+ raise
53
+
54
+ _failed_at = 0.0
55
+ logger.info(
56
+ "DeepFilterNet3 model loaded: %d Hz, %d-sample frames",
57
+ _shared_model.sample_rate,
58
+ _shared_model.frame_size,
59
+ )
60
+ return _shared_model
61
+
62
+
63
+ def prewarm() -> None:
64
+ """Download and load the model up front.
65
+
66
+ The first call otherwise pays for a network fetch and an ONNX session build
67
+ inside its first audio frame, on the event loop. Call this from a worker's
68
+ setup hook.
69
+ """
70
+
71
+ _get_model()
72
+
73
+
74
+ class NeuralEnhancer:
75
+ """Per-call DeepFilterNet stream. Always processes at 48 kHz internally.
76
+
77
+ Feeding 8/16 kHz chunks through `process(sr=...)` builds a large resample
78
+ buffer and adds hundreds of milliseconds of delay, so we resample ourselves
79
+ and call `process_frame` on whole model frames.
80
+
81
+ Delay across this stage is the model's own 32 ms lookahead plus a few ms of
82
+ resampler group delay: measured with the model running, roughly 43 ms at
83
+ 48 kHz, 48 ms at 16 kHz and 53 ms at 8 kHz. The model dominates, which is
84
+ the point of `_resample`; a general-purpose resampler put 8 kHz at 172 ms.
85
+ The delay is constant for the life of the stream -- see
86
+ tests/test_neural_plumbing.py, which pins the resampler part with the model
87
+ stubbed out.
88
+
89
+ One stream per call. Not thread-safe, and closing it is final.
90
+ """
91
+
92
+ def __init__(self) -> None:
93
+ model = _get_model()
94
+ try:
95
+ self._stream = model.new_stream()
96
+ except Exception:
97
+ logger.exception("DeepFilterNet3 stream failed; falling back to WebRTC NS")
98
+ raise
99
+
100
+ # Read the geometry off the loaded graph. Hardcoding 512 meant a model
101
+ # asset with a different hop raised on every single frame.
102
+ self._model_rate = int(model.sample_rate)
103
+ self._frame = int(model.frame_size)
104
+
105
+ self._closed = False
106
+ self._rate = 0
107
+ self._up_rs: StreamResampler | None = None
108
+ self._down_rs: StreamResampler | None = None
109
+ self._up = np.zeros(0, dtype=np.float32)
110
+ self._down = np.zeros(0, dtype=np.float32)
111
+
112
+ def _configure(self, sample_rate: int) -> None:
113
+ """Rebuild the resamplers and pre-roll the output buffer for a new rate."""
114
+
115
+ self._rate = sample_rate
116
+ self._stream.reset()
117
+ self._up = np.zeros(0, dtype=np.float32)
118
+ self._down = np.zeros(0, dtype=np.float32)
119
+
120
+ # These carry their filter state across blocks. Resampling each block
121
+ # independently restarts that state every time, and the transient at
122
+ # each block edge is loud enough to hear as distortion. At the model's
123
+ # own rate they are pass-throughs and cost nothing.
124
+ self._up_rs = StreamResampler(sample_rate, self._model_rate)
125
+ self._down_rs = StreamResampler(self._model_rate, sample_rate)
126
+
127
+ # Output arrives in whole model frames and trails input by the
128
+ # group delay of both resamplers, so the buffer starts empty and stays
129
+ # that way for a while. Left alone it runs dry mid-call and we splice in
130
+ # silence, which both clicks and pushes the audio permanently later --
131
+ # that is how latency crept past 300 ms at 8 kHz. Push silence through
132
+ # the whole chain now so the delay is flushed before real audio arrives,
133
+ # and keep one frame in hand to absorb the frame-boundary jitter.
134
+ quantum = self.quantum
135
+ for _ in range(_PRIME_LIMIT):
136
+ if self._down.size >= quantum:
137
+ break
138
+ self._pump(np.zeros(quantum, dtype=np.float32))
139
+ else:
140
+ logger.warning(
141
+ "DeepFilterNet pre-roll did not fill at %d Hz; expect a brief warm-up",
142
+ sample_rate,
143
+ )
144
+
145
+ @property
146
+ def quantum(self) -> int:
147
+ """One model frame expressed in caller-rate samples.
148
+
149
+ The output buffer can only ever fall this far behind the input, whatever
150
+ block size the caller uses, so it is also the pre-roll target.
151
+ """
152
+
153
+ return int(np.ceil(self._frame * self._rate / self._model_rate))
154
+
155
+ def _pump(self, mono: np.ndarray) -> None:
156
+ """Run one block of float32 audio through resample -> model -> resample."""
157
+
158
+ assert self._up_rs is not None and self._down_rs is not None
159
+
160
+ up = self._up_rs.process(mono)
161
+ if up.size:
162
+ self._up = np.concatenate((self._up, up))
163
+
164
+ enhanced: list[np.ndarray] = []
165
+ while self._up.size >= self._frame:
166
+ frame = self._up[: self._frame]
167
+ self._up = self._up[self._frame :]
168
+ enhanced.append(self._stream.process_frame(frame))
169
+
170
+ if not enhanced:
171
+ return
172
+
173
+ chunk = np.concatenate(enhanced).astype(np.float32)
174
+ if not np.isfinite(chunk).all():
175
+ # The resampler keeps a window of history, so a single NaN out of the
176
+ # model would smear across every later block instead of one frame.
177
+ logger.warning("DeepFilterNet3 returned non-finite samples; zeroing them")
178
+ chunk = np.nan_to_num(chunk, nan=0.0, posinf=0.0, neginf=0.0)
179
+
180
+ down = self._down_rs.process(chunk)
181
+ if down.size:
182
+ self._down = np.concatenate((self._down, down))
183
+
184
+ def process(self, pcm: np.ndarray, sample_rate: int) -> np.ndarray:
185
+ """Enhance a mono int16 block. Returns the same number of int16 samples."""
186
+
187
+ if self._closed:
188
+ raise RuntimeError("NeuralEnhancer is closed")
189
+ if sample_rate != self._rate:
190
+ self._configure(sample_rate)
191
+
192
+ wanted = int(pcm.size)
193
+ self._pump(pcm.astype(np.float32) * (1.0 / 32768.0))
194
+
195
+ if self._down.size >= wanted:
196
+ out = self._down[:wanted]
197
+ self._down = self._down[wanted:]
198
+ else:
199
+ # The pre-roll is sized so this cannot happen in steady state. Reaching
200
+ # it means the chain lost samples, so splice silence to keep the frame
201
+ # contract rather than hand back a short block.
202
+ pad = wanted - int(self._down.size)
203
+ logger.warning("DeepFilterNet output ran dry, padding %d samples", pad)
204
+ out = np.concatenate((self._down, np.zeros(pad, dtype=np.float32)))
205
+ self._down = np.zeros(0, dtype=np.float32)
206
+
207
+ return np.clip(np.rint(out * 32768.0), -32768, 32767).astype(np.int16)
208
+
209
+ def close(self) -> None:
210
+ """Release the stream. The enhancer cannot be used again afterwards."""
211
+
212
+ if self._closed:
213
+ return
214
+ self._closed = True
215
+ self._rate = 0
216
+ self._up_rs = self._down_rs = None
217
+ self._up = np.zeros(0, dtype=np.float32)
218
+ self._down = np.zeros(0, dtype=np.float32)
219
+ try:
220
+ self._stream.reset()
221
+ except Exception:
222
+ logger.debug("DeepFilterNet stream reset failed on close", exc_info=True)
@@ -0,0 +1,455 @@
1
+ """LiveKit frame processor for runtime telephony noise and echo suppression."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import threading
6
+ from dataclasses import dataclass
7
+ from typing import Literal, get_args
8
+
9
+ import numpy as np
10
+ from livekit import rtc
11
+
12
+ from ._buffers import Int16Buffer, to_mono
13
+ from .log import logger
14
+ from .neural import NeuralEnhancer
15
+
16
+ _BLOCK_MS = 10
17
+ """The WebRTC APM contract: audio must be handed over in 10 ms blocks."""
18
+
19
+ _RATE_QUANTUM = 1000 // _BLOCK_MS
20
+ """A supported rate must divide by this, so a block is a whole number of samples."""
21
+
22
+ _MAX_PENDING_RENDER_MS = 2000.0
23
+ """How much far-end audio to hold before the first inbound frame sets the format."""
24
+
25
+ Enhancer = Literal["deepfilter", "webrtc"]
26
+
27
+
28
+ @dataclass
29
+ class DenoiseOptions:
30
+ """Tuning for `TelephonyDenoiser`."""
31
+
32
+ echo_cancellation: bool = True
33
+ noise_suppression: bool = True
34
+ high_pass_filter: bool = True
35
+ auto_gain_control: bool = True
36
+
37
+ enhancer: Enhancer = "deepfilter"
38
+ """Noise suppressor for unknown caller environments.
39
+
40
+ `deepfilter` (default) runs DeepFilterNet3 after AEC and handles gym / café /
41
+ babble as well as line hiss. `webrtc` keeps the classical WebRTC NS for
42
+ minimum latency. Never stack both.
43
+ """
44
+
45
+ stream_delay_ms: int = 120
46
+ """Round trip delay between us sending audio and its echo arriving back.
47
+
48
+ AEC3 refines this internally, but a sane starting point matters on SIP where
49
+ the PSTN leg adds far more delay than a WebRTC client would.
50
+ """
51
+
52
+ def __post_init__(self) -> None:
53
+ if self.enhancer not in get_args(Enhancer):
54
+ raise ValueError(
55
+ f"enhancer must be one of {get_args(Enhancer)}, got {self.enhancer!r}"
56
+ )
57
+ if self.stream_delay_ms < 0:
58
+ raise ValueError(
59
+ f"stream_delay_ms must not be negative, got {self.stream_delay_ms}"
60
+ )
61
+
62
+
63
+ class TelephonyDenoiser(rtc.FrameProcessor[rtc.AudioFrame]):
64
+ """Suppresses background noise and acoustic echo on an inbound audio track.
65
+
66
+ The inbound (caller) audio flows through `process`, which LiveKit calls for
67
+ every frame. For echo cancellation the filter also needs the far-end
68
+ reference, meaning the audio the agent sends to the caller; feed that in via
69
+ `push_render_frame`, or let `EchoReferenceTap` do it for you.
70
+
71
+ One instance holds the echo path and noise estimate for a single
72
+ conversation. Build a new one per call; never share across calls.
73
+ """
74
+
75
+ def __init__(self, opts: DenoiseOptions | None = None) -> None:
76
+ self._opts = opts or DenoiseOptions()
77
+ self._enabled = True
78
+ self._closed = False
79
+
80
+ # The format we gave up on, if any. Held rather than a bare flag so a
81
+ # renegotiation to a format we do support starts filtering again.
82
+ self._degraded: tuple[int, int] | None = None
83
+ self._logged: set[str] = set()
84
+
85
+ # `process` runs on the room's event loop while `push_render_frame` is
86
+ # driven by the TTS output path, which may be a different thread.
87
+ self._lock = threading.Lock()
88
+
89
+ self._apm: rtc.AudioProcessingModule | None = None
90
+ self._rate = 0
91
+ self._channels = 0
92
+ self._block = 0
93
+
94
+ self._capture_in = Int16Buffer()
95
+ self._capture_out = Int16Buffer()
96
+
97
+ self._render_in = Int16Buffer()
98
+ self._render_resampler: rtc.AudioResampler | None = None
99
+ self._render_resampler_rate = 0
100
+ self._pending_render: list[rtc.AudioFrame] = []
101
+ self._pending_render_ms = 0.0
102
+
103
+ self._neural: NeuralEnhancer | None = None
104
+ self._use_neural = False
105
+ self._webrtc_ns = False
106
+
107
+ @property
108
+ def enabled(self) -> bool:
109
+ return self._enabled
110
+
111
+ @enabled.setter
112
+ def enabled(self, value: bool) -> None:
113
+ self._enabled = value
114
+
115
+ @property
116
+ def enhancer(self) -> Enhancer | None:
117
+ """Active noise suppressor, or None if it is off or not yet negotiated.
118
+
119
+ The backend is only known once the first frame has revealed the stream
120
+ format, because that is when a DeepFilterNet failure would fall back.
121
+ """
122
+
123
+ if not self._opts.noise_suppression or self._apm is None:
124
+ return None
125
+ return "deepfilter" if self._use_neural else "webrtc"
126
+
127
+ def prepare(self, sample_rate: int, num_channels: int = 1) -> bool:
128
+ """Build the filter ahead of the first frame.
129
+
130
+ Optional: the format is picked up from the first inbound frame anyway.
131
+ Doing it up front keeps the model load off the first frame's deadline and
132
+ lets far-end audio be used as a reference from the very first word.
133
+
134
+ Returns False if the format is not supported, in which case audio will
135
+ pass through untouched.
136
+ """
137
+
138
+ with self._lock:
139
+ if self._closed:
140
+ return False
141
+ return self._ensure_apm(sample_rate, num_channels)
142
+
143
+ def process(self, frame: rtc.AudioFrame) -> rtc.AudioFrame:
144
+ """Filter one inbound frame. Returns a frame of the same shape."""
145
+
146
+ return self._process(frame)
147
+
148
+ def close(self) -> None:
149
+ """Release the filter. Further frames pass through untouched."""
150
+
151
+ self._close()
152
+
153
+ def push_render_frame(self, frame: rtc.AudioFrame) -> None:
154
+ """Supply far-end audio (what the agent is saying) as the AEC reference.
155
+
156
+ Call this for every frame the agent publishes, as close to playout as
157
+ possible. Without it, noise suppression still works but echo
158
+ cancellation has nothing to subtract.
159
+ """
160
+
161
+ if self._closed or not self._opts.echo_cancellation:
162
+ return
163
+
164
+ try:
165
+ with self._lock:
166
+ # Re-check under the lock; `_close` may have run since.
167
+ if self._closed:
168
+ return
169
+ if self._apm is None:
170
+ self._stash_render(frame)
171
+ return
172
+ self._feed_render(frame)
173
+ except Exception:
174
+ self._log_once(
175
+ "render", "failed to push render frame, echo cancellation may degrade"
176
+ )
177
+
178
+ def _process(self, frame: rtc.AudioFrame) -> rtc.AudioFrame:
179
+ if not self._enabled or self._closed:
180
+ return frame
181
+ if (frame.sample_rate, frame.num_channels) == self._degraded:
182
+ return frame
183
+
184
+ try:
185
+ with self._lock:
186
+ if self._closed:
187
+ return frame
188
+ return self._process_capture(frame)
189
+ except Exception:
190
+ self._log_once(
191
+ "capture", "audio processing failed, passing frames through untouched"
192
+ )
193
+ return frame
194
+
195
+ def _close(self) -> None:
196
+ with self._lock:
197
+ self._closed = True
198
+ self._apm = None
199
+ if self._neural is not None:
200
+ self._neural.close()
201
+ self._neural = None
202
+ self._capture_in.clear()
203
+ self._capture_out.clear()
204
+ self._render_in.clear()
205
+ self._render_resampler = None
206
+ self._pending_render.clear()
207
+ self._pending_render_ms = 0.0
208
+
209
+ def _log_once(self, key: str, message: str) -> None:
210
+ """Report a recurring failure once.
211
+
212
+ These fire from the per-frame path, so an unfiltered `logger.exception`
213
+ would emit a traceback every 10 ms for the rest of the call.
214
+ """
215
+
216
+ if key in self._logged:
217
+ return
218
+ self._logged.add(key)
219
+ logger.exception("%s (further occurrences suppressed)", message)
220
+
221
+ def _process_capture(self, frame: rtc.AudioFrame) -> rtc.AudioFrame:
222
+ if not self._ensure_apm(frame.sample_rate, frame.num_channels):
223
+ return frame
224
+
225
+ wanted = frame.samples_per_channel * frame.num_channels
226
+ self._capture_in.append(self._samples(frame))
227
+
228
+ stride = self._block * self._channels
229
+ try:
230
+ while self._capture_in.size >= stride:
231
+ self._capture_out.append(
232
+ self._process_block(self._capture_in.take(stride))
233
+ )
234
+ except Exception:
235
+ # Whatever is queued is now out of step with the filter state. Left
236
+ # in place it is never drained and shows up as latency that grows
237
+ # for the rest of the call, so start the stream over instead.
238
+ self._reset_stream_buffers()
239
+ raise
240
+
241
+ processed = self._capture_out.take(wanted)
242
+ if processed.size < wanted:
243
+ processed = np.concatenate(
244
+ (processed, np.zeros(wanted - processed.size, dtype=np.int16))
245
+ )
246
+ return rtc.AudioFrame(
247
+ processed.tobytes(),
248
+ frame.sample_rate,
249
+ frame.num_channels,
250
+ frame.samples_per_channel,
251
+ )
252
+
253
+ @staticmethod
254
+ def _samples(frame: rtc.AudioFrame) -> np.ndarray:
255
+ """Declared samples of a frame, ignoring any slack in its backing buffer."""
256
+
257
+ pcm = np.frombuffer(frame.data, dtype=np.int16)
258
+ wanted = frame.samples_per_channel * frame.num_channels
259
+ if pcm.size < wanted:
260
+ raise ValueError(f"frame declares {wanted} samples but carries {pcm.size}")
261
+ # A buffer longer than the declared frame would otherwise queue up as
262
+ # latency that never drains.
263
+ return pcm[:wanted]
264
+
265
+ def _process_block(self, block: np.ndarray) -> np.ndarray:
266
+ assert self._apm is not None
267
+
268
+ frame = rtc.AudioFrame(block.tobytes(), self._rate, self._channels, self._block)
269
+ self._apm.process_stream(frame)
270
+ out = np.frombuffer(frame.data, dtype=np.int16).copy()
271
+
272
+ if self._use_neural and self._neural is not None:
273
+ out = self._neural.process(out, self._rate)
274
+
275
+ return out
276
+
277
+ def _resolve_ns_backend(self) -> None:
278
+ """Pick neural or WebRTC NS. Never enable both."""
279
+
280
+ self._use_neural = False
281
+ self._webrtc_ns = False
282
+ if self._neural is not None:
283
+ self._neural.close()
284
+ self._neural = None
285
+
286
+ if not self._opts.noise_suppression:
287
+ return
288
+
289
+ if self._opts.enhancer == "deepfilter":
290
+ try:
291
+ self._neural = NeuralEnhancer()
292
+ self._use_neural = True
293
+ return
294
+ except Exception:
295
+ # neural._get_model logs the first failure with a traceback.
296
+ logger.warning(
297
+ "DeepFilterNet3 unavailable, using WebRTC noise suppression"
298
+ )
299
+ self._webrtc_ns = True
300
+ return
301
+
302
+ self._webrtc_ns = True
303
+
304
+ def _ensure_apm(self, sample_rate: int, num_channels: int) -> bool:
305
+ if (
306
+ self._apm is not None
307
+ and sample_rate == self._rate
308
+ and num_channels == self._channels
309
+ ):
310
+ return True
311
+
312
+ if not self._supported(sample_rate, num_channels):
313
+ return False
314
+
315
+ self._rate = sample_rate
316
+ self._channels = num_channels
317
+ self._block = sample_rate // _RATE_QUANTUM
318
+ self._degraded = None
319
+
320
+ self._resolve_ns_backend()
321
+
322
+ self._apm = rtc.AudioProcessingModule(
323
+ echo_cancellation=self._opts.echo_cancellation,
324
+ noise_suppression=self._webrtc_ns,
325
+ high_pass_filter=self._opts.high_pass_filter,
326
+ auto_gain_control=self._opts.auto_gain_control,
327
+ )
328
+ if self._opts.echo_cancellation:
329
+ # Stored state, not a per-frame argument, and each call is a full
330
+ # FFI round trip -- so set it here rather than every 10 ms.
331
+ self._apm.set_stream_delay_ms(self._opts.stream_delay_ms)
332
+
333
+ self._reset_stream_buffers()
334
+ self._render_resampler = None
335
+ self._render_resampler_rate = 0
336
+
337
+ logger.info(
338
+ "audio filter ready: %d Hz, %d ch, aec=%s ns=%s enhancer=%s agc=%s hpf=%s",
339
+ sample_rate,
340
+ num_channels,
341
+ self._opts.echo_cancellation,
342
+ self._opts.noise_suppression,
343
+ self.enhancer or "off",
344
+ self._opts.auto_gain_control,
345
+ self._opts.high_pass_filter,
346
+ )
347
+
348
+ self._replay_render()
349
+ return True
350
+
351
+ def _supported(self, sample_rate: int, num_channels: int) -> bool:
352
+ """Check the stream format, and step aside for the rest of the call if not.
353
+
354
+ Degrading beats raising: an unsupported format is a property of the call,
355
+ so raising would log a traceback for every frame until it ends.
356
+ """
357
+
358
+ reason = None
359
+ if num_channels != 1:
360
+ reason = f"{num_channels} channels; telephony denoising is mono only"
361
+ elif sample_rate % _RATE_QUANTUM:
362
+ reason = (
363
+ f"{sample_rate} Hz, which is not a whole number of samples per "
364
+ f"{_BLOCK_MS} ms block"
365
+ )
366
+
367
+ if reason is None:
368
+ return True
369
+
370
+ if self._degraded != (sample_rate, num_channels):
371
+ self._degraded = (sample_rate, num_channels)
372
+ logger.error("cannot filter %s; passing audio through unfiltered", reason)
373
+ return False
374
+
375
+ def _reset_stream_buffers(self) -> None:
376
+ self._capture_in.clear()
377
+ self._capture_out.clear()
378
+ self._render_in.clear()
379
+
380
+ # Prime the output with one block of silence so that every inbound frame
381
+ # can be answered with an equal number of samples. Costs 10 ms of
382
+ # added latency and keeps the stream sample-accurate. Neural cold-start
383
+ # pads inside NeuralEnhancer so we do not need a larger prime here.
384
+ self._capture_out.append(np.zeros(self._block * self._channels, dtype=np.int16))
385
+
386
+ def _stash_render(self, frame: rtc.AudioFrame) -> None:
387
+ """Hold far-end audio that arrived before the format was known.
388
+
389
+ The agent usually speaks first, so without this the whole greeting is
390
+ missing from the echo reference and its echo cannot be subtracted.
391
+ """
392
+
393
+ self._pending_render.append(frame)
394
+ self._pending_render_ms += frame.duration * 1000.0
395
+ while self._pending_render_ms > _MAX_PENDING_RENDER_MS and self._pending_render:
396
+ self._pending_render_ms -= self._pending_render.pop(0).duration * 1000.0
397
+
398
+ def _replay_render(self) -> None:
399
+ pending, self._pending_render = self._pending_render, []
400
+ self._pending_render_ms = 0.0
401
+ if not pending or not self._opts.echo_cancellation:
402
+ return
403
+ try:
404
+ for frame in pending:
405
+ self._feed_render(frame)
406
+ except Exception:
407
+ self._log_once(
408
+ "render", "failed to replay far-end audio into the echo canceller"
409
+ )
410
+
411
+ def _feed_render(self, frame: rtc.AudioFrame) -> None:
412
+ assert self._apm is not None
413
+
414
+ pcm = to_mono(self._samples(frame), frame.num_channels)
415
+
416
+ if frame.sample_rate != self._rate:
417
+ pcm = self._resample_render(pcm, frame.sample_rate)
418
+
419
+ self._render_in.append(pcm)
420
+
421
+ stride = self._block * self._channels
422
+ while self._render_in.size >= stride:
423
+ chunk = self._render_in.take(stride)
424
+ self._apm.process_reverse_stream(
425
+ rtc.AudioFrame(chunk.tobytes(), self._rate, self._channels, self._block)
426
+ )
427
+
428
+ def _resample_render(self, pcm: np.ndarray, src_rate: int) -> np.ndarray:
429
+ chunks: list[np.ndarray] = []
430
+
431
+ if (
432
+ self._render_resampler is not None
433
+ and self._render_resampler_rate != src_rate
434
+ ):
435
+ # Push out the tail still inside the old filter before dropping it,
436
+ # otherwise the reference loses samples and the echo path shifts.
437
+ chunks += [
438
+ np.frombuffer(f.data, dtype=np.int16)
439
+ for f in self._render_resampler.flush()
440
+ ]
441
+ self._render_resampler = None
442
+
443
+ if self._render_resampler is None:
444
+ self._render_resampler = rtc.AudioResampler(
445
+ src_rate, self._rate, num_channels=self._channels
446
+ )
447
+ self._render_resampler_rate = src_rate
448
+
449
+ chunks += [
450
+ np.frombuffer(f.data, dtype=np.int16)
451
+ for f in self._render_resampler.push(bytearray(pcm.tobytes()))
452
+ ]
453
+ if not chunks:
454
+ return np.zeros(0, dtype=np.int16)
455
+ return np.concatenate(chunks)
File without changes
@@ -0,0 +1 @@
1
+ __version__ = "0.1.0"