livekit-plugins-denoise 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- livekit/plugins/telephony_denoise/__init__.py +32 -0
- livekit/plugins/telephony_denoise/_buffers.py +53 -0
- livekit/plugins/telephony_denoise/_resample.py +257 -0
- livekit/plugins/telephony_denoise/echo_reference.py +159 -0
- livekit/plugins/telephony_denoise/log.py +3 -0
- livekit/plugins/telephony_denoise/neural.py +222 -0
- livekit/plugins/telephony_denoise/processor.py +455 -0
- livekit/plugins/telephony_denoise/py.typed +0 -0
- livekit/plugins/telephony_denoise/version.py +1 -0
- livekit_plugins_denoise-0.1.0.dist-info/METADATA +136 -0
- livekit_plugins_denoise-0.1.0.dist-info/RECORD +14 -0
- livekit_plugins_denoise-0.1.0.dist-info/WHEEL +4 -0
- livekit_plugins_denoise-0.1.0.dist-info/licenses/LICENSE +21 -0
- livekit_plugins_denoise-0.1.0.dist-info/licenses/NOTICE +11 -0
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
"""DeepFilterNet3 noise suppression for unknown telephony environments."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import threading
|
|
6
|
+
import time
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
|
|
11
|
+
from ._resample import StreamResampler
|
|
12
|
+
from .log import logger
|
|
13
|
+
|
|
14
|
+
_PRIME_LIMIT = 64 # pre-roll blocks before we give up and warn
|
|
15
|
+
|
|
16
|
+
# How long to sit out after a failed load. Retrying per call would stall every
|
|
17
|
+
# call on the same slow failure; never retrying would let one flaky download at
|
|
18
|
+
# worker start disable neural suppression for the life of the process.
|
|
19
|
+
_RETRY_AFTER_SECONDS = 60.0
|
|
20
|
+
|
|
21
|
+
_model_lock = threading.Lock()
|
|
22
|
+
_shared_model: Any | None = None
|
|
23
|
+
_failed_at = 0.0
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _get_model() -> Any:
|
|
27
|
+
"""Load one ONNX session per process. Thread-safe to share; streams are not."""
|
|
28
|
+
|
|
29
|
+
global _shared_model, _failed_at
|
|
30
|
+
|
|
31
|
+
# Loading downloads the weights on first use, so hold the lock across the
|
|
32
|
+
# whole thing: concurrent calls should wait, not each start a download.
|
|
33
|
+
with _model_lock:
|
|
34
|
+
if _shared_model is not None:
|
|
35
|
+
return _shared_model
|
|
36
|
+
|
|
37
|
+
waited = time.monotonic() - _failed_at
|
|
38
|
+
if _failed_at and waited < _RETRY_AFTER_SECONDS:
|
|
39
|
+
raise RuntimeError(
|
|
40
|
+
f"DeepFilterNet3 unavailable, retrying in "
|
|
41
|
+
f"{_RETRY_AFTER_SECONDS - waited:.0f}s"
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
try:
|
|
45
|
+
from deepfilter_stream import DeepFilterModel
|
|
46
|
+
|
|
47
|
+
# One ONNX thread per session so concurrent SIP calls do not thrash.
|
|
48
|
+
_shared_model = DeepFilterModel(intra_op_num_threads=1)
|
|
49
|
+
except Exception:
|
|
50
|
+
_failed_at = time.monotonic()
|
|
51
|
+
logger.exception("DeepFilterNet3 failed to load; falling back to WebRTC NS")
|
|
52
|
+
raise
|
|
53
|
+
|
|
54
|
+
_failed_at = 0.0
|
|
55
|
+
logger.info(
|
|
56
|
+
"DeepFilterNet3 model loaded: %d Hz, %d-sample frames",
|
|
57
|
+
_shared_model.sample_rate,
|
|
58
|
+
_shared_model.frame_size,
|
|
59
|
+
)
|
|
60
|
+
return _shared_model
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def prewarm() -> None:
|
|
64
|
+
"""Download and load the model up front.
|
|
65
|
+
|
|
66
|
+
The first call otherwise pays for a network fetch and an ONNX session build
|
|
67
|
+
inside its first audio frame, on the event loop. Call this from a worker's
|
|
68
|
+
setup hook.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
_get_model()
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class NeuralEnhancer:
|
|
75
|
+
"""Per-call DeepFilterNet stream. Always processes at 48 kHz internally.
|
|
76
|
+
|
|
77
|
+
Feeding 8/16 kHz chunks through `process(sr=...)` builds a large resample
|
|
78
|
+
buffer and adds hundreds of milliseconds of delay, so we resample ourselves
|
|
79
|
+
and call `process_frame` on whole model frames.
|
|
80
|
+
|
|
81
|
+
Delay across this stage is the model's own 32 ms lookahead plus a few ms of
|
|
82
|
+
resampler group delay: measured with the model running, roughly 43 ms at
|
|
83
|
+
48 kHz, 48 ms at 16 kHz and 53 ms at 8 kHz. The model dominates, which is
|
|
84
|
+
the point of `_resample`; a general-purpose resampler put 8 kHz at 172 ms.
|
|
85
|
+
The delay is constant for the life of the stream -- see
|
|
86
|
+
tests/test_neural_plumbing.py, which pins the resampler part with the model
|
|
87
|
+
stubbed out.
|
|
88
|
+
|
|
89
|
+
One stream per call. Not thread-safe, and closing it is final.
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
def __init__(self) -> None:
|
|
93
|
+
model = _get_model()
|
|
94
|
+
try:
|
|
95
|
+
self._stream = model.new_stream()
|
|
96
|
+
except Exception:
|
|
97
|
+
logger.exception("DeepFilterNet3 stream failed; falling back to WebRTC NS")
|
|
98
|
+
raise
|
|
99
|
+
|
|
100
|
+
# Read the geometry off the loaded graph. Hardcoding 512 meant a model
|
|
101
|
+
# asset with a different hop raised on every single frame.
|
|
102
|
+
self._model_rate = int(model.sample_rate)
|
|
103
|
+
self._frame = int(model.frame_size)
|
|
104
|
+
|
|
105
|
+
self._closed = False
|
|
106
|
+
self._rate = 0
|
|
107
|
+
self._up_rs: StreamResampler | None = None
|
|
108
|
+
self._down_rs: StreamResampler | None = None
|
|
109
|
+
self._up = np.zeros(0, dtype=np.float32)
|
|
110
|
+
self._down = np.zeros(0, dtype=np.float32)
|
|
111
|
+
|
|
112
|
+
def _configure(self, sample_rate: int) -> None:
|
|
113
|
+
"""Rebuild the resamplers and pre-roll the output buffer for a new rate."""
|
|
114
|
+
|
|
115
|
+
self._rate = sample_rate
|
|
116
|
+
self._stream.reset()
|
|
117
|
+
self._up = np.zeros(0, dtype=np.float32)
|
|
118
|
+
self._down = np.zeros(0, dtype=np.float32)
|
|
119
|
+
|
|
120
|
+
# These carry their filter state across blocks. Resampling each block
|
|
121
|
+
# independently restarts that state every time, and the transient at
|
|
122
|
+
# each block edge is loud enough to hear as distortion. At the model's
|
|
123
|
+
# own rate they are pass-throughs and cost nothing.
|
|
124
|
+
self._up_rs = StreamResampler(sample_rate, self._model_rate)
|
|
125
|
+
self._down_rs = StreamResampler(self._model_rate, sample_rate)
|
|
126
|
+
|
|
127
|
+
# Output arrives in whole model frames and trails input by the
|
|
128
|
+
# group delay of both resamplers, so the buffer starts empty and stays
|
|
129
|
+
# that way for a while. Left alone it runs dry mid-call and we splice in
|
|
130
|
+
# silence, which both clicks and pushes the audio permanently later --
|
|
131
|
+
# that is how latency crept past 300 ms at 8 kHz. Push silence through
|
|
132
|
+
# the whole chain now so the delay is flushed before real audio arrives,
|
|
133
|
+
# and keep one frame in hand to absorb the frame-boundary jitter.
|
|
134
|
+
quantum = self.quantum
|
|
135
|
+
for _ in range(_PRIME_LIMIT):
|
|
136
|
+
if self._down.size >= quantum:
|
|
137
|
+
break
|
|
138
|
+
self._pump(np.zeros(quantum, dtype=np.float32))
|
|
139
|
+
else:
|
|
140
|
+
logger.warning(
|
|
141
|
+
"DeepFilterNet pre-roll did not fill at %d Hz; expect a brief warm-up",
|
|
142
|
+
sample_rate,
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
@property
|
|
146
|
+
def quantum(self) -> int:
|
|
147
|
+
"""One model frame expressed in caller-rate samples.
|
|
148
|
+
|
|
149
|
+
The output buffer can only ever fall this far behind the input, whatever
|
|
150
|
+
block size the caller uses, so it is also the pre-roll target.
|
|
151
|
+
"""
|
|
152
|
+
|
|
153
|
+
return int(np.ceil(self._frame * self._rate / self._model_rate))
|
|
154
|
+
|
|
155
|
+
def _pump(self, mono: np.ndarray) -> None:
|
|
156
|
+
"""Run one block of float32 audio through resample -> model -> resample."""
|
|
157
|
+
|
|
158
|
+
assert self._up_rs is not None and self._down_rs is not None
|
|
159
|
+
|
|
160
|
+
up = self._up_rs.process(mono)
|
|
161
|
+
if up.size:
|
|
162
|
+
self._up = np.concatenate((self._up, up))
|
|
163
|
+
|
|
164
|
+
enhanced: list[np.ndarray] = []
|
|
165
|
+
while self._up.size >= self._frame:
|
|
166
|
+
frame = self._up[: self._frame]
|
|
167
|
+
self._up = self._up[self._frame :]
|
|
168
|
+
enhanced.append(self._stream.process_frame(frame))
|
|
169
|
+
|
|
170
|
+
if not enhanced:
|
|
171
|
+
return
|
|
172
|
+
|
|
173
|
+
chunk = np.concatenate(enhanced).astype(np.float32)
|
|
174
|
+
if not np.isfinite(chunk).all():
|
|
175
|
+
# The resampler keeps a window of history, so a single NaN out of the
|
|
176
|
+
# model would smear across every later block instead of one frame.
|
|
177
|
+
logger.warning("DeepFilterNet3 returned non-finite samples; zeroing them")
|
|
178
|
+
chunk = np.nan_to_num(chunk, nan=0.0, posinf=0.0, neginf=0.0)
|
|
179
|
+
|
|
180
|
+
down = self._down_rs.process(chunk)
|
|
181
|
+
if down.size:
|
|
182
|
+
self._down = np.concatenate((self._down, down))
|
|
183
|
+
|
|
184
|
+
def process(self, pcm: np.ndarray, sample_rate: int) -> np.ndarray:
|
|
185
|
+
"""Enhance a mono int16 block. Returns the same number of int16 samples."""
|
|
186
|
+
|
|
187
|
+
if self._closed:
|
|
188
|
+
raise RuntimeError("NeuralEnhancer is closed")
|
|
189
|
+
if sample_rate != self._rate:
|
|
190
|
+
self._configure(sample_rate)
|
|
191
|
+
|
|
192
|
+
wanted = int(pcm.size)
|
|
193
|
+
self._pump(pcm.astype(np.float32) * (1.0 / 32768.0))
|
|
194
|
+
|
|
195
|
+
if self._down.size >= wanted:
|
|
196
|
+
out = self._down[:wanted]
|
|
197
|
+
self._down = self._down[wanted:]
|
|
198
|
+
else:
|
|
199
|
+
# The pre-roll is sized so this cannot happen in steady state. Reaching
|
|
200
|
+
# it means the chain lost samples, so splice silence to keep the frame
|
|
201
|
+
# contract rather than hand back a short block.
|
|
202
|
+
pad = wanted - int(self._down.size)
|
|
203
|
+
logger.warning("DeepFilterNet output ran dry, padding %d samples", pad)
|
|
204
|
+
out = np.concatenate((self._down, np.zeros(pad, dtype=np.float32)))
|
|
205
|
+
self._down = np.zeros(0, dtype=np.float32)
|
|
206
|
+
|
|
207
|
+
return np.clip(np.rint(out * 32768.0), -32768, 32767).astype(np.int16)
|
|
208
|
+
|
|
209
|
+
def close(self) -> None:
|
|
210
|
+
"""Release the stream. The enhancer cannot be used again afterwards."""
|
|
211
|
+
|
|
212
|
+
if self._closed:
|
|
213
|
+
return
|
|
214
|
+
self._closed = True
|
|
215
|
+
self._rate = 0
|
|
216
|
+
self._up_rs = self._down_rs = None
|
|
217
|
+
self._up = np.zeros(0, dtype=np.float32)
|
|
218
|
+
self._down = np.zeros(0, dtype=np.float32)
|
|
219
|
+
try:
|
|
220
|
+
self._stream.reset()
|
|
221
|
+
except Exception:
|
|
222
|
+
logger.debug("DeepFilterNet stream reset failed on close", exc_info=True)
|
|
@@ -0,0 +1,455 @@
|
|
|
1
|
+
"""LiveKit frame processor for runtime telephony noise and echo suppression."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import threading
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from typing import Literal, get_args
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
from livekit import rtc
|
|
11
|
+
|
|
12
|
+
from ._buffers import Int16Buffer, to_mono
|
|
13
|
+
from .log import logger
|
|
14
|
+
from .neural import NeuralEnhancer
|
|
15
|
+
|
|
16
|
+
_BLOCK_MS = 10
|
|
17
|
+
"""The WebRTC APM contract: audio must be handed over in 10 ms blocks."""
|
|
18
|
+
|
|
19
|
+
_RATE_QUANTUM = 1000 // _BLOCK_MS
|
|
20
|
+
"""A supported rate must divide by this, so a block is a whole number of samples."""
|
|
21
|
+
|
|
22
|
+
_MAX_PENDING_RENDER_MS = 2000.0
|
|
23
|
+
"""How much far-end audio to hold before the first inbound frame sets the format."""
|
|
24
|
+
|
|
25
|
+
Enhancer = Literal["deepfilter", "webrtc"]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class DenoiseOptions:
|
|
30
|
+
"""Tuning for `TelephonyDenoiser`."""
|
|
31
|
+
|
|
32
|
+
echo_cancellation: bool = True
|
|
33
|
+
noise_suppression: bool = True
|
|
34
|
+
high_pass_filter: bool = True
|
|
35
|
+
auto_gain_control: bool = True
|
|
36
|
+
|
|
37
|
+
enhancer: Enhancer = "deepfilter"
|
|
38
|
+
"""Noise suppressor for unknown caller environments.
|
|
39
|
+
|
|
40
|
+
`deepfilter` (default) runs DeepFilterNet3 after AEC and handles gym / café /
|
|
41
|
+
babble as well as line hiss. `webrtc` keeps the classical WebRTC NS for
|
|
42
|
+
minimum latency. Never stack both.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
stream_delay_ms: int = 120
|
|
46
|
+
"""Round trip delay between us sending audio and its echo arriving back.
|
|
47
|
+
|
|
48
|
+
AEC3 refines this internally, but a sane starting point matters on SIP where
|
|
49
|
+
the PSTN leg adds far more delay than a WebRTC client would.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
def __post_init__(self) -> None:
|
|
53
|
+
if self.enhancer not in get_args(Enhancer):
|
|
54
|
+
raise ValueError(
|
|
55
|
+
f"enhancer must be one of {get_args(Enhancer)}, got {self.enhancer!r}"
|
|
56
|
+
)
|
|
57
|
+
if self.stream_delay_ms < 0:
|
|
58
|
+
raise ValueError(
|
|
59
|
+
f"stream_delay_ms must not be negative, got {self.stream_delay_ms}"
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class TelephonyDenoiser(rtc.FrameProcessor[rtc.AudioFrame]):
|
|
64
|
+
"""Suppresses background noise and acoustic echo on an inbound audio track.
|
|
65
|
+
|
|
66
|
+
The inbound (caller) audio flows through `process`, which LiveKit calls for
|
|
67
|
+
every frame. For echo cancellation the filter also needs the far-end
|
|
68
|
+
reference, meaning the audio the agent sends to the caller; feed that in via
|
|
69
|
+
`push_render_frame`, or let `EchoReferenceTap` do it for you.
|
|
70
|
+
|
|
71
|
+
One instance holds the echo path and noise estimate for a single
|
|
72
|
+
conversation. Build a new one per call; never share across calls.
|
|
73
|
+
"""
|
|
74
|
+
|
|
75
|
+
def __init__(self, opts: DenoiseOptions | None = None) -> None:
|
|
76
|
+
self._opts = opts or DenoiseOptions()
|
|
77
|
+
self._enabled = True
|
|
78
|
+
self._closed = False
|
|
79
|
+
|
|
80
|
+
# The format we gave up on, if any. Held rather than a bare flag so a
|
|
81
|
+
# renegotiation to a format we do support starts filtering again.
|
|
82
|
+
self._degraded: tuple[int, int] | None = None
|
|
83
|
+
self._logged: set[str] = set()
|
|
84
|
+
|
|
85
|
+
# `process` runs on the room's event loop while `push_render_frame` is
|
|
86
|
+
# driven by the TTS output path, which may be a different thread.
|
|
87
|
+
self._lock = threading.Lock()
|
|
88
|
+
|
|
89
|
+
self._apm: rtc.AudioProcessingModule | None = None
|
|
90
|
+
self._rate = 0
|
|
91
|
+
self._channels = 0
|
|
92
|
+
self._block = 0
|
|
93
|
+
|
|
94
|
+
self._capture_in = Int16Buffer()
|
|
95
|
+
self._capture_out = Int16Buffer()
|
|
96
|
+
|
|
97
|
+
self._render_in = Int16Buffer()
|
|
98
|
+
self._render_resampler: rtc.AudioResampler | None = None
|
|
99
|
+
self._render_resampler_rate = 0
|
|
100
|
+
self._pending_render: list[rtc.AudioFrame] = []
|
|
101
|
+
self._pending_render_ms = 0.0
|
|
102
|
+
|
|
103
|
+
self._neural: NeuralEnhancer | None = None
|
|
104
|
+
self._use_neural = False
|
|
105
|
+
self._webrtc_ns = False
|
|
106
|
+
|
|
107
|
+
@property
|
|
108
|
+
def enabled(self) -> bool:
|
|
109
|
+
return self._enabled
|
|
110
|
+
|
|
111
|
+
@enabled.setter
|
|
112
|
+
def enabled(self, value: bool) -> None:
|
|
113
|
+
self._enabled = value
|
|
114
|
+
|
|
115
|
+
@property
|
|
116
|
+
def enhancer(self) -> Enhancer | None:
|
|
117
|
+
"""Active noise suppressor, or None if it is off or not yet negotiated.
|
|
118
|
+
|
|
119
|
+
The backend is only known once the first frame has revealed the stream
|
|
120
|
+
format, because that is when a DeepFilterNet failure would fall back.
|
|
121
|
+
"""
|
|
122
|
+
|
|
123
|
+
if not self._opts.noise_suppression or self._apm is None:
|
|
124
|
+
return None
|
|
125
|
+
return "deepfilter" if self._use_neural else "webrtc"
|
|
126
|
+
|
|
127
|
+
def prepare(self, sample_rate: int, num_channels: int = 1) -> bool:
|
|
128
|
+
"""Build the filter ahead of the first frame.
|
|
129
|
+
|
|
130
|
+
Optional: the format is picked up from the first inbound frame anyway.
|
|
131
|
+
Doing it up front keeps the model load off the first frame's deadline and
|
|
132
|
+
lets far-end audio be used as a reference from the very first word.
|
|
133
|
+
|
|
134
|
+
Returns False if the format is not supported, in which case audio will
|
|
135
|
+
pass through untouched.
|
|
136
|
+
"""
|
|
137
|
+
|
|
138
|
+
with self._lock:
|
|
139
|
+
if self._closed:
|
|
140
|
+
return False
|
|
141
|
+
return self._ensure_apm(sample_rate, num_channels)
|
|
142
|
+
|
|
143
|
+
def process(self, frame: rtc.AudioFrame) -> rtc.AudioFrame:
|
|
144
|
+
"""Filter one inbound frame. Returns a frame of the same shape."""
|
|
145
|
+
|
|
146
|
+
return self._process(frame)
|
|
147
|
+
|
|
148
|
+
def close(self) -> None:
|
|
149
|
+
"""Release the filter. Further frames pass through untouched."""
|
|
150
|
+
|
|
151
|
+
self._close()
|
|
152
|
+
|
|
153
|
+
def push_render_frame(self, frame: rtc.AudioFrame) -> None:
|
|
154
|
+
"""Supply far-end audio (what the agent is saying) as the AEC reference.
|
|
155
|
+
|
|
156
|
+
Call this for every frame the agent publishes, as close to playout as
|
|
157
|
+
possible. Without it, noise suppression still works but echo
|
|
158
|
+
cancellation has nothing to subtract.
|
|
159
|
+
"""
|
|
160
|
+
|
|
161
|
+
if self._closed or not self._opts.echo_cancellation:
|
|
162
|
+
return
|
|
163
|
+
|
|
164
|
+
try:
|
|
165
|
+
with self._lock:
|
|
166
|
+
# Re-check under the lock; `_close` may have run since.
|
|
167
|
+
if self._closed:
|
|
168
|
+
return
|
|
169
|
+
if self._apm is None:
|
|
170
|
+
self._stash_render(frame)
|
|
171
|
+
return
|
|
172
|
+
self._feed_render(frame)
|
|
173
|
+
except Exception:
|
|
174
|
+
self._log_once(
|
|
175
|
+
"render", "failed to push render frame, echo cancellation may degrade"
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
def _process(self, frame: rtc.AudioFrame) -> rtc.AudioFrame:
|
|
179
|
+
if not self._enabled or self._closed:
|
|
180
|
+
return frame
|
|
181
|
+
if (frame.sample_rate, frame.num_channels) == self._degraded:
|
|
182
|
+
return frame
|
|
183
|
+
|
|
184
|
+
try:
|
|
185
|
+
with self._lock:
|
|
186
|
+
if self._closed:
|
|
187
|
+
return frame
|
|
188
|
+
return self._process_capture(frame)
|
|
189
|
+
except Exception:
|
|
190
|
+
self._log_once(
|
|
191
|
+
"capture", "audio processing failed, passing frames through untouched"
|
|
192
|
+
)
|
|
193
|
+
return frame
|
|
194
|
+
|
|
195
|
+
def _close(self) -> None:
|
|
196
|
+
with self._lock:
|
|
197
|
+
self._closed = True
|
|
198
|
+
self._apm = None
|
|
199
|
+
if self._neural is not None:
|
|
200
|
+
self._neural.close()
|
|
201
|
+
self._neural = None
|
|
202
|
+
self._capture_in.clear()
|
|
203
|
+
self._capture_out.clear()
|
|
204
|
+
self._render_in.clear()
|
|
205
|
+
self._render_resampler = None
|
|
206
|
+
self._pending_render.clear()
|
|
207
|
+
self._pending_render_ms = 0.0
|
|
208
|
+
|
|
209
|
+
def _log_once(self, key: str, message: str) -> None:
|
|
210
|
+
"""Report a recurring failure once.
|
|
211
|
+
|
|
212
|
+
These fire from the per-frame path, so an unfiltered `logger.exception`
|
|
213
|
+
would emit a traceback every 10 ms for the rest of the call.
|
|
214
|
+
"""
|
|
215
|
+
|
|
216
|
+
if key in self._logged:
|
|
217
|
+
return
|
|
218
|
+
self._logged.add(key)
|
|
219
|
+
logger.exception("%s (further occurrences suppressed)", message)
|
|
220
|
+
|
|
221
|
+
def _process_capture(self, frame: rtc.AudioFrame) -> rtc.AudioFrame:
|
|
222
|
+
if not self._ensure_apm(frame.sample_rate, frame.num_channels):
|
|
223
|
+
return frame
|
|
224
|
+
|
|
225
|
+
wanted = frame.samples_per_channel * frame.num_channels
|
|
226
|
+
self._capture_in.append(self._samples(frame))
|
|
227
|
+
|
|
228
|
+
stride = self._block * self._channels
|
|
229
|
+
try:
|
|
230
|
+
while self._capture_in.size >= stride:
|
|
231
|
+
self._capture_out.append(
|
|
232
|
+
self._process_block(self._capture_in.take(stride))
|
|
233
|
+
)
|
|
234
|
+
except Exception:
|
|
235
|
+
# Whatever is queued is now out of step with the filter state. Left
|
|
236
|
+
# in place it is never drained and shows up as latency that grows
|
|
237
|
+
# for the rest of the call, so start the stream over instead.
|
|
238
|
+
self._reset_stream_buffers()
|
|
239
|
+
raise
|
|
240
|
+
|
|
241
|
+
processed = self._capture_out.take(wanted)
|
|
242
|
+
if processed.size < wanted:
|
|
243
|
+
processed = np.concatenate(
|
|
244
|
+
(processed, np.zeros(wanted - processed.size, dtype=np.int16))
|
|
245
|
+
)
|
|
246
|
+
return rtc.AudioFrame(
|
|
247
|
+
processed.tobytes(),
|
|
248
|
+
frame.sample_rate,
|
|
249
|
+
frame.num_channels,
|
|
250
|
+
frame.samples_per_channel,
|
|
251
|
+
)
|
|
252
|
+
|
|
253
|
+
@staticmethod
|
|
254
|
+
def _samples(frame: rtc.AudioFrame) -> np.ndarray:
|
|
255
|
+
"""Declared samples of a frame, ignoring any slack in its backing buffer."""
|
|
256
|
+
|
|
257
|
+
pcm = np.frombuffer(frame.data, dtype=np.int16)
|
|
258
|
+
wanted = frame.samples_per_channel * frame.num_channels
|
|
259
|
+
if pcm.size < wanted:
|
|
260
|
+
raise ValueError(f"frame declares {wanted} samples but carries {pcm.size}")
|
|
261
|
+
# A buffer longer than the declared frame would otherwise queue up as
|
|
262
|
+
# latency that never drains.
|
|
263
|
+
return pcm[:wanted]
|
|
264
|
+
|
|
265
|
+
def _process_block(self, block: np.ndarray) -> np.ndarray:
|
|
266
|
+
assert self._apm is not None
|
|
267
|
+
|
|
268
|
+
frame = rtc.AudioFrame(block.tobytes(), self._rate, self._channels, self._block)
|
|
269
|
+
self._apm.process_stream(frame)
|
|
270
|
+
out = np.frombuffer(frame.data, dtype=np.int16).copy()
|
|
271
|
+
|
|
272
|
+
if self._use_neural and self._neural is not None:
|
|
273
|
+
out = self._neural.process(out, self._rate)
|
|
274
|
+
|
|
275
|
+
return out
|
|
276
|
+
|
|
277
|
+
def _resolve_ns_backend(self) -> None:
|
|
278
|
+
"""Pick neural or WebRTC NS. Never enable both."""
|
|
279
|
+
|
|
280
|
+
self._use_neural = False
|
|
281
|
+
self._webrtc_ns = False
|
|
282
|
+
if self._neural is not None:
|
|
283
|
+
self._neural.close()
|
|
284
|
+
self._neural = None
|
|
285
|
+
|
|
286
|
+
if not self._opts.noise_suppression:
|
|
287
|
+
return
|
|
288
|
+
|
|
289
|
+
if self._opts.enhancer == "deepfilter":
|
|
290
|
+
try:
|
|
291
|
+
self._neural = NeuralEnhancer()
|
|
292
|
+
self._use_neural = True
|
|
293
|
+
return
|
|
294
|
+
except Exception:
|
|
295
|
+
# neural._get_model logs the first failure with a traceback.
|
|
296
|
+
logger.warning(
|
|
297
|
+
"DeepFilterNet3 unavailable, using WebRTC noise suppression"
|
|
298
|
+
)
|
|
299
|
+
self._webrtc_ns = True
|
|
300
|
+
return
|
|
301
|
+
|
|
302
|
+
self._webrtc_ns = True
|
|
303
|
+
|
|
304
|
+
def _ensure_apm(self, sample_rate: int, num_channels: int) -> bool:
|
|
305
|
+
if (
|
|
306
|
+
self._apm is not None
|
|
307
|
+
and sample_rate == self._rate
|
|
308
|
+
and num_channels == self._channels
|
|
309
|
+
):
|
|
310
|
+
return True
|
|
311
|
+
|
|
312
|
+
if not self._supported(sample_rate, num_channels):
|
|
313
|
+
return False
|
|
314
|
+
|
|
315
|
+
self._rate = sample_rate
|
|
316
|
+
self._channels = num_channels
|
|
317
|
+
self._block = sample_rate // _RATE_QUANTUM
|
|
318
|
+
self._degraded = None
|
|
319
|
+
|
|
320
|
+
self._resolve_ns_backend()
|
|
321
|
+
|
|
322
|
+
self._apm = rtc.AudioProcessingModule(
|
|
323
|
+
echo_cancellation=self._opts.echo_cancellation,
|
|
324
|
+
noise_suppression=self._webrtc_ns,
|
|
325
|
+
high_pass_filter=self._opts.high_pass_filter,
|
|
326
|
+
auto_gain_control=self._opts.auto_gain_control,
|
|
327
|
+
)
|
|
328
|
+
if self._opts.echo_cancellation:
|
|
329
|
+
# Stored state, not a per-frame argument, and each call is a full
|
|
330
|
+
# FFI round trip -- so set it here rather than every 10 ms.
|
|
331
|
+
self._apm.set_stream_delay_ms(self._opts.stream_delay_ms)
|
|
332
|
+
|
|
333
|
+
self._reset_stream_buffers()
|
|
334
|
+
self._render_resampler = None
|
|
335
|
+
self._render_resampler_rate = 0
|
|
336
|
+
|
|
337
|
+
logger.info(
|
|
338
|
+
"audio filter ready: %d Hz, %d ch, aec=%s ns=%s enhancer=%s agc=%s hpf=%s",
|
|
339
|
+
sample_rate,
|
|
340
|
+
num_channels,
|
|
341
|
+
self._opts.echo_cancellation,
|
|
342
|
+
self._opts.noise_suppression,
|
|
343
|
+
self.enhancer or "off",
|
|
344
|
+
self._opts.auto_gain_control,
|
|
345
|
+
self._opts.high_pass_filter,
|
|
346
|
+
)
|
|
347
|
+
|
|
348
|
+
self._replay_render()
|
|
349
|
+
return True
|
|
350
|
+
|
|
351
|
+
def _supported(self, sample_rate: int, num_channels: int) -> bool:
|
|
352
|
+
"""Check the stream format, and step aside for the rest of the call if not.
|
|
353
|
+
|
|
354
|
+
Degrading beats raising: an unsupported format is a property of the call,
|
|
355
|
+
so raising would log a traceback for every frame until it ends.
|
|
356
|
+
"""
|
|
357
|
+
|
|
358
|
+
reason = None
|
|
359
|
+
if num_channels != 1:
|
|
360
|
+
reason = f"{num_channels} channels; telephony denoising is mono only"
|
|
361
|
+
elif sample_rate % _RATE_QUANTUM:
|
|
362
|
+
reason = (
|
|
363
|
+
f"{sample_rate} Hz, which is not a whole number of samples per "
|
|
364
|
+
f"{_BLOCK_MS} ms block"
|
|
365
|
+
)
|
|
366
|
+
|
|
367
|
+
if reason is None:
|
|
368
|
+
return True
|
|
369
|
+
|
|
370
|
+
if self._degraded != (sample_rate, num_channels):
|
|
371
|
+
self._degraded = (sample_rate, num_channels)
|
|
372
|
+
logger.error("cannot filter %s; passing audio through unfiltered", reason)
|
|
373
|
+
return False
|
|
374
|
+
|
|
375
|
+
def _reset_stream_buffers(self) -> None:
|
|
376
|
+
self._capture_in.clear()
|
|
377
|
+
self._capture_out.clear()
|
|
378
|
+
self._render_in.clear()
|
|
379
|
+
|
|
380
|
+
# Prime the output with one block of silence so that every inbound frame
|
|
381
|
+
# can be answered with an equal number of samples. Costs 10 ms of
|
|
382
|
+
# added latency and keeps the stream sample-accurate. Neural cold-start
|
|
383
|
+
# pads inside NeuralEnhancer so we do not need a larger prime here.
|
|
384
|
+
self._capture_out.append(np.zeros(self._block * self._channels, dtype=np.int16))
|
|
385
|
+
|
|
386
|
+
def _stash_render(self, frame: rtc.AudioFrame) -> None:
|
|
387
|
+
"""Hold far-end audio that arrived before the format was known.
|
|
388
|
+
|
|
389
|
+
The agent usually speaks first, so without this the whole greeting is
|
|
390
|
+
missing from the echo reference and its echo cannot be subtracted.
|
|
391
|
+
"""
|
|
392
|
+
|
|
393
|
+
self._pending_render.append(frame)
|
|
394
|
+
self._pending_render_ms += frame.duration * 1000.0
|
|
395
|
+
while self._pending_render_ms > _MAX_PENDING_RENDER_MS and self._pending_render:
|
|
396
|
+
self._pending_render_ms -= self._pending_render.pop(0).duration * 1000.0
|
|
397
|
+
|
|
398
|
+
def _replay_render(self) -> None:
|
|
399
|
+
pending, self._pending_render = self._pending_render, []
|
|
400
|
+
self._pending_render_ms = 0.0
|
|
401
|
+
if not pending or not self._opts.echo_cancellation:
|
|
402
|
+
return
|
|
403
|
+
try:
|
|
404
|
+
for frame in pending:
|
|
405
|
+
self._feed_render(frame)
|
|
406
|
+
except Exception:
|
|
407
|
+
self._log_once(
|
|
408
|
+
"render", "failed to replay far-end audio into the echo canceller"
|
|
409
|
+
)
|
|
410
|
+
|
|
411
|
+
def _feed_render(self, frame: rtc.AudioFrame) -> None:
|
|
412
|
+
assert self._apm is not None
|
|
413
|
+
|
|
414
|
+
pcm = to_mono(self._samples(frame), frame.num_channels)
|
|
415
|
+
|
|
416
|
+
if frame.sample_rate != self._rate:
|
|
417
|
+
pcm = self._resample_render(pcm, frame.sample_rate)
|
|
418
|
+
|
|
419
|
+
self._render_in.append(pcm)
|
|
420
|
+
|
|
421
|
+
stride = self._block * self._channels
|
|
422
|
+
while self._render_in.size >= stride:
|
|
423
|
+
chunk = self._render_in.take(stride)
|
|
424
|
+
self._apm.process_reverse_stream(
|
|
425
|
+
rtc.AudioFrame(chunk.tobytes(), self._rate, self._channels, self._block)
|
|
426
|
+
)
|
|
427
|
+
|
|
428
|
+
def _resample_render(self, pcm: np.ndarray, src_rate: int) -> np.ndarray:
|
|
429
|
+
chunks: list[np.ndarray] = []
|
|
430
|
+
|
|
431
|
+
if (
|
|
432
|
+
self._render_resampler is not None
|
|
433
|
+
and self._render_resampler_rate != src_rate
|
|
434
|
+
):
|
|
435
|
+
# Push out the tail still inside the old filter before dropping it,
|
|
436
|
+
# otherwise the reference loses samples and the echo path shifts.
|
|
437
|
+
chunks += [
|
|
438
|
+
np.frombuffer(f.data, dtype=np.int16)
|
|
439
|
+
for f in self._render_resampler.flush()
|
|
440
|
+
]
|
|
441
|
+
self._render_resampler = None
|
|
442
|
+
|
|
443
|
+
if self._render_resampler is None:
|
|
444
|
+
self._render_resampler = rtc.AudioResampler(
|
|
445
|
+
src_rate, self._rate, num_channels=self._channels
|
|
446
|
+
)
|
|
447
|
+
self._render_resampler_rate = src_rate
|
|
448
|
+
|
|
449
|
+
chunks += [
|
|
450
|
+
np.frombuffer(f.data, dtype=np.int16)
|
|
451
|
+
for f in self._render_resampler.push(bytearray(pcm.tobytes()))
|
|
452
|
+
]
|
|
453
|
+
if not chunks:
|
|
454
|
+
return np.zeros(0, dtype=np.int16)
|
|
455
|
+
return np.concatenate(chunks)
|
|
File without changes
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|