agent-voice 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_voice/__init__.py +3 -0
- agent_voice/__main__.py +5 -0
- agent_voice/audio.py +297 -0
- agent_voice/cli.py +492 -0
- agent_voice/client.py +302 -0
- agent_voice/config.py +225 -0
- agent_voice/delivery.py +55 -0
- agent_voice/doctor.py +114 -0
- agent_voice/kokoro.py +332 -0
- agent_voice/media.py +9 -0
- agent_voice/model.py +146 -0
- agent_voice/paths.py +63 -0
- agent_voice/registry.py +102 -0
- agent_voice/service.py +339 -0
- agent_voice/speaking.py +375 -0
- agent_voice/templates/brand-icon.svg +17 -0
- agent_voice/templates/brand-logo.svg +32 -0
- agent_voice/templates/recording.html +122 -0
- agent_voice/viewer.py +287 -0
- agent_voice/viewer_server.py +311 -0
- agent_voice-0.5.0.dist-info/METADATA +200 -0
- agent_voice-0.5.0.dist-info/RECORD +25 -0
- agent_voice-0.5.0.dist-info/WHEEL +4 -0
- agent_voice-0.5.0.dist-info/entry_points.txt +2 -0
- agent_voice-0.5.0.dist-info/licenses/LICENSE +21 -0
agent_voice/__init__.py
ADDED
agent_voice/__main__.py
ADDED
agent_voice/audio.py
ADDED
|
@@ -0,0 +1,297 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import subprocess
|
|
5
|
+
import tempfile
|
|
6
|
+
import threading
|
|
7
|
+
import wave
|
|
8
|
+
from collections.abc import Generator
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
import imageio_ffmpeg
|
|
13
|
+
import miniaudio
|
|
14
|
+
import numpy as np
|
|
15
|
+
from numpy.typing import NDArray
|
|
16
|
+
|
|
17
|
+
from .config import FORMATS, MAX_SPEED, MIN_SPEED
|
|
18
|
+
|
|
19
|
+
PLAYBACK_SAMPLE_RATE = 24_000
|
|
20
|
+
PLAYBACK_CHANNELS = 1
|
|
21
|
+
PLAYBACK_SAMPLE_WIDTH = 2
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True)
|
|
25
|
+
class AudioRuntime:
|
|
26
|
+
ffmpeg_path: str | None
|
|
27
|
+
ffmpeg_version: str | None
|
|
28
|
+
ffmpeg_error: str | None
|
|
29
|
+
miniaudio_version: str
|
|
30
|
+
playback_backend: str | None
|
|
31
|
+
playback_error: str | None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def inspect_audio_runtime() -> AudioRuntime:
|
|
35
|
+
"""Inspect the bundled codec and native playback runtimes."""
|
|
36
|
+
try:
|
|
37
|
+
ffmpeg_path = _ffmpeg_executable()
|
|
38
|
+
ffmpeg_version = imageio_ffmpeg.get_ffmpeg_version()
|
|
39
|
+
ffmpeg_error = None
|
|
40
|
+
except RuntimeError as error:
|
|
41
|
+
ffmpeg_path = None
|
|
42
|
+
ffmpeg_version = None
|
|
43
|
+
ffmpeg_error = str(error)
|
|
44
|
+
|
|
45
|
+
try:
|
|
46
|
+
with miniaudio.PlaybackDevice(
|
|
47
|
+
output_format=miniaudio.SampleFormat.SIGNED16,
|
|
48
|
+
nchannels=PLAYBACK_CHANNELS,
|
|
49
|
+
sample_rate=PLAYBACK_SAMPLE_RATE,
|
|
50
|
+
app_name="Agent Voice",
|
|
51
|
+
) as device:
|
|
52
|
+
playback_backend = device.backend
|
|
53
|
+
playback_error = None
|
|
54
|
+
except miniaudio.MiniaudioError as error:
|
|
55
|
+
playback_backend = None
|
|
56
|
+
playback_error = str(error)
|
|
57
|
+
|
|
58
|
+
return AudioRuntime(
|
|
59
|
+
ffmpeg_path=ffmpeg_path,
|
|
60
|
+
ffmpeg_version=ffmpeg_version,
|
|
61
|
+
ffmpeg_error=ffmpeg_error,
|
|
62
|
+
miniaudio_version=miniaudio.__version__,
|
|
63
|
+
playback_backend=playback_backend,
|
|
64
|
+
playback_error=playback_error,
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def change_tempo(
|
|
69
|
+
samples: NDArray[np.floating], sample_rate: int, factor: float
|
|
70
|
+
) -> NDArray[np.float32]:
|
|
71
|
+
"""Change speech tempo with bundled FFmpeg while preserving pitch."""
|
|
72
|
+
if factor == 1.0:
|
|
73
|
+
return np.asarray(samples, dtype=np.float32).reshape(-1)
|
|
74
|
+
if not MIN_SPEED <= factor <= MAX_SPEED:
|
|
75
|
+
raise ValueError(f"Tempo factor must be between {MIN_SPEED} and {MAX_SPEED}")
|
|
76
|
+
|
|
77
|
+
source = np.asarray(samples, dtype="<f4").reshape(-1)
|
|
78
|
+
tempo_filter = ",".join(
|
|
79
|
+
f"atempo={tempo_factor}" for tempo_factor in _tempo_factors(factor)
|
|
80
|
+
)
|
|
81
|
+
command = [
|
|
82
|
+
_ffmpeg_executable(),
|
|
83
|
+
"-hide_banner",
|
|
84
|
+
"-loglevel",
|
|
85
|
+
"error",
|
|
86
|
+
"-f",
|
|
87
|
+
"f32le",
|
|
88
|
+
"-ar",
|
|
89
|
+
str(sample_rate),
|
|
90
|
+
"-ac",
|
|
91
|
+
"1",
|
|
92
|
+
"-i",
|
|
93
|
+
"pipe:0",
|
|
94
|
+
"-filter:a",
|
|
95
|
+
tempo_filter,
|
|
96
|
+
"-f",
|
|
97
|
+
"f32le",
|
|
98
|
+
"pipe:1",
|
|
99
|
+
]
|
|
100
|
+
try:
|
|
101
|
+
completed = subprocess.run(
|
|
102
|
+
command, input=source.tobytes(), check=True, capture_output=True
|
|
103
|
+
)
|
|
104
|
+
except subprocess.CalledProcessError as error:
|
|
105
|
+
detail = error.stderr.decode(errors="replace").strip()
|
|
106
|
+
raise RuntimeError(f"ffmpeg could not change speech tempo: {detail}") from error
|
|
107
|
+
return np.frombuffer(completed.stdout, dtype="<f4").copy()
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _tempo_factors(factor: float) -> list[float]:
|
|
111
|
+
"""Split aggressive speedups into pitch-preserving FFmpeg tempo stages."""
|
|
112
|
+
factors: list[float] = []
|
|
113
|
+
remaining = float(factor)
|
|
114
|
+
while remaining > 2.0:
|
|
115
|
+
factors.append(2.0)
|
|
116
|
+
remaining /= 2.0
|
|
117
|
+
factors.append(remaining)
|
|
118
|
+
return factors
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def write_audio(
|
|
122
|
+
samples: NDArray[np.floating],
|
|
123
|
+
sample_rate: int,
|
|
124
|
+
destination: Path,
|
|
125
|
+
audio_format: str,
|
|
126
|
+
) -> Path:
|
|
127
|
+
audio_format = audio_format.lower()
|
|
128
|
+
if audio_format not in FORMATS:
|
|
129
|
+
raise ValueError(
|
|
130
|
+
f"Unsupported format '{audio_format}'. Choose: {', '.join(FORMATS)}"
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
destination = destination.expanduser().resolve()
|
|
134
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
135
|
+
handle, temporary_name = tempfile.mkstemp(
|
|
136
|
+
prefix=f".{destination.name}.",
|
|
137
|
+
suffix=f".{audio_format}",
|
|
138
|
+
dir=destination.parent,
|
|
139
|
+
)
|
|
140
|
+
os.close(handle)
|
|
141
|
+
temporary = Path(temporary_name)
|
|
142
|
+
try:
|
|
143
|
+
if audio_format == "wav":
|
|
144
|
+
_write_wav(samples, sample_rate, temporary)
|
|
145
|
+
else:
|
|
146
|
+
_encode_audio(samples, sample_rate, temporary, audio_format)
|
|
147
|
+
temporary.replace(destination)
|
|
148
|
+
finally:
|
|
149
|
+
temporary.unlink(missing_ok=True)
|
|
150
|
+
return destination
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def write_audio_bytes(data: bytes, destination: Path) -> Path:
|
|
154
|
+
"""Atomically persist audio returned by the localhost service."""
|
|
155
|
+
if not data:
|
|
156
|
+
raise RuntimeError("The Agent Voice service returned an empty audio response")
|
|
157
|
+
destination = destination.expanduser().resolve()
|
|
158
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
159
|
+
handle, temporary_name = tempfile.mkstemp(
|
|
160
|
+
prefix=f".{destination.name}.", dir=destination.parent
|
|
161
|
+
)
|
|
162
|
+
temporary = Path(temporary_name)
|
|
163
|
+
try:
|
|
164
|
+
with os.fdopen(handle, "wb") as output:
|
|
165
|
+
output.write(data)
|
|
166
|
+
temporary.replace(destination)
|
|
167
|
+
finally:
|
|
168
|
+
temporary.unlink(missing_ok=True)
|
|
169
|
+
return destination
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def play_audio(path: Path) -> None:
|
|
173
|
+
"""Decode a recording with bundled FFmpeg and play it through miniaudio."""
|
|
174
|
+
pcm = _decode_for_playback(path)
|
|
175
|
+
_play_pcm(pcm)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _encode_audio(
|
|
179
|
+
samples: NDArray[np.floating],
|
|
180
|
+
sample_rate: int,
|
|
181
|
+
destination: Path,
|
|
182
|
+
audio_format: str,
|
|
183
|
+
) -> None:
|
|
184
|
+
with tempfile.TemporaryDirectory(prefix="agent-voice-") as directory:
|
|
185
|
+
source = Path(directory) / "source.wav"
|
|
186
|
+
_write_wav(samples, sample_rate, source)
|
|
187
|
+
command = [
|
|
188
|
+
_ffmpeg_executable(),
|
|
189
|
+
"-hide_banner",
|
|
190
|
+
"-loglevel",
|
|
191
|
+
"error",
|
|
192
|
+
"-y",
|
|
193
|
+
"-i",
|
|
194
|
+
str(source),
|
|
195
|
+
]
|
|
196
|
+
if audio_format == "mp3":
|
|
197
|
+
command += ["-codec:a", "libmp3lame", "-q:a", "3"]
|
|
198
|
+
elif audio_format == "opus":
|
|
199
|
+
command += ["-codec:a", "libopus", "-b:a", "48k"]
|
|
200
|
+
else:
|
|
201
|
+
command += ["-codec:a", "aac", "-b:a", "128k"]
|
|
202
|
+
command.append(str(destination))
|
|
203
|
+
try:
|
|
204
|
+
subprocess.run(command, check=True, capture_output=True)
|
|
205
|
+
except subprocess.CalledProcessError as error:
|
|
206
|
+
detail = error.stderr.decode(errors="replace").strip()
|
|
207
|
+
raise RuntimeError(
|
|
208
|
+
f"ffmpeg could not create {audio_format}: {detail}"
|
|
209
|
+
) from error
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _ffmpeg_executable() -> str:
|
|
213
|
+
try:
|
|
214
|
+
return imageio_ffmpeg.get_ffmpeg_exe()
|
|
215
|
+
except RuntimeError as error:
|
|
216
|
+
raise RuntimeError(
|
|
217
|
+
"Bundled FFmpeg is unavailable; reinstall Agent Voice"
|
|
218
|
+
) from error
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _decode_for_playback(path: Path) -> bytes:
|
|
222
|
+
command = [
|
|
223
|
+
_ffmpeg_executable(),
|
|
224
|
+
"-hide_banner",
|
|
225
|
+
"-loglevel",
|
|
226
|
+
"error",
|
|
227
|
+
"-i",
|
|
228
|
+
str(path),
|
|
229
|
+
"-f",
|
|
230
|
+
"s16le",
|
|
231
|
+
"-acodec",
|
|
232
|
+
"pcm_s16le",
|
|
233
|
+
"-ar",
|
|
234
|
+
str(PLAYBACK_SAMPLE_RATE),
|
|
235
|
+
"-ac",
|
|
236
|
+
str(PLAYBACK_CHANNELS),
|
|
237
|
+
"pipe:1",
|
|
238
|
+
]
|
|
239
|
+
try:
|
|
240
|
+
completed = subprocess.run(command, check=True, capture_output=True)
|
|
241
|
+
except subprocess.CalledProcessError as error:
|
|
242
|
+
detail = error.stderr.decode(errors="replace").strip()
|
|
243
|
+
raise RuntimeError(
|
|
244
|
+
f"ffmpeg could not decode audio for playback: {detail}"
|
|
245
|
+
) from error
|
|
246
|
+
if not completed.stdout:
|
|
247
|
+
raise RuntimeError("ffmpeg returned empty audio for playback")
|
|
248
|
+
return completed.stdout
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _play_pcm(pcm: bytes) -> None:
|
|
252
|
+
finished = threading.Event()
|
|
253
|
+
stream = _pcm_stream(pcm, finished)
|
|
254
|
+
next(stream)
|
|
255
|
+
try:
|
|
256
|
+
with miniaudio.PlaybackDevice(
|
|
257
|
+
output_format=miniaudio.SampleFormat.SIGNED16,
|
|
258
|
+
nchannels=PLAYBACK_CHANNELS,
|
|
259
|
+
sample_rate=PLAYBACK_SAMPLE_RATE,
|
|
260
|
+
app_name="Agent Voice",
|
|
261
|
+
) as device:
|
|
262
|
+
device.start(stream)
|
|
263
|
+
duration = len(pcm) / (
|
|
264
|
+
PLAYBACK_SAMPLE_RATE * PLAYBACK_CHANNELS * PLAYBACK_SAMPLE_WIDTH
|
|
265
|
+
)
|
|
266
|
+
if not finished.wait(timeout=max(5.0, duration + 5.0)):
|
|
267
|
+
raise RuntimeError("audio playback timed out")
|
|
268
|
+
except miniaudio.MiniaudioError as error:
|
|
269
|
+
raise RuntimeError(f"audio playback failed: {error}") from error
|
|
270
|
+
finally:
|
|
271
|
+
stream.close()
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def _pcm_stream(
|
|
275
|
+
pcm: bytes, finished: threading.Event
|
|
276
|
+
) -> Generator[bytes | memoryview, int, None]:
|
|
277
|
+
offset = 0
|
|
278
|
+
frame_width = PLAYBACK_CHANNELS * PLAYBACK_SAMPLE_WIDTH
|
|
279
|
+
try:
|
|
280
|
+
required_frames = yield b""
|
|
281
|
+
while offset < len(pcm):
|
|
282
|
+
requested_bytes = (required_frames or 4096) * frame_width
|
|
283
|
+
end = min(offset + requested_bytes, len(pcm))
|
|
284
|
+
required_frames = yield memoryview(pcm)[offset:end]
|
|
285
|
+
offset = end
|
|
286
|
+
finally:
|
|
287
|
+
finished.set()
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _write_wav(samples: NDArray[np.floating], sample_rate: int, path: Path) -> None:
|
|
291
|
+
normalized = np.asarray(samples, dtype=np.float32).reshape(-1)
|
|
292
|
+
pcm = (np.clip(normalized, -1.0, 1.0) * 32767).astype("<i2")
|
|
293
|
+
with wave.open(str(path), "wb") as output:
|
|
294
|
+
output.setnchannels(1)
|
|
295
|
+
output.setsampwidth(2)
|
|
296
|
+
output.setframerate(sample_rate)
|
|
297
|
+
output.writeframes(pcm.tobytes())
|