agent-voice 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ """Local voice artifacts for AI agents."""
2
+
3
+ __version__ = "0.5.0"
@@ -0,0 +1,5 @@
1
+ from .cli import main
2
+
3
+
4
+ if __name__ == "__main__":
5
+ main()
agent_voice/audio.py ADDED
@@ -0,0 +1,297 @@
1
+ from __future__ import annotations
2
+
3
+ import os
4
+ import subprocess
5
+ import tempfile
6
+ import threading
7
+ import wave
8
+ from collections.abc import Generator
9
+ from dataclasses import dataclass
10
+ from pathlib import Path
11
+
12
+ import imageio_ffmpeg
13
+ import miniaudio
14
+ import numpy as np
15
+ from numpy.typing import NDArray
16
+
17
+ from .config import FORMATS, MAX_SPEED, MIN_SPEED
18
+
19
+ PLAYBACK_SAMPLE_RATE = 24_000
20
+ PLAYBACK_CHANNELS = 1
21
+ PLAYBACK_SAMPLE_WIDTH = 2
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class AudioRuntime:
26
+ ffmpeg_path: str | None
27
+ ffmpeg_version: str | None
28
+ ffmpeg_error: str | None
29
+ miniaudio_version: str
30
+ playback_backend: str | None
31
+ playback_error: str | None
32
+
33
+
34
+ def inspect_audio_runtime() -> AudioRuntime:
35
+ """Inspect the bundled codec and native playback runtimes."""
36
+ try:
37
+ ffmpeg_path = _ffmpeg_executable()
38
+ ffmpeg_version = imageio_ffmpeg.get_ffmpeg_version()
39
+ ffmpeg_error = None
40
+ except RuntimeError as error:
41
+ ffmpeg_path = None
42
+ ffmpeg_version = None
43
+ ffmpeg_error = str(error)
44
+
45
+ try:
46
+ with miniaudio.PlaybackDevice(
47
+ output_format=miniaudio.SampleFormat.SIGNED16,
48
+ nchannels=PLAYBACK_CHANNELS,
49
+ sample_rate=PLAYBACK_SAMPLE_RATE,
50
+ app_name="Agent Voice",
51
+ ) as device:
52
+ playback_backend = device.backend
53
+ playback_error = None
54
+ except miniaudio.MiniaudioError as error:
55
+ playback_backend = None
56
+ playback_error = str(error)
57
+
58
+ return AudioRuntime(
59
+ ffmpeg_path=ffmpeg_path,
60
+ ffmpeg_version=ffmpeg_version,
61
+ ffmpeg_error=ffmpeg_error,
62
+ miniaudio_version=miniaudio.__version__,
63
+ playback_backend=playback_backend,
64
+ playback_error=playback_error,
65
+ )
66
+
67
+
68
+ def change_tempo(
69
+ samples: NDArray[np.floating], sample_rate: int, factor: float
70
+ ) -> NDArray[np.float32]:
71
+ """Change speech tempo with bundled FFmpeg while preserving pitch."""
72
+ if factor == 1.0:
73
+ return np.asarray(samples, dtype=np.float32).reshape(-1)
74
+ if not MIN_SPEED <= factor <= MAX_SPEED:
75
+ raise ValueError(f"Tempo factor must be between {MIN_SPEED} and {MAX_SPEED}")
76
+
77
+ source = np.asarray(samples, dtype="<f4").reshape(-1)
78
+ tempo_filter = ",".join(
79
+ f"atempo={tempo_factor}" for tempo_factor in _tempo_factors(factor)
80
+ )
81
+ command = [
82
+ _ffmpeg_executable(),
83
+ "-hide_banner",
84
+ "-loglevel",
85
+ "error",
86
+ "-f",
87
+ "f32le",
88
+ "-ar",
89
+ str(sample_rate),
90
+ "-ac",
91
+ "1",
92
+ "-i",
93
+ "pipe:0",
94
+ "-filter:a",
95
+ tempo_filter,
96
+ "-f",
97
+ "f32le",
98
+ "pipe:1",
99
+ ]
100
+ try:
101
+ completed = subprocess.run(
102
+ command, input=source.tobytes(), check=True, capture_output=True
103
+ )
104
+ except subprocess.CalledProcessError as error:
105
+ detail = error.stderr.decode(errors="replace").strip()
106
+ raise RuntimeError(f"ffmpeg could not change speech tempo: {detail}") from error
107
+ return np.frombuffer(completed.stdout, dtype="<f4").copy()
108
+
109
+
110
+ def _tempo_factors(factor: float) -> list[float]:
111
+ """Split aggressive speedups into pitch-preserving FFmpeg tempo stages."""
112
+ factors: list[float] = []
113
+ remaining = float(factor)
114
+ while remaining > 2.0:
115
+ factors.append(2.0)
116
+ remaining /= 2.0
117
+ factors.append(remaining)
118
+ return factors
119
+
120
+
121
+ def write_audio(
122
+ samples: NDArray[np.floating],
123
+ sample_rate: int,
124
+ destination: Path,
125
+ audio_format: str,
126
+ ) -> Path:
127
+ audio_format = audio_format.lower()
128
+ if audio_format not in FORMATS:
129
+ raise ValueError(
130
+ f"Unsupported format '{audio_format}'. Choose: {', '.join(FORMATS)}"
131
+ )
132
+
133
+ destination = destination.expanduser().resolve()
134
+ destination.parent.mkdir(parents=True, exist_ok=True)
135
+ handle, temporary_name = tempfile.mkstemp(
136
+ prefix=f".{destination.name}.",
137
+ suffix=f".{audio_format}",
138
+ dir=destination.parent,
139
+ )
140
+ os.close(handle)
141
+ temporary = Path(temporary_name)
142
+ try:
143
+ if audio_format == "wav":
144
+ _write_wav(samples, sample_rate, temporary)
145
+ else:
146
+ _encode_audio(samples, sample_rate, temporary, audio_format)
147
+ temporary.replace(destination)
148
+ finally:
149
+ temporary.unlink(missing_ok=True)
150
+ return destination
151
+
152
+
153
+ def write_audio_bytes(data: bytes, destination: Path) -> Path:
154
+ """Atomically persist audio returned by the localhost service."""
155
+ if not data:
156
+ raise RuntimeError("The Agent Voice service returned an empty audio response")
157
+ destination = destination.expanduser().resolve()
158
+ destination.parent.mkdir(parents=True, exist_ok=True)
159
+ handle, temporary_name = tempfile.mkstemp(
160
+ prefix=f".{destination.name}.", dir=destination.parent
161
+ )
162
+ temporary = Path(temporary_name)
163
+ try:
164
+ with os.fdopen(handle, "wb") as output:
165
+ output.write(data)
166
+ temporary.replace(destination)
167
+ finally:
168
+ temporary.unlink(missing_ok=True)
169
+ return destination
170
+
171
+
172
+ def play_audio(path: Path) -> None:
173
+ """Decode a recording with bundled FFmpeg and play it through miniaudio."""
174
+ pcm = _decode_for_playback(path)
175
+ _play_pcm(pcm)
176
+
177
+
178
+ def _encode_audio(
179
+ samples: NDArray[np.floating],
180
+ sample_rate: int,
181
+ destination: Path,
182
+ audio_format: str,
183
+ ) -> None:
184
+ with tempfile.TemporaryDirectory(prefix="agent-voice-") as directory:
185
+ source = Path(directory) / "source.wav"
186
+ _write_wav(samples, sample_rate, source)
187
+ command = [
188
+ _ffmpeg_executable(),
189
+ "-hide_banner",
190
+ "-loglevel",
191
+ "error",
192
+ "-y",
193
+ "-i",
194
+ str(source),
195
+ ]
196
+ if audio_format == "mp3":
197
+ command += ["-codec:a", "libmp3lame", "-q:a", "3"]
198
+ elif audio_format == "opus":
199
+ command += ["-codec:a", "libopus", "-b:a", "48k"]
200
+ else:
201
+ command += ["-codec:a", "aac", "-b:a", "128k"]
202
+ command.append(str(destination))
203
+ try:
204
+ subprocess.run(command, check=True, capture_output=True)
205
+ except subprocess.CalledProcessError as error:
206
+ detail = error.stderr.decode(errors="replace").strip()
207
+ raise RuntimeError(
208
+ f"ffmpeg could not create {audio_format}: {detail}"
209
+ ) from error
210
+
211
+
212
+ def _ffmpeg_executable() -> str:
213
+ try:
214
+ return imageio_ffmpeg.get_ffmpeg_exe()
215
+ except RuntimeError as error:
216
+ raise RuntimeError(
217
+ "Bundled FFmpeg is unavailable; reinstall Agent Voice"
218
+ ) from error
219
+
220
+
221
+ def _decode_for_playback(path: Path) -> bytes:
222
+ command = [
223
+ _ffmpeg_executable(),
224
+ "-hide_banner",
225
+ "-loglevel",
226
+ "error",
227
+ "-i",
228
+ str(path),
229
+ "-f",
230
+ "s16le",
231
+ "-acodec",
232
+ "pcm_s16le",
233
+ "-ar",
234
+ str(PLAYBACK_SAMPLE_RATE),
235
+ "-ac",
236
+ str(PLAYBACK_CHANNELS),
237
+ "pipe:1",
238
+ ]
239
+ try:
240
+ completed = subprocess.run(command, check=True, capture_output=True)
241
+ except subprocess.CalledProcessError as error:
242
+ detail = error.stderr.decode(errors="replace").strip()
243
+ raise RuntimeError(
244
+ f"ffmpeg could not decode audio for playback: {detail}"
245
+ ) from error
246
+ if not completed.stdout:
247
+ raise RuntimeError("ffmpeg returned empty audio for playback")
248
+ return completed.stdout
249
+
250
+
251
+ def _play_pcm(pcm: bytes) -> None:
252
+ finished = threading.Event()
253
+ stream = _pcm_stream(pcm, finished)
254
+ next(stream)
255
+ try:
256
+ with miniaudio.PlaybackDevice(
257
+ output_format=miniaudio.SampleFormat.SIGNED16,
258
+ nchannels=PLAYBACK_CHANNELS,
259
+ sample_rate=PLAYBACK_SAMPLE_RATE,
260
+ app_name="Agent Voice",
261
+ ) as device:
262
+ device.start(stream)
263
+ duration = len(pcm) / (
264
+ PLAYBACK_SAMPLE_RATE * PLAYBACK_CHANNELS * PLAYBACK_SAMPLE_WIDTH
265
+ )
266
+ if not finished.wait(timeout=max(5.0, duration + 5.0)):
267
+ raise RuntimeError("audio playback timed out")
268
+ except miniaudio.MiniaudioError as error:
269
+ raise RuntimeError(f"audio playback failed: {error}") from error
270
+ finally:
271
+ stream.close()
272
+
273
+
274
+ def _pcm_stream(
275
+ pcm: bytes, finished: threading.Event
276
+ ) -> Generator[bytes | memoryview, int, None]:
277
+ offset = 0
278
+ frame_width = PLAYBACK_CHANNELS * PLAYBACK_SAMPLE_WIDTH
279
+ try:
280
+ required_frames = yield b""
281
+ while offset < len(pcm):
282
+ requested_bytes = (required_frames or 4096) * frame_width
283
+ end = min(offset + requested_bytes, len(pcm))
284
+ required_frames = yield memoryview(pcm)[offset:end]
285
+ offset = end
286
+ finally:
287
+ finished.set()
288
+
289
+
290
+ def _write_wav(samples: NDArray[np.floating], sample_rate: int, path: Path) -> None:
291
+ normalized = np.asarray(samples, dtype=np.float32).reshape(-1)
292
+ pcm = (np.clip(normalized, -1.0, 1.0) * 32767).astype("<i2")
293
+ with wave.open(str(path), "wb") as output:
294
+ output.setnchannels(1)
295
+ output.setsampwidth(2)
296
+ output.setframerate(sample_rate)
297
+ output.writeframes(pcm.tobytes())