agent-voice 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_voice-0.5.0/.github/scripts/verify_package.py +181 -0
- agent_voice-0.5.0/.github/workflows/ci.yml +110 -0
- agent_voice-0.5.0/.github/workflows/publish.yml +51 -0
- agent_voice-0.5.0/.gitignore +21 -0
- agent_voice-0.5.0/.python-version +1 -0
- agent_voice-0.5.0/LICENSE +21 -0
- agent_voice-0.5.0/PKG-INFO +200 -0
- agent_voice-0.5.0/README.md +172 -0
- agent_voice-0.5.0/THIRD_PARTY_NOTICES.md +10 -0
- agent_voice-0.5.0/agent-voice +6 -0
- agent_voice-0.5.0/assets/brand/agent-voice-icon-voiceprint.png +0 -0
- agent_voice-0.5.0/assets/brand/agent-voice-icon-voiceprint.svg +17 -0
- agent_voice-0.5.0/assets/brand/agent-voice-logo-voiceprint-dark.png +0 -0
- agent_voice-0.5.0/assets/brand/agent-voice-logo-voiceprint-dark.svg +32 -0
- agent_voice-0.5.0/assets/brand/agent-voice-logo-voiceprint.png +0 -0
- agent_voice-0.5.0/assets/brand/agent-voice-logo-voiceprint.svg +32 -0
- agent_voice-0.5.0/docs/viewer-update.html +74 -0
- agent_voice-0.5.0/docs/voice-sampler/audio/af_bella.mp3 +0 -0
- agent_voice-0.5.0/docs/voice-sampler/audio/af_heart.mp3 +0 -0
- agent_voice-0.5.0/docs/voice-sampler/audio/af_nova.mp3 +0 -0
- agent_voice-0.5.0/docs/voice-sampler/audio/af_sky.mp3 +0 -0
- agent_voice-0.5.0/docs/voice-sampler/audio/am_adam.mp3 +0 -0
- agent_voice-0.5.0/docs/voice-sampler/audio/am_michael.mp3 +0 -0
- agent_voice-0.5.0/docs/voice-sampler/audio/bf_emma.mp3 +0 -0
- agent_voice-0.5.0/docs/voice-sampler/audio/bm_george.mp3 +0 -0
- agent_voice-0.5.0/docs/voice-sampler/index.html +594 -0
- agent_voice-0.5.0/models/.gitkeep +1 -0
- agent_voice-0.5.0/pyproject.toml +49 -0
- agent_voice-0.5.0/recordings/.gitkeep +1 -0
- agent_voice-0.5.0/skills/create-speech-recording/SKILL.md +59 -0
- agent_voice-0.5.0/skills/create-speech-recording/agents/openai.yaml +7 -0
- agent_voice-0.5.0/skills/create-speech-recording/references/recording-delivery.md +14 -0
- agent_voice-0.5.0/skills/spoken-response/SKILL.md +71 -0
- agent_voice-0.5.0/skills/spoken-response/agents/openai.yaml +7 -0
- agent_voice-0.5.0/skills/spoken-response/references/recording-delivery.md +14 -0
- agent_voice-0.5.0/src/agent_voice/__init__.py +3 -0
- agent_voice-0.5.0/src/agent_voice/__main__.py +5 -0
- agent_voice-0.5.0/src/agent_voice/audio.py +297 -0
- agent_voice-0.5.0/src/agent_voice/cli.py +492 -0
- agent_voice-0.5.0/src/agent_voice/client.py +302 -0
- agent_voice-0.5.0/src/agent_voice/config.py +225 -0
- agent_voice-0.5.0/src/agent_voice/delivery.py +55 -0
- agent_voice-0.5.0/src/agent_voice/doctor.py +114 -0
- agent_voice-0.5.0/src/agent_voice/kokoro.py +332 -0
- agent_voice-0.5.0/src/agent_voice/media.py +9 -0
- agent_voice-0.5.0/src/agent_voice/model.py +146 -0
- agent_voice-0.5.0/src/agent_voice/paths.py +63 -0
- agent_voice-0.5.0/src/agent_voice/registry.py +102 -0
- agent_voice-0.5.0/src/agent_voice/service.py +339 -0
- agent_voice-0.5.0/src/agent_voice/speaking.py +375 -0
- agent_voice-0.5.0/src/agent_voice/templates/brand-icon.svg +17 -0
- agent_voice-0.5.0/src/agent_voice/templates/brand-logo.svg +32 -0
- agent_voice-0.5.0/src/agent_voice/templates/recording.html +122 -0
- agent_voice-0.5.0/src/agent_voice/viewer.py +287 -0
- agent_voice-0.5.0/src/agent_voice/viewer_server.py +311 -0
- agent_voice-0.5.0/tests/test_audio.py +185 -0
- agent_voice-0.5.0/tests/test_cli.py +352 -0
- agent_voice-0.5.0/tests/test_config.py +203 -0
- agent_voice-0.5.0/tests/test_delivery.py +130 -0
- agent_voice-0.5.0/tests/test_doctor.py +104 -0
- agent_voice-0.5.0/tests/test_engine.py +94 -0
- agent_voice-0.5.0/tests/test_models.py +157 -0
- agent_voice-0.5.0/tests/test_service.py +582 -0
- agent_voice-0.5.0/tests/test_skills.py +105 -0
- agent_voice-0.5.0/tests/test_speaking.py +612 -0
- agent_voice-0.5.0/tests/test_viewer.py +302 -0
- agent_voice-0.5.0/uv.lock +774 -0
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import platform
|
|
6
|
+
import re
|
|
7
|
+
import subprocess
|
|
8
|
+
import sys
|
|
9
|
+
import time
|
|
10
|
+
import urllib.request
|
|
11
|
+
import wave
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
import imageio_ffmpeg
|
|
15
|
+
|
|
16
|
+
SERVICE_URL = "http://127.0.0.1:18765"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def run_cli(cli: Path, *args: str) -> dict[str, object]:
|
|
20
|
+
completed = subprocess.run(
|
|
21
|
+
[str(cli), *args],
|
|
22
|
+
check=True,
|
|
23
|
+
capture_output=True,
|
|
24
|
+
text=True,
|
|
25
|
+
)
|
|
26
|
+
return json.loads(completed.stdout.strip().splitlines()[-1])
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def validate_decodable(path: Path) -> None:
|
|
30
|
+
ffmpeg = imageio_ffmpeg.get_ffmpeg_exe()
|
|
31
|
+
assert "imageio_ffmpeg" in Path(ffmpeg).as_posix()
|
|
32
|
+
completed = subprocess.run(
|
|
33
|
+
[
|
|
34
|
+
ffmpeg,
|
|
35
|
+
"-v",
|
|
36
|
+
"error",
|
|
37
|
+
"-i",
|
|
38
|
+
str(path),
|
|
39
|
+
"-f",
|
|
40
|
+
"s16le",
|
|
41
|
+
"-acodec",
|
|
42
|
+
"pcm_s16le",
|
|
43
|
+
"pipe:1",
|
|
44
|
+
],
|
|
45
|
+
check=True,
|
|
46
|
+
capture_output=True,
|
|
47
|
+
)
|
|
48
|
+
assert completed.stdout
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def validate_wav(path: Path) -> None:
|
|
52
|
+
with wave.open(str(path), "rb") as audio:
|
|
53
|
+
assert audio.getnchannels() == 1
|
|
54
|
+
assert audio.getframerate() == 24_000
|
|
55
|
+
assert audio.getnframes() > 0
|
|
56
|
+
validate_decodable(path)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def validate_mp3(path: Path) -> None:
|
|
60
|
+
assert path.suffix == ".mp3"
|
|
61
|
+
validate_decodable(path)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def check_doctor(report: dict[str, object], service_status: str) -> None:
|
|
65
|
+
assert report["ok"] is True
|
|
66
|
+
checks = {check["name"]: check for check in report["checks"]}
|
|
67
|
+
assert checks["model"]["status"] == "pass"
|
|
68
|
+
assert checks["runtime"]["status"] == "pass"
|
|
69
|
+
assert checks["compressed audio"]["status"] == "pass"
|
|
70
|
+
assert "bundled by imageio-ffmpeg" in checks["compressed audio"]["detail"]
|
|
71
|
+
assert checks["playback"]["status"] in {"pass", "warn"}
|
|
72
|
+
assert "miniaudio" in checks["playback"]["detail"]
|
|
73
|
+
assert checks["service"]["status"] == service_status
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def main() -> None:
|
|
77
|
+
cli = Path(sys.argv[1]).resolve()
|
|
78
|
+
system = platform.system()
|
|
79
|
+
output_dir = Path(os.environ["RUNNER_TEMP"]) / "agent-voice-package-e2e"
|
|
80
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
81
|
+
|
|
82
|
+
subprocess.run([str(cli), "setup", "--model", "int8"], check=True)
|
|
83
|
+
offline_doctor = run_cli(cli, "doctor", "--service-url", SERVICE_URL, "--json")
|
|
84
|
+
check_doctor(offline_doctor, "warn")
|
|
85
|
+
|
|
86
|
+
local_wav = output_dir / "local.wav"
|
|
87
|
+
local = run_cli(
|
|
88
|
+
cli,
|
|
89
|
+
"speak",
|
|
90
|
+
f"{system} generation verification.",
|
|
91
|
+
"--service",
|
|
92
|
+
"off",
|
|
93
|
+
"--output",
|
|
94
|
+
str(local_wav),
|
|
95
|
+
)
|
|
96
|
+
assert local["backend"] == "local"
|
|
97
|
+
assert local["played"] is False
|
|
98
|
+
validate_wav(local_wav)
|
|
99
|
+
|
|
100
|
+
labeled = run_cli(
|
|
101
|
+
cli,
|
|
102
|
+
"speak",
|
|
103
|
+
f"{system} labeled speed verification.",
|
|
104
|
+
"--service",
|
|
105
|
+
"off",
|
|
106
|
+
"--label",
|
|
107
|
+
"Package E2E",
|
|
108
|
+
"--format",
|
|
109
|
+
"mp3",
|
|
110
|
+
"--speed",
|
|
111
|
+
"1.5",
|
|
112
|
+
)
|
|
113
|
+
labeled_path = Path(str(labeled["path"]))
|
|
114
|
+
assert labeled["backend"] == "local"
|
|
115
|
+
assert labeled["speed"] == 1.5
|
|
116
|
+
assert re.fullmatch(
|
|
117
|
+
r"Package-E2E-\d{2}-\d{2}-\d{2}-at-\d{2}-\d{2}\.mp3",
|
|
118
|
+
labeled_path.name,
|
|
119
|
+
)
|
|
120
|
+
assert (
|
|
121
|
+
labeled_path.parent
|
|
122
|
+
== (Path(os.environ["AGENT_VOICE_HOME"]) / "recordings").resolve()
|
|
123
|
+
)
|
|
124
|
+
validate_mp3(labeled_path)
|
|
125
|
+
|
|
126
|
+
log_path = output_dir / "service.log"
|
|
127
|
+
with log_path.open("w", encoding="utf-8") as log:
|
|
128
|
+
service = subprocess.Popen(
|
|
129
|
+
[str(cli), "serve", "--port", "18765"],
|
|
130
|
+
stdout=log,
|
|
131
|
+
stderr=subprocess.STDOUT,
|
|
132
|
+
text=True,
|
|
133
|
+
)
|
|
134
|
+
try:
|
|
135
|
+
for _ in range(120):
|
|
136
|
+
if service.poll() is not None:
|
|
137
|
+
raise RuntimeError(f"service exited with {service.returncode}")
|
|
138
|
+
try:
|
|
139
|
+
with urllib.request.urlopen(
|
|
140
|
+
f"{SERVICE_URL}/health", timeout=1
|
|
141
|
+
) as response:
|
|
142
|
+
if json.loads(response.read()).get("status") == "ok":
|
|
143
|
+
break
|
|
144
|
+
except OSError:
|
|
145
|
+
time.sleep(1)
|
|
146
|
+
else:
|
|
147
|
+
raise RuntimeError("service did not become healthy")
|
|
148
|
+
|
|
149
|
+
online_doctor = run_cli(
|
|
150
|
+
cli, "doctor", "--service-url", SERVICE_URL, "--json"
|
|
151
|
+
)
|
|
152
|
+
check_doctor(online_doctor, "pass")
|
|
153
|
+
|
|
154
|
+
service_wav = output_dir / "service.wav"
|
|
155
|
+
remote = run_cli(
|
|
156
|
+
cli,
|
|
157
|
+
"speak",
|
|
158
|
+
f"{system} localhost service verification.",
|
|
159
|
+
"--service",
|
|
160
|
+
"on",
|
|
161
|
+
"--service-url",
|
|
162
|
+
SERVICE_URL,
|
|
163
|
+
"--output",
|
|
164
|
+
str(service_wav),
|
|
165
|
+
)
|
|
166
|
+
assert remote["backend"] == "service"
|
|
167
|
+
assert remote["played"] is False
|
|
168
|
+
validate_wav(service_wav)
|
|
169
|
+
finally:
|
|
170
|
+
service.terminate()
|
|
171
|
+
try:
|
|
172
|
+
service.wait(timeout=10)
|
|
173
|
+
except subprocess.TimeoutExpired:
|
|
174
|
+
service.kill()
|
|
175
|
+
service.wait(timeout=10)
|
|
176
|
+
print(log_path.read_text(encoding="utf-8", errors="replace"))
|
|
177
|
+
print(f"Verified installed package on {system}")
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
if __name__ == "__main__":
|
|
181
|
+
main()
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
concurrency:
|
|
12
|
+
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
|
|
13
|
+
cancel-in-progress: true
|
|
14
|
+
|
|
15
|
+
jobs:
|
|
16
|
+
test:
|
|
17
|
+
name: ${{ matrix.os }} / Python ${{ matrix.python-version }}
|
|
18
|
+
runs-on: ${{ matrix.os }}
|
|
19
|
+
strategy:
|
|
20
|
+
fail-fast: false
|
|
21
|
+
matrix:
|
|
22
|
+
os: [ubuntu-latest, macos-latest]
|
|
23
|
+
python-version: ["3.11", "3.13"]
|
|
24
|
+
|
|
25
|
+
steps:
|
|
26
|
+
- uses: actions/checkout@v6
|
|
27
|
+
|
|
28
|
+
- name: Install uv
|
|
29
|
+
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
|
|
30
|
+
with:
|
|
31
|
+
enable-cache: true
|
|
32
|
+
python-version: ${{ matrix.python-version }}
|
|
33
|
+
|
|
34
|
+
- name: Cache verified Kokoro model
|
|
35
|
+
if: matrix.python-version == '3.13'
|
|
36
|
+
uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
|
37
|
+
with:
|
|
38
|
+
path: ${{ runner.temp }}/agent-voice-data/models
|
|
39
|
+
key: kokoro-v1-int8-${{ runner.os }}-${{ runner.arch }}
|
|
40
|
+
|
|
41
|
+
- name: Lint
|
|
42
|
+
if: matrix.os == 'ubuntu-latest' && matrix.python-version == '3.11'
|
|
43
|
+
run: uv run --frozen ruff check src tests .github/scripts
|
|
44
|
+
|
|
45
|
+
- name: Test
|
|
46
|
+
run: uv run --frozen pytest -q
|
|
47
|
+
|
|
48
|
+
- name: Build package
|
|
49
|
+
if: matrix.python-version == '3.13'
|
|
50
|
+
run: uv build
|
|
51
|
+
|
|
52
|
+
- name: Install built wheel
|
|
53
|
+
if: matrix.python-version == '3.13'
|
|
54
|
+
run: |
|
|
55
|
+
uv venv --python ${{ matrix.python-version }} .ci-venv
|
|
56
|
+
uv pip install --python .ci-venv/bin/python dist/*.whl
|
|
57
|
+
|
|
58
|
+
- name: Verify installed package with real model, audio, doctor, and service
|
|
59
|
+
if: matrix.python-version == '3.13'
|
|
60
|
+
env:
|
|
61
|
+
AGENT_VOICE_HOME: ${{ runner.temp }}/agent-voice-data
|
|
62
|
+
run: |
|
|
63
|
+
.ci-venv/bin/python .github/scripts/verify_package.py .ci-venv/bin/agent-voice
|
|
64
|
+
|
|
65
|
+
windows:
|
|
66
|
+
name: windows-latest / Python ${{ matrix.python-version }}
|
|
67
|
+
runs-on: windows-latest
|
|
68
|
+
strategy:
|
|
69
|
+
fail-fast: false
|
|
70
|
+
matrix:
|
|
71
|
+
python-version: ["3.11", "3.12", "3.13"]
|
|
72
|
+
|
|
73
|
+
steps:
|
|
74
|
+
- uses: actions/checkout@v6
|
|
75
|
+
|
|
76
|
+
- name: Install uv
|
|
77
|
+
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
|
|
78
|
+
with:
|
|
79
|
+
enable-cache: true
|
|
80
|
+
python-version: ${{ matrix.python-version }}
|
|
81
|
+
|
|
82
|
+
- name: Cache verified Kokoro model
|
|
83
|
+
if: matrix.python-version == '3.13'
|
|
84
|
+
uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
|
85
|
+
with:
|
|
86
|
+
path: ${{ runner.temp }}/agent-voice-data/models
|
|
87
|
+
key: kokoro-v1-int8-${{ runner.os }}-${{ runner.arch }}
|
|
88
|
+
|
|
89
|
+
- name: Test
|
|
90
|
+
run: uv run --frozen pytest -q
|
|
91
|
+
|
|
92
|
+
- name: Build package
|
|
93
|
+
if: matrix.python-version == '3.13'
|
|
94
|
+
run: uv build
|
|
95
|
+
|
|
96
|
+
- name: Install built wheel
|
|
97
|
+
if: matrix.python-version == '3.13'
|
|
98
|
+
shell: pwsh
|
|
99
|
+
run: |
|
|
100
|
+
uv venv --python ${{ matrix.python-version }} .ci-venv
|
|
101
|
+
$wheel = (Get-ChildItem dist\*.whl).FullName
|
|
102
|
+
uv pip install --python .ci-venv\Scripts\python.exe $wheel
|
|
103
|
+
|
|
104
|
+
- name: Verify installed package with real model, audio, doctor, and service
|
|
105
|
+
if: matrix.python-version == '3.13'
|
|
106
|
+
shell: pwsh
|
|
107
|
+
env:
|
|
108
|
+
AGENT_VOICE_HOME: ${{ runner.temp }}/agent-voice-data
|
|
109
|
+
run: |
|
|
110
|
+
.ci-venv\Scripts\python.exe .github\scripts\verify_package.py .ci-venv\Scripts\agent-voice.exe
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
release:
|
|
5
|
+
types: [published]
|
|
6
|
+
|
|
7
|
+
permissions:
|
|
8
|
+
contents: read
|
|
9
|
+
|
|
10
|
+
jobs:
|
|
11
|
+
build:
|
|
12
|
+
name: Build distribution
|
|
13
|
+
runs-on: ubuntu-latest
|
|
14
|
+
|
|
15
|
+
steps:
|
|
16
|
+
- uses: actions/checkout@v6
|
|
17
|
+
|
|
18
|
+
- name: Install uv
|
|
19
|
+
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
|
|
20
|
+
|
|
21
|
+
- name: Build wheel and source distribution
|
|
22
|
+
run: uv build
|
|
23
|
+
|
|
24
|
+
- name: Validate package metadata
|
|
25
|
+
run: uvx twine check dist/*
|
|
26
|
+
|
|
27
|
+
- name: Upload distribution
|
|
28
|
+
uses: actions/upload-artifact@v4
|
|
29
|
+
with:
|
|
30
|
+
name: python-package-distributions
|
|
31
|
+
path: dist/
|
|
32
|
+
|
|
33
|
+
publish:
|
|
34
|
+
name: Publish distribution
|
|
35
|
+
needs: build
|
|
36
|
+
runs-on: ubuntu-latest
|
|
37
|
+
environment:
|
|
38
|
+
name: pypi
|
|
39
|
+
url: https://pypi.org/p/agent-voice
|
|
40
|
+
permissions:
|
|
41
|
+
id-token: write
|
|
42
|
+
|
|
43
|
+
steps:
|
|
44
|
+
- name: Download distribution
|
|
45
|
+
uses: actions/download-artifact@v4
|
|
46
|
+
with:
|
|
47
|
+
name: python-package-distributions
|
|
48
|
+
path: dist/
|
|
49
|
+
|
|
50
|
+
- name: Publish to PyPI
|
|
51
|
+
uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
.venv/
|
|
2
|
+
.DS_Store
|
|
3
|
+
__pycache__/
|
|
4
|
+
.pytest_cache/
|
|
5
|
+
.ruff_cache/
|
|
6
|
+
.agents/
|
|
7
|
+
skills-lock.json
|
|
8
|
+
*.pyc
|
|
9
|
+
*.egg-info/
|
|
10
|
+
dist/
|
|
11
|
+
config.json
|
|
12
|
+
service-start.lock
|
|
13
|
+
viewer.lock
|
|
14
|
+
viewer.json
|
|
15
|
+
models/*.onnx
|
|
16
|
+
models/*.bin
|
|
17
|
+
models/*.lock
|
|
18
|
+
recordings/*
|
|
19
|
+
IDEAS.md
|
|
20
|
+
!models/.gitkeep
|
|
21
|
+
!recordings/.gitkeep
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
3.11
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Yoav Gal
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: agent-voice
|
|
3
|
+
Version: 0.5.0
|
|
4
|
+
Summary: Local, free voice artifacts for AI agents
|
|
5
|
+
Project-URL: Repository, https://github.com/yoav0gal/agent-voice
|
|
6
|
+
Project-URL: Issues, https://github.com/yoav0gal/agent-voice/issues
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Keywords: agents,audio,cli,kokoro,speech,text-to-speech,tts
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Operating System :: MacOS
|
|
14
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
15
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
21
|
+
Requires-Python: <3.14,>=3.11
|
|
22
|
+
Requires-Dist: filelock<4,>=3.20
|
|
23
|
+
Requires-Dist: imageio-ffmpeg==0.6.0
|
|
24
|
+
Requires-Dist: kokoro-onnx==0.5.0
|
|
25
|
+
Requires-Dist: miniaudio==1.71
|
|
26
|
+
Requires-Dist: numpy<3,>=2
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
|
|
29
|
+
# Agent Voice
|
|
30
|
+
|
|
31
|
+
<picture>
|
|
32
|
+
<source media="(prefers-color-scheme: dark)" srcset="assets/brand/agent-voice-logo-voiceprint-dark.svg">
|
|
33
|
+
<source media="(prefers-color-scheme: light)" srcset="assets/brand/agent-voice-logo-voiceprint.svg">
|
|
34
|
+
<img src="assets/brand/agent-voice-logo-voiceprint.svg" alt="Agent Voice logo" width="720">
|
|
35
|
+
</picture>
|
|
36
|
+
|
|
37
|
+
https://github.com/user-attachments/assets/975dcfd0-17ec-4912-b3b1-ec084077f858
|
|
38
|
+
|
|
39
|
+
Local text-to-speech for people and AI agents, powered by
|
|
40
|
+
[Kokoro-82M](https://huggingface.co/hexgrad/Kokoro-82M) (only English is supported).
|
|
41
|
+
|
|
42
|
+
Agent Voice creates WAV, MP3, Opus, or M4A recordings on macOS, Linux, and
|
|
43
|
+
Windows without an API key.
|
|
44
|
+
|
|
45
|
+
Prebuilt dependencies cover macOS arm64/x64, Linux x64, and Windows x64.
|
|
46
|
+
Linux arm64 currently needs a C build toolchain for miniaudio. Native Windows
|
|
47
|
+
arm64 lacks an `imageio-ffmpeg` wheel; use x64 Python under Windows emulation.
|
|
48
|
+
|
|
49
|
+
## Quick start
|
|
50
|
+
|
|
51
|
+
```sh
|
|
52
|
+
uv tool install agent-voice
|
|
53
|
+
agent-voice setup
|
|
54
|
+
agent-voice speak "Hello from Agent Voice." --play
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
`agent-voice setup` downloads and verifies the speech model.
|
|
58
|
+
|
|
59
|
+
## How to use
|
|
60
|
+
|
|
61
|
+
```sh
|
|
62
|
+
# See every recording option
|
|
63
|
+
agent-voice speak --help
|
|
64
|
+
|
|
65
|
+
# Create a recording
|
|
66
|
+
agent-voice speak "The build is finished."
|
|
67
|
+
|
|
68
|
+
# Play an existing recording
|
|
69
|
+
agent-voice play "/absolute/path/recording.mp3"
|
|
70
|
+
|
|
71
|
+
# Create an MP3 with a readable filename
|
|
72
|
+
agent-voice speak "Here is your summary." --format mp3 --label summary
|
|
73
|
+
|
|
74
|
+
# Choose a voice, speed, and exact output
|
|
75
|
+
agent-voice speak "A slower reading." \
|
|
76
|
+
--voice bf_emma --speed 0.85 --output recording.opus
|
|
77
|
+
|
|
78
|
+
# Safely pass agent text through stdin
|
|
79
|
+
printf '%s' "$VISIBLE_TEXT" |
|
|
80
|
+
agent-voice speak --format mp3
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Every `agent-voice speak` command prints one machine-readable JSON receipt
|
|
84
|
+
containing the recording's absolute `path`, percent-encoded `file_uri`, audio
|
|
85
|
+
metadata, and a `delivery` object with optional `browser_url`, `audio_url`, and
|
|
86
|
+
`recording_path` viewer facts.
|
|
87
|
+
|
|
88
|
+
Agent Voice stores the recording plus private transcript metadata in its managed
|
|
89
|
+
recordings directory. A lightweight localhost viewer renders the branded player
|
|
90
|
+
document—with the recording name, native audio controls, and response text—and
|
|
91
|
+
serves the actual WAV, MP3, Opus, or M4A recording. It starts automatically on
|
|
92
|
+
port `8779`, or on a free port when `8779` is occupied.
|
|
93
|
+
|
|
94
|
+
Agents use exactly two delivery routes:
|
|
95
|
+
|
|
96
|
+
1. Render `path` with the current surface's native audio player.
|
|
97
|
+
2. Otherwise render the installed skill's editable `recording-delivery.md`
|
|
98
|
+
template using the structured receipt values:
|
|
99
|
+
|
|
100
|
+
````markdown
|
|
101
|
+
Agent Voice recording recording.mp3
|
|
102
|
+
Listen: [web player](http://127.0.0.1:8779/player/recording.html) · [media app](file:///absolute/path/recording.mp3) · [web audio](http://127.0.0.1:8779/recordings/recording.mp3)
|
|
103
|
+
```sh
|
|
104
|
+
agent-voice play "/absolute/path/recording.mp3"
|
|
105
|
+
```
|
|
106
|
+
````
|
|
107
|
+
|
|
108
|
+
The CLI owns delivery facts, not agent-facing recording prose. Each
|
|
109
|
+
independently installable skill carries `references/recording-delivery.md`,
|
|
110
|
+
where its wording can be customized. The skill derives the media link and
|
|
111
|
+
playback command from the receipt's `file_uri` and `path`, omitting viewer links
|
|
112
|
+
when those optional delivery facts are unavailable.
|
|
113
|
+
|
|
114
|
+
The viewer prefers port `8779` so links survive restarts. If that port is
|
|
115
|
+
occupied, it selects a free port and reports it in the receipt. The web player
|
|
116
|
+
renders the complete branded document, the media-app link opens the local file
|
|
117
|
+
with the operating system default, and web audio serves the recording directly
|
|
118
|
+
over HTTP.
|
|
119
|
+
|
|
120
|
+
The virtual player URL uses the recording name with an `.html` extension. If
|
|
121
|
+
that name already belongs to another format, Agent Voice adds `-2`, `-3`, and
|
|
122
|
+
so on.
|
|
123
|
+
|
|
124
|
+
Manage the lightweight viewer explicitly when needed:
|
|
125
|
+
|
|
126
|
+
```sh
|
|
127
|
+
agent-voice viewer start
|
|
128
|
+
agent-voice viewer stop
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
`--output` still writes the exact requested path. For HTTP delivery, Agent
|
|
132
|
+
Voice copies that audio into the managed recordings directory instead of
|
|
133
|
+
serving arbitrary filesystem paths or creating symlinks.
|
|
134
|
+
|
|
135
|
+
Explore the available voices and models, manage defaults, or check that Agent
|
|
136
|
+
Voice is ready:
|
|
137
|
+
|
|
138
|
+
```sh
|
|
139
|
+
agent-voice voices
|
|
140
|
+
agent-voice models
|
|
141
|
+
agent-voice config --voice bf_emma --speed 1.15
|
|
142
|
+
agent-voice doctor --json
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
## Defaults
|
|
146
|
+
|
|
147
|
+
Run `agent-voice config` to view the active persisted settings and their
|
|
148
|
+
configuration file.
|
|
149
|
+
|
|
150
|
+
| Setting | Built-in default | Save as default | Override once |
|
|
151
|
+
| --- | --- | --- | --- |
|
|
152
|
+
| Voice | `af_heart` | `config --voice NAME` | `speak --voice NAME` |
|
|
153
|
+
| Speed | `1.0×` | `config --speed NUMBER` | `speak --speed NUMBER` |
|
|
154
|
+
| Audio format | MP3 | `config --format FORMAT` | `speak --format FORMAT` |
|
|
155
|
+
| Recording directory | Agent Voice's `recordings/` directory | `config --output-dir DIR` | `speak --output-dir DIR` |
|
|
156
|
+
| Service | `timed` for `10` minutes | `config --service MODE [--service-timeout MINUTES]` | `speak --service MODE [--service-timeout MINUTES]` |
|
|
157
|
+
|
|
158
|
+
`on` leaves the service running, `off` uses embedded inference, and `timed`
|
|
159
|
+
stops the service after the configured number of idle minutes.
|
|
160
|
+
|
|
161
|
+
The service setting is stored as one object. Timed mode includes its duration:
|
|
162
|
+
|
|
163
|
+
```json
|
|
164
|
+
{
|
|
165
|
+
"service": {
|
|
166
|
+
"mode": "timed",
|
|
167
|
+
"timeout_minutes": 10
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
## Agent skills
|
|
173
|
+
|
|
174
|
+
[View Agent Voice on skills.sh](https://skills.sh/b/yoav0gal/agent-voice).
|
|
175
|
+
|
|
176
|
+
```sh
|
|
177
|
+
# Create speech recordings or read text aloud
|
|
178
|
+
npx skills add yoav0gal/agent-voice --skill create-speech-recording --global --agent codex --yes
|
|
179
|
+
|
|
180
|
+
# Add audio to requested written responses
|
|
181
|
+
npx skills add yoav0gal/agent-voice --skill spoken-response --global --agent codex --yes
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
Use `--agent '*'` instead of `--agent codex` to install the same skills for all
|
|
185
|
+
agent destinations recognized by the skills CLI.
|
|
186
|
+
|
|
187
|
+
## Local speech API
|
|
188
|
+
|
|
189
|
+
```sh
|
|
190
|
+
agent-voice serve
|
|
191
|
+
|
|
192
|
+
curl http://127.0.0.1:8765/v1/audio/speech \
|
|
193
|
+
-H 'Content-Type: application/json' \
|
|
194
|
+
-d '{"input":"The task is complete.","voice":"af_heart","response_format":"mp3"}' \
|
|
195
|
+
--output speech.mp3
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
The speech API binds only to localhost. It is separate from the lightweight
|
|
199
|
+
recording viewer. `agent-voice speak` uses the speech API automatically when
|
|
200
|
+
available and falls back to embedded inference.
|