agent-voice 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. agent_voice-0.5.0/.github/scripts/verify_package.py +181 -0
  2. agent_voice-0.5.0/.github/workflows/ci.yml +110 -0
  3. agent_voice-0.5.0/.github/workflows/publish.yml +51 -0
  4. agent_voice-0.5.0/.gitignore +21 -0
  5. agent_voice-0.5.0/.python-version +1 -0
  6. agent_voice-0.5.0/LICENSE +21 -0
  7. agent_voice-0.5.0/PKG-INFO +200 -0
  8. agent_voice-0.5.0/README.md +172 -0
  9. agent_voice-0.5.0/THIRD_PARTY_NOTICES.md +10 -0
  10. agent_voice-0.5.0/agent-voice +6 -0
  11. agent_voice-0.5.0/assets/brand/agent-voice-icon-voiceprint.png +0 -0
  12. agent_voice-0.5.0/assets/brand/agent-voice-icon-voiceprint.svg +17 -0
  13. agent_voice-0.5.0/assets/brand/agent-voice-logo-voiceprint-dark.png +0 -0
  14. agent_voice-0.5.0/assets/brand/agent-voice-logo-voiceprint-dark.svg +32 -0
  15. agent_voice-0.5.0/assets/brand/agent-voice-logo-voiceprint.png +0 -0
  16. agent_voice-0.5.0/assets/brand/agent-voice-logo-voiceprint.svg +32 -0
  17. agent_voice-0.5.0/docs/viewer-update.html +74 -0
  18. agent_voice-0.5.0/docs/voice-sampler/audio/af_bella.mp3 +0 -0
  19. agent_voice-0.5.0/docs/voice-sampler/audio/af_heart.mp3 +0 -0
  20. agent_voice-0.5.0/docs/voice-sampler/audio/af_nova.mp3 +0 -0
  21. agent_voice-0.5.0/docs/voice-sampler/audio/af_sky.mp3 +0 -0
  22. agent_voice-0.5.0/docs/voice-sampler/audio/am_adam.mp3 +0 -0
  23. agent_voice-0.5.0/docs/voice-sampler/audio/am_michael.mp3 +0 -0
  24. agent_voice-0.5.0/docs/voice-sampler/audio/bf_emma.mp3 +0 -0
  25. agent_voice-0.5.0/docs/voice-sampler/audio/bm_george.mp3 +0 -0
  26. agent_voice-0.5.0/docs/voice-sampler/index.html +594 -0
  27. agent_voice-0.5.0/models/.gitkeep +1 -0
  28. agent_voice-0.5.0/pyproject.toml +49 -0
  29. agent_voice-0.5.0/recordings/.gitkeep +1 -0
  30. agent_voice-0.5.0/skills/create-speech-recording/SKILL.md +59 -0
  31. agent_voice-0.5.0/skills/create-speech-recording/agents/openai.yaml +7 -0
  32. agent_voice-0.5.0/skills/create-speech-recording/references/recording-delivery.md +14 -0
  33. agent_voice-0.5.0/skills/spoken-response/SKILL.md +71 -0
  34. agent_voice-0.5.0/skills/spoken-response/agents/openai.yaml +7 -0
  35. agent_voice-0.5.0/skills/spoken-response/references/recording-delivery.md +14 -0
  36. agent_voice-0.5.0/src/agent_voice/__init__.py +3 -0
  37. agent_voice-0.5.0/src/agent_voice/__main__.py +5 -0
  38. agent_voice-0.5.0/src/agent_voice/audio.py +297 -0
  39. agent_voice-0.5.0/src/agent_voice/cli.py +492 -0
  40. agent_voice-0.5.0/src/agent_voice/client.py +302 -0
  41. agent_voice-0.5.0/src/agent_voice/config.py +225 -0
  42. agent_voice-0.5.0/src/agent_voice/delivery.py +55 -0
  43. agent_voice-0.5.0/src/agent_voice/doctor.py +114 -0
  44. agent_voice-0.5.0/src/agent_voice/kokoro.py +332 -0
  45. agent_voice-0.5.0/src/agent_voice/media.py +9 -0
  46. agent_voice-0.5.0/src/agent_voice/model.py +146 -0
  47. agent_voice-0.5.0/src/agent_voice/paths.py +63 -0
  48. agent_voice-0.5.0/src/agent_voice/registry.py +102 -0
  49. agent_voice-0.5.0/src/agent_voice/service.py +339 -0
  50. agent_voice-0.5.0/src/agent_voice/speaking.py +375 -0
  51. agent_voice-0.5.0/src/agent_voice/templates/brand-icon.svg +17 -0
  52. agent_voice-0.5.0/src/agent_voice/templates/brand-logo.svg +32 -0
  53. agent_voice-0.5.0/src/agent_voice/templates/recording.html +122 -0
  54. agent_voice-0.5.0/src/agent_voice/viewer.py +287 -0
  55. agent_voice-0.5.0/src/agent_voice/viewer_server.py +311 -0
  56. agent_voice-0.5.0/tests/test_audio.py +185 -0
  57. agent_voice-0.5.0/tests/test_cli.py +352 -0
  58. agent_voice-0.5.0/tests/test_config.py +203 -0
  59. agent_voice-0.5.0/tests/test_delivery.py +130 -0
  60. agent_voice-0.5.0/tests/test_doctor.py +104 -0
  61. agent_voice-0.5.0/tests/test_engine.py +94 -0
  62. agent_voice-0.5.0/tests/test_models.py +157 -0
  63. agent_voice-0.5.0/tests/test_service.py +582 -0
  64. agent_voice-0.5.0/tests/test_skills.py +105 -0
  65. agent_voice-0.5.0/tests/test_speaking.py +612 -0
  66. agent_voice-0.5.0/tests/test_viewer.py +302 -0
  67. agent_voice-0.5.0/uv.lock +774 -0
@@ -0,0 +1,181 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import os
5
+ import platform
6
+ import re
7
+ import subprocess
8
+ import sys
9
+ import time
10
+ import urllib.request
11
+ import wave
12
+ from pathlib import Path
13
+
14
+ import imageio_ffmpeg
15
+
16
+ SERVICE_URL = "http://127.0.0.1:18765"
17
+
18
+
19
+ def run_cli(cli: Path, *args: str) -> dict[str, object]:
20
+ completed = subprocess.run(
21
+ [str(cli), *args],
22
+ check=True,
23
+ capture_output=True,
24
+ text=True,
25
+ )
26
+ return json.loads(completed.stdout.strip().splitlines()[-1])
27
+
28
+
29
+ def validate_decodable(path: Path) -> None:
30
+ ffmpeg = imageio_ffmpeg.get_ffmpeg_exe()
31
+ assert "imageio_ffmpeg" in Path(ffmpeg).as_posix()
32
+ completed = subprocess.run(
33
+ [
34
+ ffmpeg,
35
+ "-v",
36
+ "error",
37
+ "-i",
38
+ str(path),
39
+ "-f",
40
+ "s16le",
41
+ "-acodec",
42
+ "pcm_s16le",
43
+ "pipe:1",
44
+ ],
45
+ check=True,
46
+ capture_output=True,
47
+ )
48
+ assert completed.stdout
49
+
50
+
51
+ def validate_wav(path: Path) -> None:
52
+ with wave.open(str(path), "rb") as audio:
53
+ assert audio.getnchannels() == 1
54
+ assert audio.getframerate() == 24_000
55
+ assert audio.getnframes() > 0
56
+ validate_decodable(path)
57
+
58
+
59
+ def validate_mp3(path: Path) -> None:
60
+ assert path.suffix == ".mp3"
61
+ validate_decodable(path)
62
+
63
+
64
+ def check_doctor(report: dict[str, object], service_status: str) -> None:
65
+ assert report["ok"] is True
66
+ checks = {check["name"]: check for check in report["checks"]}
67
+ assert checks["model"]["status"] == "pass"
68
+ assert checks["runtime"]["status"] == "pass"
69
+ assert checks["compressed audio"]["status"] == "pass"
70
+ assert "bundled by imageio-ffmpeg" in checks["compressed audio"]["detail"]
71
+ assert checks["playback"]["status"] in {"pass", "warn"}
72
+ assert "miniaudio" in checks["playback"]["detail"]
73
+ assert checks["service"]["status"] == service_status
74
+
75
+
76
+ def main() -> None:
77
+ cli = Path(sys.argv[1]).resolve()
78
+ system = platform.system()
79
+ output_dir = Path(os.environ["RUNNER_TEMP"]) / "agent-voice-package-e2e"
80
+ output_dir.mkdir(parents=True, exist_ok=True)
81
+
82
+ subprocess.run([str(cli), "setup", "--model", "int8"], check=True)
83
+ offline_doctor = run_cli(cli, "doctor", "--service-url", SERVICE_URL, "--json")
84
+ check_doctor(offline_doctor, "warn")
85
+
86
+ local_wav = output_dir / "local.wav"
87
+ local = run_cli(
88
+ cli,
89
+ "speak",
90
+ f"{system} generation verification.",
91
+ "--service",
92
+ "off",
93
+ "--output",
94
+ str(local_wav),
95
+ )
96
+ assert local["backend"] == "local"
97
+ assert local["played"] is False
98
+ validate_wav(local_wav)
99
+
100
+ labeled = run_cli(
101
+ cli,
102
+ "speak",
103
+ f"{system} labeled speed verification.",
104
+ "--service",
105
+ "off",
106
+ "--label",
107
+ "Package E2E",
108
+ "--format",
109
+ "mp3",
110
+ "--speed",
111
+ "1.5",
112
+ )
113
+ labeled_path = Path(str(labeled["path"]))
114
+ assert labeled["backend"] == "local"
115
+ assert labeled["speed"] == 1.5
116
+ assert re.fullmatch(
117
+ r"Package-E2E-\d{2}-\d{2}-\d{2}-at-\d{2}-\d{2}\.mp3",
118
+ labeled_path.name,
119
+ )
120
+ assert (
121
+ labeled_path.parent
122
+ == (Path(os.environ["AGENT_VOICE_HOME"]) / "recordings").resolve()
123
+ )
124
+ validate_mp3(labeled_path)
125
+
126
+ log_path = output_dir / "service.log"
127
+ with log_path.open("w", encoding="utf-8") as log:
128
+ service = subprocess.Popen(
129
+ [str(cli), "serve", "--port", "18765"],
130
+ stdout=log,
131
+ stderr=subprocess.STDOUT,
132
+ text=True,
133
+ )
134
+ try:
135
+ for _ in range(120):
136
+ if service.poll() is not None:
137
+ raise RuntimeError(f"service exited with {service.returncode}")
138
+ try:
139
+ with urllib.request.urlopen(
140
+ f"{SERVICE_URL}/health", timeout=1
141
+ ) as response:
142
+ if json.loads(response.read()).get("status") == "ok":
143
+ break
144
+ except OSError:
145
+ time.sleep(1)
146
+ else:
147
+ raise RuntimeError("service did not become healthy")
148
+
149
+ online_doctor = run_cli(
150
+ cli, "doctor", "--service-url", SERVICE_URL, "--json"
151
+ )
152
+ check_doctor(online_doctor, "pass")
153
+
154
+ service_wav = output_dir / "service.wav"
155
+ remote = run_cli(
156
+ cli,
157
+ "speak",
158
+ f"{system} localhost service verification.",
159
+ "--service",
160
+ "on",
161
+ "--service-url",
162
+ SERVICE_URL,
163
+ "--output",
164
+ str(service_wav),
165
+ )
166
+ assert remote["backend"] == "service"
167
+ assert remote["played"] is False
168
+ validate_wav(service_wav)
169
+ finally:
170
+ service.terminate()
171
+ try:
172
+ service.wait(timeout=10)
173
+ except subprocess.TimeoutExpired:
174
+ service.kill()
175
+ service.wait(timeout=10)
176
+ print(log_path.read_text(encoding="utf-8", errors="replace"))
177
+ print(f"Verified installed package on {system}")
178
+
179
+
180
+ if __name__ == "__main__":
181
+ main()
@@ -0,0 +1,110 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+
8
+ permissions:
9
+ contents: read
10
+
11
+ concurrency:
12
+ group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
13
+ cancel-in-progress: true
14
+
15
+ jobs:
16
+ test:
17
+ name: ${{ matrix.os }} / Python ${{ matrix.python-version }}
18
+ runs-on: ${{ matrix.os }}
19
+ strategy:
20
+ fail-fast: false
21
+ matrix:
22
+ os: [ubuntu-latest, macos-latest]
23
+ python-version: ["3.11", "3.13"]
24
+
25
+ steps:
26
+ - uses: actions/checkout@v6
27
+
28
+ - name: Install uv
29
+ uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
30
+ with:
31
+ enable-cache: true
32
+ python-version: ${{ matrix.python-version }}
33
+
34
+ - name: Cache verified Kokoro model
35
+ if: matrix.python-version == '3.13'
36
+ uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
37
+ with:
38
+ path: ${{ runner.temp }}/agent-voice-data/models
39
+ key: kokoro-v1-int8-${{ runner.os }}-${{ runner.arch }}
40
+
41
+ - name: Lint
42
+ if: matrix.os == 'ubuntu-latest' && matrix.python-version == '3.11'
43
+ run: uv run --frozen ruff check src tests .github/scripts
44
+
45
+ - name: Test
46
+ run: uv run --frozen pytest -q
47
+
48
+ - name: Build package
49
+ if: matrix.python-version == '3.13'
50
+ run: uv build
51
+
52
+ - name: Install built wheel
53
+ if: matrix.python-version == '3.13'
54
+ run: |
55
+ uv venv --python ${{ matrix.python-version }} .ci-venv
56
+ uv pip install --python .ci-venv/bin/python dist/*.whl
57
+
58
+ - name: Verify installed package with real model, audio, doctor, and service
59
+ if: matrix.python-version == '3.13'
60
+ env:
61
+ AGENT_VOICE_HOME: ${{ runner.temp }}/agent-voice-data
62
+ run: |
63
+ .ci-venv/bin/python .github/scripts/verify_package.py .ci-venv/bin/agent-voice
64
+
65
+ windows:
66
+ name: windows-latest / Python ${{ matrix.python-version }}
67
+ runs-on: windows-latest
68
+ strategy:
69
+ fail-fast: false
70
+ matrix:
71
+ python-version: ["3.11", "3.12", "3.13"]
72
+
73
+ steps:
74
+ - uses: actions/checkout@v6
75
+
76
+ - name: Install uv
77
+ uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
78
+ with:
79
+ enable-cache: true
80
+ python-version: ${{ matrix.python-version }}
81
+
82
+ - name: Cache verified Kokoro model
83
+ if: matrix.python-version == '3.13'
84
+ uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
85
+ with:
86
+ path: ${{ runner.temp }}/agent-voice-data/models
87
+ key: kokoro-v1-int8-${{ runner.os }}-${{ runner.arch }}
88
+
89
+ - name: Test
90
+ run: uv run --frozen pytest -q
91
+
92
+ - name: Build package
93
+ if: matrix.python-version == '3.13'
94
+ run: uv build
95
+
96
+ - name: Install built wheel
97
+ if: matrix.python-version == '3.13'
98
+ shell: pwsh
99
+ run: |
100
+ uv venv --python ${{ matrix.python-version }} .ci-venv
101
+ $wheel = (Get-ChildItem dist\*.whl).FullName
102
+ uv pip install --python .ci-venv\Scripts\python.exe $wheel
103
+
104
+ - name: Verify installed package with real model, audio, doctor, and service
105
+ if: matrix.python-version == '3.13'
106
+ shell: pwsh
107
+ env:
108
+ AGENT_VOICE_HOME: ${{ runner.temp }}/agent-voice-data
109
+ run: |
110
+ .ci-venv\Scripts\python.exe .github\scripts\verify_package.py .ci-venv\Scripts\agent-voice.exe
@@ -0,0 +1,51 @@
1
+ name: Publish to PyPI
2
+
3
+ on:
4
+ release:
5
+ types: [published]
6
+
7
+ permissions:
8
+ contents: read
9
+
10
+ jobs:
11
+ build:
12
+ name: Build distribution
13
+ runs-on: ubuntu-latest
14
+
15
+ steps:
16
+ - uses: actions/checkout@v6
17
+
18
+ - name: Install uv
19
+ uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
20
+
21
+ - name: Build wheel and source distribution
22
+ run: uv build
23
+
24
+ - name: Validate package metadata
25
+ run: uvx twine check dist/*
26
+
27
+ - name: Upload distribution
28
+ uses: actions/upload-artifact@v4
29
+ with:
30
+ name: python-package-distributions
31
+ path: dist/
32
+
33
+ publish:
34
+ name: Publish distribution
35
+ needs: build
36
+ runs-on: ubuntu-latest
37
+ environment:
38
+ name: pypi
39
+ url: https://pypi.org/p/agent-voice
40
+ permissions:
41
+ id-token: write
42
+
43
+ steps:
44
+ - name: Download distribution
45
+ uses: actions/download-artifact@v4
46
+ with:
47
+ name: python-package-distributions
48
+ path: dist/
49
+
50
+ - name: Publish to PyPI
51
+ uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,21 @@
1
+ .venv/
2
+ .DS_Store
3
+ __pycache__/
4
+ .pytest_cache/
5
+ .ruff_cache/
6
+ .agents/
7
+ skills-lock.json
8
+ *.pyc
9
+ *.egg-info/
10
+ dist/
11
+ config.json
12
+ service-start.lock
13
+ viewer.lock
14
+ viewer.json
15
+ models/*.onnx
16
+ models/*.bin
17
+ models/*.lock
18
+ recordings/*
19
+ IDEAS.md
20
+ !models/.gitkeep
21
+ !recordings/.gitkeep
@@ -0,0 +1 @@
1
+ 3.11
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Yoav Gal
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,200 @@
1
+ Metadata-Version: 2.4
2
+ Name: agent-voice
3
+ Version: 0.5.0
4
+ Summary: Local, free voice artifacts for AI agents
5
+ Project-URL: Repository, https://github.com/yoav0gal/agent-voice
6
+ Project-URL: Issues, https://github.com/yoav0gal/agent-voice/issues
7
+ License-Expression: MIT
8
+ License-File: LICENSE
9
+ Keywords: agents,audio,cli,kokoro,speech,text-to-speech,tts
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Environment :: Console
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Operating System :: MacOS
14
+ Classifier: Operating System :: Microsoft :: Windows
15
+ Classifier: Operating System :: POSIX :: Linux
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
21
+ Requires-Python: <3.14,>=3.11
22
+ Requires-Dist: filelock<4,>=3.20
23
+ Requires-Dist: imageio-ffmpeg==0.6.0
24
+ Requires-Dist: kokoro-onnx==0.5.0
25
+ Requires-Dist: miniaudio==1.71
26
+ Requires-Dist: numpy<3,>=2
27
+ Description-Content-Type: text/markdown
28
+
29
+ # Agent Voice
30
+
31
+ <picture>
32
+ <source media="(prefers-color-scheme: dark)" srcset="assets/brand/agent-voice-logo-voiceprint-dark.svg">
33
+ <source media="(prefers-color-scheme: light)" srcset="assets/brand/agent-voice-logo-voiceprint.svg">
34
+ <img src="assets/brand/agent-voice-logo-voiceprint.svg" alt="Agent Voice logo" width="720">
35
+ </picture>
36
+
37
+ https://github.com/user-attachments/assets/975dcfd0-17ec-4912-b3b1-ec084077f858
38
+
39
+ Local text-to-speech for people and AI agents, powered by
40
+ [Kokoro-82M](https://huggingface.co/hexgrad/Kokoro-82M) (only English is supported).
41
+
42
+ Agent Voice creates WAV, MP3, Opus, or M4A recordings on macOS, Linux, and
43
+ Windows without an API key.
44
+
45
+ Prebuilt dependencies cover macOS arm64/x64, Linux x64, and Windows x64.
46
+ Linux arm64 currently needs a C build toolchain for miniaudio. Native Windows
47
+ arm64 lacks an `imageio-ffmpeg` wheel; use x64 Python under Windows emulation.
48
+
49
+ ## Quick start
50
+
51
+ ```sh
52
+ uv tool install agent-voice
53
+ agent-voice setup
54
+ agent-voice speak "Hello from Agent Voice." --play
55
+ ```
56
+
57
+ `agent-voice setup` downloads and verifies the speech model.
58
+
59
+ ## How to use
60
+
61
+ ```sh
62
+ # See every recording option
63
+ agent-voice speak --help
64
+
65
+ # Create a recording
66
+ agent-voice speak "The build is finished."
67
+
68
+ # Play an existing recording
69
+ agent-voice play "/absolute/path/recording.mp3"
70
+
71
+ # Create an MP3 with a readable filename
72
+ agent-voice speak "Here is your summary." --format mp3 --label summary
73
+
74
+ # Choose a voice, speed, and exact output
75
+ agent-voice speak "A slower reading." \
76
+ --voice bf_emma --speed 0.85 --output recording.opus
77
+
78
+ # Safely pass agent text through stdin
79
+ printf '%s' "$VISIBLE_TEXT" |
80
+ agent-voice speak --format mp3
81
+ ```
82
+
83
+ Every `agent-voice speak` command prints one machine-readable JSON receipt
84
+ containing the recording's absolute `path`, percent-encoded `file_uri`, audio
85
+ metadata, and a `delivery` object with optional `browser_url`, `audio_url`, and
86
+ `recording_path` viewer facts.
87
+
88
+ Agent Voice stores the recording plus private transcript metadata in its managed
89
+ recordings directory. A lightweight localhost viewer renders the branded player
90
+ document—with the recording name, native audio controls, and response text—and
91
+ serves the actual WAV, MP3, Opus, or M4A recording. It starts automatically on
92
+ port `8779`, or on a free port when `8779` is occupied.
93
+
94
+ Agents use exactly two delivery routes:
95
+
96
+ 1. Render `path` with the current surface's native audio player.
97
+ 2. Otherwise render the installed skill's editable `recording-delivery.md`
98
+ template using the structured receipt values:
99
+
100
+ ````markdown
101
+ Agent Voice recording recording.mp3
102
+ Listen: [web player](http://127.0.0.1:8779/player/recording.html) · [media app](file:///absolute/path/recording.mp3) · [web audio](http://127.0.0.1:8779/recordings/recording.mp3)
103
+ ```sh
104
+ agent-voice play "/absolute/path/recording.mp3"
105
+ ```
106
+ ````
107
+
108
+ The CLI owns delivery facts, not agent-facing recording prose. Each
109
+ independently installable skill carries `references/recording-delivery.md`,
110
+ where its wording can be customized. The skill derives the media link and
111
+ playback command from the receipt's `file_uri` and `path`, omitting viewer links
112
+ when those optional delivery facts are unavailable.
113
+
114
+ The viewer prefers port `8779` so links survive restarts. If that port is
115
+ occupied, it selects a free port and reports it in the receipt. The web player
116
+ renders the complete branded document, the media-app link opens the local file
117
+ with the operating system default, and web audio serves the recording directly
118
+ over HTTP.
119
+
120
+ The virtual player URL uses the recording name with an `.html` extension. If
121
+ that name already belongs to another format, Agent Voice adds `-2`, `-3`, and
122
+ so on.
123
+
124
+ Manage the lightweight viewer explicitly when needed:
125
+
126
+ ```sh
127
+ agent-voice viewer start
128
+ agent-voice viewer stop
129
+ ```
130
+
131
+ `--output` still writes the exact requested path. For HTTP delivery, Agent
132
+ Voice copies that audio into the managed recordings directory instead of
133
+ serving arbitrary filesystem paths or creating symlinks.
134
+
135
+ Explore the available voices and models, manage defaults, or check that Agent
136
+ Voice is ready:
137
+
138
+ ```sh
139
+ agent-voice voices
140
+ agent-voice models
141
+ agent-voice config --voice bf_emma --speed 1.15
142
+ agent-voice doctor --json
143
+ ```
144
+
145
+ ## Defaults
146
+
147
+ Run `agent-voice config` to view the active persisted settings and their
148
+ configuration file.
149
+
150
+ | Setting | Built-in default | Save as default | Override once |
151
+ | --- | --- | --- | --- |
152
+ | Voice | `af_heart` | `config --voice NAME` | `speak --voice NAME` |
153
+ | Speed | `1.0×` | `config --speed NUMBER` | `speak --speed NUMBER` |
154
+ | Audio format | MP3 | `config --format FORMAT` | `speak --format FORMAT` |
155
+ | Recording directory | Agent Voice's `recordings/` directory | `config --output-dir DIR` | `speak --output-dir DIR` |
156
+ | Service | `timed` for `10` minutes | `config --service MODE [--service-timeout MINUTES]` | `speak --service MODE [--service-timeout MINUTES]` |
157
+
158
+ `on` leaves the service running, `off` uses embedded inference, and `timed`
159
+ stops the service after the configured number of idle minutes.
160
+
161
+ The service setting is stored as one object. Timed mode includes its duration:
162
+
163
+ ```json
164
+ {
165
+ "service": {
166
+ "mode": "timed",
167
+ "timeout_minutes": 10
168
+ }
169
+ }
170
+ ```
171
+
172
+ ## Agent skills
173
+
174
+ [View Agent Voice on skills.sh](https://skills.sh/b/yoav0gal/agent-voice).
175
+
176
+ ```sh
177
+ # Create speech recordings or read text aloud
178
+ npx skills add yoav0gal/agent-voice --skill create-speech-recording --global --agent codex --yes
179
+
180
+ # Add audio to requested written responses
181
+ npx skills add yoav0gal/agent-voice --skill spoken-response --global --agent codex --yes
182
+ ```
183
+
184
+ Use `--agent '*'` instead of `--agent codex` to install the same skills for all
185
+ agent destinations recognized by the skills CLI.
186
+
187
+ ## Local speech API
188
+
189
+ ```sh
190
+ agent-voice serve
191
+
192
+ curl http://127.0.0.1:8765/v1/audio/speech \
193
+ -H 'Content-Type: application/json' \
194
+ -d '{"input":"The task is complete.","voice":"af_heart","response_format":"mp3"}' \
195
+ --output speech.mp3
196
+ ```
197
+
198
+ The speech API binds only to localhost. It is separate from the lightweight
199
+ recording viewer. `agent-voice speak` uses the speech API automatically when
200
+ available and falls back to embedded inference.