vv-synth 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
main.py ADDED
@@ -0,0 +1,131 @@
1
+ """Text-to-speech CLI for VOICEVOX Engine (``vv-synth``)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+ from pathlib import Path
7
+ from typing import Never
8
+
9
+ import typer
10
+
11
+ from vv_synth.engine_client import (
12
+ DEFAULT_ENGINE_URL,
13
+ DEFAULT_SPEED_SCALE,
14
+ DEFAULT_STYLE_ID,
15
+ EngineClientError,
16
+ synthesize_text_to_file,
17
+ )
18
+ from vv_synth.output_paths import (
19
+ DEFAULT_OUTPUT_DIR,
20
+ resolve_output_file,
21
+ )
22
+
23
+ app = typer.Typer(
24
+ name="vv-synth",
25
+ add_completion=False,
26
+ no_args_is_help=True,
27
+ rich_markup_mode=None,
28
+ help="Synthesize text to WAV via VOICEVOX Engine.",
29
+ )
30
+
31
+
32
+ def _validate_speed(value: float) -> float:
33
+ """Reject non-finite CLI speech rates.
34
+
35
+ Args:
36
+ value: Parsed speech rate.
37
+
38
+ Returns:
39
+ Validated speech rate.
40
+ """
41
+ if not math.isfinite(value):
42
+ msg = "Speed must be a finite number."
43
+ raise typer.BadParameter(msg)
44
+
45
+ return value
46
+
47
+
48
+ def _exit_with_error(message: str) -> Never:
49
+ """Print a single-line synthesis error and exit.
50
+
51
+ Args:
52
+ message: Error description to print on stderr.
53
+ """
54
+ typer.echo(" ".join(message.splitlines()), err=True)
55
+ raise typer.Exit(code=1) from None
56
+
57
+
58
+ @app.command()
59
+ def synth(
60
+ message: str = typer.Argument(
61
+ ...,
62
+ metavar="MESSAGE",
63
+ help="Text to speak.",
64
+ ),
65
+ output: Path | None = typer.Option(
66
+ None,
67
+ "--output",
68
+ "-o",
69
+ help="Output WAV file name or path.",
70
+ ),
71
+ output_dir: Path = typer.Option(
72
+ DEFAULT_OUTPUT_DIR,
73
+ "--output-dir",
74
+ help="Directory for WAV output files.",
75
+ envvar="VV_SYNTH_OUTPUT_DIR",
76
+ ),
77
+ speaker: int = typer.Option(
78
+ DEFAULT_STYLE_ID,
79
+ "--speaker",
80
+ "-s",
81
+ help="Speaker style ID (Engine GET /speakers).",
82
+ ),
83
+ speed: float = typer.Option(
84
+ DEFAULT_SPEED_SCALE,
85
+ "--speed",
86
+ min=0.01,
87
+ max=10.0,
88
+ callback=_validate_speed,
89
+ help="Finite speech rate; 1.0 is normal, higher is faster.",
90
+ ),
91
+ engine_url: str = typer.Option(
92
+ DEFAULT_ENGINE_URL,
93
+ "--engine-url",
94
+ help="VOICEVOX Engine base URL.",
95
+ envvar="VOICEVOX_ENGINE_URL",
96
+ ),
97
+ ) -> None:
98
+ """Synthesize MESSAGE to a WAV file.
99
+
100
+ Without -o, writes a local-time timestamped file under output/
101
+ in the current directory (auto-created). Use --output-dir or
102
+ set VV_SYNTH_OUTPUT_DIR to change the directory.
103
+ """
104
+ try:
105
+ output_path = resolve_output_file(output_dir=output_dir, filename=output)
106
+ except OSError as exc:
107
+ _exit_with_error(f"Could not create the output directory: {exc}")
108
+
109
+ try:
110
+ saved = synthesize_text_to_file(
111
+ message,
112
+ output_path,
113
+ base_url=engine_url,
114
+ style_id=speaker,
115
+ speed_scale=speed,
116
+ )
117
+ except (EngineClientError, ValueError, TypeError) as exc:
118
+ _exit_with_error(str(exc))
119
+ except OSError as exc:
120
+ _exit_with_error(f"Could not write the WAV file: {exc}")
121
+
122
+ typer.echo(f"INFO: wrote {saved.resolve()}")
123
+
124
+
125
+ def main() -> None:
126
+ """Entry point for the ``vv-synth`` command."""
127
+ app()
128
+
129
+
130
+ if __name__ == "__main__":
131
+ main()
vv_synth/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """Library package for the ``vv-synth`` CLI."""
2
+
3
+ from __future__ import annotations
@@ -0,0 +1,168 @@
1
+ """VOICEVOX Engine HTTP API client."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import http.client
6
+ import json
7
+ import math
8
+ import urllib.error
9
+ import urllib.parse
10
+ import urllib.request
11
+ from typing import TYPE_CHECKING, Any, cast
12
+
13
+ if TYPE_CHECKING:
14
+ from pathlib import Path
15
+
16
+ # Type alias for the AudioQuery JSON object returned by the Engine API.
17
+ AudioQuery = dict[str, Any]
18
+
19
+ DEFAULT_ENGINE_URL = "http://127.0.0.1:50021"
20
+ DEFAULT_STYLE_ID = 2
21
+ DEFAULT_SPEED_SCALE = 1.0
22
+
23
+ # HTTP timeouts (seconds) per Engine endpoint.
24
+ AUDIO_QUERY_TIMEOUT_S = 60
25
+ SYNTHESIS_TIMEOUT_S = 120
26
+
27
+
28
+ class EngineClientError(Exception):
29
+ """Error raised when a VOICEVOX Engine call fails."""
30
+
31
+
32
+ def _post(url: str, body: bytes | None, *, timeout: int) -> bytes:
33
+ """POST a request body and return the raw response bytes."""
34
+ request = urllib.request.Request(
35
+ url,
36
+ data=body,
37
+ method="POST",
38
+ headers={"Content-Type": "application/json"},
39
+ )
40
+
41
+ with urllib.request.urlopen(request, timeout=timeout) as response:
42
+ return cast("bytes", response.read())
43
+
44
+
45
+ def _post_json(url: str) -> object:
46
+ """POST without a body and return the JSON response as a Python object."""
47
+ raw = _post(url, None, timeout=AUDIO_QUERY_TIMEOUT_S)
48
+
49
+ return json.loads(raw.decode("utf-8"))
50
+
51
+
52
+ def _post_wav(url: str, audio_query: AudioQuery) -> bytes:
53
+ """POST an AudioQuery and return the synthesized WAV bytes."""
54
+ body = json.dumps(audio_query).encode("utf-8")
55
+
56
+ return _post(url, body, timeout=SYNTHESIS_TIMEOUT_S)
57
+
58
+
59
+ def create_audio_query(
60
+ base_url: str,
61
+ text: str,
62
+ style_id: int,
63
+ ) -> AudioQuery:
64
+ """Create an AudioQuery from text."""
65
+ query = urllib.parse.urlencode({"text": text, "speaker": style_id})
66
+ url = f"{base_url.rstrip('/')}/audio_query?{query}"
67
+ result = _post_json(url)
68
+
69
+ if not isinstance(result, dict):
70
+ msg = f"audio_query response is not a dict: {type(result)!r}"
71
+ raise TypeError(msg)
72
+
73
+ return cast("AudioQuery", result)
74
+
75
+
76
+ def synthesize_wav(
77
+ base_url: str,
78
+ audio_query: AudioQuery,
79
+ style_id: int,
80
+ ) -> bytes:
81
+ """Synthesize WAV audio from an AudioQuery."""
82
+ query = urllib.parse.urlencode({"speaker": style_id})
83
+ url = f"{base_url.rstrip('/')}/synthesis?{query}"
84
+
85
+ return _post_wav(url, audio_query)
86
+
87
+
88
+ def save_wav(path: Path, wav_data: bytes) -> None:
89
+ """Save WAV bytes to a file."""
90
+ path.parent.mkdir(parents=True, exist_ok=True)
91
+
92
+ path.write_bytes(wav_data)
93
+
94
+
95
+ def apply_speed_scale(audio_query: AudioQuery, speed_scale: float) -> AudioQuery:
96
+ """Set the speech rate (speedScale) on an AudioQuery."""
97
+ if not math.isfinite(speed_scale) or speed_scale <= 0:
98
+ msg = f"speed_scale must be finite and positive: {speed_scale}"
99
+ raise ValueError(msg)
100
+
101
+ audio_query["speedScale"] = speed_scale
102
+
103
+ return audio_query
104
+
105
+
106
+ def synthesize_text_to_file(
107
+ text: str,
108
+ output: Path,
109
+ *,
110
+ base_url: str = DEFAULT_ENGINE_URL,
111
+ style_id: int = DEFAULT_STYLE_ID,
112
+ speed_scale: float = DEFAULT_SPEED_SCALE,
113
+ ) -> Path:
114
+ """Synthesize text and save it to a WAV file.
115
+
116
+ Args:
117
+ text: Text to speak.
118
+ output: Destination WAV file.
119
+ base_url: Engine base URL.
120
+ style_id: Speaker style ID.
121
+ speed_scale: Speech rate. 1.0 is normal; higher is faster.
122
+
123
+ Returns:
124
+ Path to the saved file.
125
+
126
+ Raises:
127
+ EngineClientError: Engine is unavailable, the API returns an error, etc.
128
+ """
129
+ try:
130
+ audio_query = create_audio_query(base_url, text, style_id)
131
+ apply_speed_scale(audio_query, speed_scale)
132
+ wav_data = synthesize_wav(base_url, audio_query, style_id)
133
+
134
+ except urllib.error.HTTPError as exc:
135
+ msg = (
136
+ f"Engine API returned an error (HTTP {exc.code}). "
137
+ "Check that the speaker style ID is valid for this Engine."
138
+ )
139
+ raise EngineClientError(msg) from exc
140
+
141
+ except urllib.error.URLError as exc:
142
+ msg = (
143
+ "Could not connect to VOICEVOX Engine. "
144
+ "Start the Engine (e.g. Docker voicevox/voicevox_engine:cpu-latest "
145
+ "on port 50021) and ensure it is reachable."
146
+ )
147
+ raise EngineClientError(msg) from exc
148
+
149
+ # urlopen() wraps only connection failures in URLError. A timeout or a dropped
150
+ # connection while waiting for the response escapes unwrapped.
151
+ except TimeoutError as exc:
152
+ msg = (
153
+ "VOICEVOX Engine did not respond in time. "
154
+ "Split long text into shorter runs, or check that the Engine is not busy."
155
+ )
156
+ raise EngineClientError(msg) from exc
157
+
158
+ except (OSError, http.client.HTTPException) as exc:
159
+ msg = (
160
+ "Lost connection to VOICEVOX Engine before the response completed. "
161
+ "Check that the Engine is still running."
162
+ )
163
+ raise EngineClientError(msg) from exc
164
+
165
+ # Write failures stay OSError so they are not reported as Engine problems.
166
+ save_wav(output, wav_data)
167
+
168
+ return output
@@ -0,0 +1,52 @@
1
+ """Manage output paths for synthesized audio artifacts."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from time import strftime
7
+
8
+ # Default output directory for artifacts such as synthesized WAV files.
9
+ DEFAULT_OUTPUT_DIR = Path("output")
10
+
11
+
12
+ def ensure_output_dir(output_dir: Path) -> Path:
13
+ """Create the output directory if it does not exist.
14
+
15
+ Args:
16
+ output_dir: Directory to create.
17
+
18
+ Returns:
19
+ Path to the created directory.
20
+ """
21
+ output_dir.mkdir(parents=True, exist_ok=True)
22
+
23
+ return output_dir
24
+
25
+
26
+ def resolve_output_file(
27
+ *,
28
+ output_dir: Path,
29
+ filename: Path | None = None,
30
+ ) -> Path:
31
+ """Resolve the destination path for an artifact WAV file.
32
+
33
+ When ``filename`` is omitted, use a timestamped name under ``output_dir``.
34
+ When only a file name is provided, place it under ``output_dir``.
35
+ When a path with directories is provided, use that path as-is.
36
+
37
+ Args:
38
+ output_dir: Directory that stores artifacts.
39
+ filename: Output file name or path. Auto-generated when None.
40
+
41
+ Returns:
42
+ Destination file path.
43
+ """
44
+ if filename is None:
45
+ timestamp = strftime("%Y%m%d-%H%M%S")
46
+ return ensure_output_dir(output_dir) / f"{timestamp}.wav"
47
+
48
+ if filename.parent.parts:
49
+ ensure_output_dir(filename.parent)
50
+ return filename
51
+
52
+ return ensure_output_dir(output_dir) / filename.name
@@ -0,0 +1,490 @@
1
+ Metadata-Version: 2.5
2
+ Name: vv-synth
3
+ Version: 0.1.0
4
+ Summary: VOICEVOX Engine text-to-speech CLI (vv-synth)
5
+ Project-URL: Homepage, https://github.com/ru-461/vv-synth
6
+ Project-URL: Repository, https://github.com/ru-461/vv-synth
7
+ Project-URL: Issues, https://github.com/ru-461/vv-synth/issues
8
+ Project-URL: Changelog, https://github.com/ru-461/vv-synth/releases
9
+ Author: ru-461
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: cli,japanese,speech-synthesis,text-to-speech,tts,voicevox
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Environment :: Console
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3.14
19
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
20
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
21
+ Classifier: Typing :: Typed
22
+ Requires-Python: >=3.14
23
+ Requires-Dist: typer>=0.15
24
+ Description-Content-Type: text/markdown
25
+
26
+ # vv-synth
27
+
28
+ [日本語版](README.ja.md)
29
+
30
+ `vv-synth` is a small CLI that sends text to [VOICEVOX](https://voicevox.hiroshiba.jp/) Engine and writes the synthesized audio as a local WAV file.
31
+
32
+ The project intentionally stays thin:
33
+
34
+ - `vv-synth` talks to a separately prepared VOICEVOX Engine over HTTP.
35
+ - VOICEVOX Engine, voice libraries, model files, Docker images, official binaries, and generated WAV files are not bundled in this repository.
36
+ - Runtime behavior is centered in [`vv_synth/engine_client.py`](vv_synth/engine_client.py), [`vv_synth/output_paths.py`](vv_synth/output_paths.py), and [`main.py`](main.py).
37
+
38
+ ## Documentation
39
+
40
+ - **Users:** [Quick Start](#quick-start) / [Usage](#usage)
41
+ - **VOICEVOX terms:** [OSS Publication And VOICEVOX Terms](#oss-publication-and-voicevox-terms)
42
+ - **Maintainers:** [Architecture](#architecture) / [Maintenance](#maintenance)
43
+ - **AI coding agents (Codex CLI / Claude Code recommended):** [`AGENTS.md`](AGENTS.md) / [`CLAUDE.md`](CLAUDE.md) / [`.claude/rules/`](.claude/rules/) / [`skills/README.md`](skills/README.md)
44
+ - **Agent Skills:** [`skills/vv-synth/`](skills/vv-synth/) for portable TTS, [`skills/vv-synth-dev/`](skills/vv-synth-dev/) for this repository
45
+
46
+ ## Requirements
47
+
48
+ | Requirement | Purpose |
49
+ |-------------|---------|
50
+ | Python 3.14+ | Runtime |
51
+ | [uv](https://docs.astral.sh/uv/) | Dependency and tool management |
52
+ | [Docker](https://www.docker.com/) | Optional way to run VOICEVOX Engine |
53
+ | VOICEVOX Engine | HTTP API on `http://127.0.0.1:50021`, prepared with Docker or an official binary |
54
+
55
+ Windows is supported as long as Python, uv, and a reachable VOICEVOX Engine are available. For Windows NVIDIA GPU usage, see [Windows + NVIDIA GPU](#windows--nvidia-gpu-docker-desktop).
56
+
57
+ VOICEVOX CORE (`voicevox_core/`) is not required by `vv-synth`.
58
+
59
+ ## OSS Publication And VOICEVOX Terms
60
+
61
+ `vv-synth` is open source under the MIT License and contains only the small HTTP client CLI. It does not vendor VOICEVOX Engine itself, voice libraries, model files, Docker images, official Windows/macOS/Linux binaries, or generated WAV files.
62
+
63
+ Users are responsible for checking and following the latest official VOICEVOX terms. Using generated audio requires credit identifying VOICEVOX and compliance with each voice library / speaker's own terms. The required credit may be omitted only where an applicable license explicitly permits it. If generated audio is embedded in an application or redistributed, the final distribution must also satisfy those terms and credit requirements.
64
+
65
+ When granting others permission to use generated audio, you must require them to comply with the applicable voice library terms and to pass on these same obligations whenever they grant further permission to use the audio, as required by items 2 and 3 of the official VOICEVOX software terms.
66
+
67
+ References:
68
+
69
+ - [VOICEVOX official software terms](https://voicevox.hiroshiba.jp/term/)
70
+ - [voicevox/voicevox_engine on Docker Hub](https://hub.docker.com/r/voicevox/voicevox_engine)
71
+ - [VOICEVOX Engine Releases](https://github.com/VOICEVOX/voicevox_engine/releases)
72
+ - [VOICEVOX Q&A](https://voicevox.hiroshiba.jp/qa/)
73
+
74
+ Maintainer checklist for every change and release:
75
+
76
+ - Confirm the code license in `LICENSE` (MIT) matches the `pyproject.toml` license metadata.
77
+ - Confirm `version` in `pyproject.toml`, `metadata.version` in both `skills/*/SKILL.md` files, and the release tag `v<version>` (created by `gh skill publish --tag`) match.
78
+ - Confirm the VOICEVOX official links in this README are current.
79
+ - Confirm VOICEVOX Engine, voice libraries, model files, and generated WAV files are not tracked by Git.
80
+ - If sample audio is ever distributed, confirm speaker-specific terms and credit notation first.
81
+
82
+ ## Prepare VOICEVOX Engine
83
+
84
+ `vv-synth` only needs an HTTP Engine listening on `http://127.0.0.1:50021`. The VOICEVOX desktop GUI is not required.
85
+
86
+ Docker users should use the official image. If you do not want Docker, use an official Engine binary from [VOICEVOX Engine Releases](https://github.com/VOICEVOX/voicevox_engine/releases).
87
+
88
+ Sources: [voicevox/voicevox_engine on Docker Hub](https://hub.docker.com/r/voicevox/voicevox_engine), [Docker Desktop GPU support](https://docs.docker.com/desktop/features/gpu/), [VOICEVOX Q&A](https://voicevox.hiroshiba.jp/qa/)
89
+
90
+ ### Docker CPU
91
+
92
+ Pull the image:
93
+
94
+ ```shell
95
+ docker pull voicevox/voicevox_engine:cpu-latest
96
+ ```
97
+
98
+ Run in the foreground:
99
+
100
+ ```shell
101
+ docker run --rm -it -p '127.0.0.1:50021:50021' voicevox/voicevox_engine:cpu-latest
102
+ ```
103
+
104
+ Run in the background:
105
+
106
+ ```shell
107
+ docker run --rm -d -p '127.0.0.1:50021:50021' voicevox/voicevox_engine:cpu-latest
108
+ ```
109
+
110
+ Stop a background container with `docker ps`, then `docker stop <id>`.
111
+
112
+ ### Windows + NVIDIA GPU (Docker Desktop)
113
+
114
+ Windows GPU usage with Docker requires:
115
+
116
+ - Windows 10 / 11 with an NVIDIA GPU
117
+ - Docker Desktop with the WSL2 backend enabled
118
+ - A current NVIDIA driver that supports WSL2 GPU usage
119
+ - A current WSL2 Linux kernel (`wsl --update` in PowerShell)
120
+
121
+ Check whether Docker can see the GPU:
122
+
123
+ ```shell
124
+ docker run --rm -it --gpus=all nvcr.io/nvidia/k8s/cuda-sample:nbody nbody -gpu -benchmark
125
+ ```
126
+
127
+ Run the NVIDIA GPU Engine image:
128
+
129
+ ```shell
130
+ docker pull voicevox/voicevox_engine:nvidia-latest
131
+ docker run --rm -it --gpus all -p '127.0.0.1:50021:50021' voicevox/voicevox_engine:nvidia-latest
132
+ ```
133
+
134
+ Background:
135
+
136
+ ```shell
137
+ docker run --rm -d --gpus all -p '127.0.0.1:50021:50021' voicevox/voicevox_engine:nvidia-latest
138
+ ```
139
+
140
+ After the Engine is listening, `vv-synth` usage is the same as CPU mode. GPU selection is an Engine concern; `vv-synth` does not need a GPU-specific option.
141
+
142
+ ### Official Engine Binaries
143
+
144
+ If you do not use Docker, download an official binary for your OS from [VOICEVOX Engine Releases](https://github.com/VOICEVOX/voicevox_engine/releases). Windows builds include CPU, GPU/DirectML, and GPU/CUDA variants.
145
+
146
+ Once `http://127.0.0.1:50021/version` responds, `vv-synth` can use the binary Engine exactly like the Docker Engine. If you change the port, pass `--engine-url` or set `VOICEVOX_ENGINE_URL`.
147
+
148
+ ### Verify The Engine
149
+
150
+ In another terminal:
151
+
152
+ ```shell
153
+ curl -sSf http://127.0.0.1:50021/version
154
+ ```
155
+
156
+ If JSON is returned, the Engine is ready. Speaker style IDs are available from `/speakers` in [http://127.0.0.1:50021/docs](http://127.0.0.1:50021/docs).
157
+
158
+ ### Engine Troubleshooting
159
+
160
+ | Symptom | Check |
161
+ |---------|-------|
162
+ | `Cannot connect to the Docker daemon` | Start Docker Desktop, or use an official Engine binary |
163
+ | `port is already allocated` on 50021 | Stop other Engine containers or the VOICEVOX desktop app |
164
+ | GPU container cannot select a driver | Confirm Docker Desktop WSL2 backend, NVIDIA driver, and `wsl --update` |
165
+ | `vv-synth` cannot connect | Confirm `curl .../version` works and the port is `127.0.0.1:50021` |
166
+ | HTTP 4xx for speaker | Pick a valid style ID from `/speakers` |
167
+
168
+ Do not run multiple Engines on port 50021 at the same time.
169
+
170
+ ## Quick Start
171
+
172
+ ```shell
173
+ git clone https://github.com/ru-461/vv-synth.git
174
+ cd vv-synth
175
+ uv sync
176
+ uv run vv-synth --help
177
+ ```
178
+
179
+ After starting VOICEVOX Engine with Docker or an official binary:
180
+
181
+ ```shell
182
+ uv run vv-synth "こんにちは、音声合成のテストです。"
183
+ ```
184
+
185
+ The default output is `output/YYYYMMDD-HHMMSS.wav` (local time) under the directory where you run the command. The `output/` directory is created automatically on first run. Set `VV_SYNTH_OUTPUT_DIR` to change the default directory globally.
186
+
187
+ ## Global Install
188
+
189
+ Install directly from GitHub (no clone needed):
190
+
191
+ ```shell
192
+ uv tool install git+https://github.com/ru-461/vv-synth
193
+ ```
194
+
195
+ From a local clone (editable):
196
+
197
+ ```shell
198
+ uv tool install --editable .
199
+ ```
200
+
201
+ Two commands are installed and behave identically:
202
+
203
+ - `vv-synth` — the canonical name
204
+ - `vvs` — short alias
205
+
206
+ If `~/.local/bin` is not on `PATH`:
207
+
208
+ ```shell
209
+ uv tool update-shell
210
+ # or
211
+ export PATH="$HOME/.local/bin:$PATH"
212
+ ```
213
+
214
+ Uninstall:
215
+
216
+ ```shell
217
+ uv tool uninstall vv-synth
218
+ ```
219
+
220
+ If you installed in editable mode and the repository path changed, reinstall:
221
+
222
+ ```shell
223
+ uv tool uninstall vv-synth
224
+ cd /path/to/vv-synth
225
+ uv tool install --editable .
226
+ ```
227
+
228
+ ## Agent Skill Local Install
229
+
230
+ The portable `vv-synth` Agent Skill lets coding agents use this CLI from other projects. `gh skill` is the recommended installation path; `npx skills` (Vercel Labs) is a supported alternative.
231
+
232
+ ### Recommended: `gh skill`
233
+
234
+ Requires GitHub CLI v2.90+:
235
+
236
+ ```shell
237
+ cd /path/to/vv-synth
238
+ gh skill install . vv-synth --from-local --scope user --agent universal
239
+ ```
240
+
241
+ Use a specific agent target if you only want the skill installed for one host:
242
+
243
+ ```shell
244
+ gh skill install . vv-synth --from-local --scope user --agent codex
245
+ gh skill install . vv-synth --from-local --scope user --agent claude-code
246
+ ```
247
+
248
+ Preview the published skill before installing (reads from GitHub; in a local clone, read `skills/vv-synth/SKILL.md` directly):
249
+
250
+ ```shell
251
+ gh skill preview ru-461/vv-synth vv-synth
252
+ ```
253
+
254
+ ### Alternative: `npx skills`
255
+
256
+ Requires Node.js; no global install needed. `--global` targets your user directory (omit it for the current project only):
257
+
258
+ ```shell
259
+ cd /path/to/vv-synth
260
+ npx skills@latest add . --skill vv-synth --global
261
+ ```
262
+
263
+ Target one agent, or list the repository's skills before installing:
264
+
265
+ ```shell
266
+ npx skills@latest add . --skill vv-synth --global --agent claude-code
267
+ npx skills@latest add . --list
268
+ ```
269
+
270
+ See [`skills/README.md`](skills/README.md) for publishing, updates, and project-scoped installs.
271
+
272
+ ## Usage
273
+
274
+ ```shell
275
+ vv-synth MESSAGE [OPTIONS]
276
+ ```
277
+
278
+ `vv-synth --help` is written in English.
279
+
280
+ A successful run prints `INFO: wrote <absolute-path>` to stdout. Synthesis and file I/O errors print one English line to stderr and exit with code 1. Invalid arguments or options produce a Typer usage error on stderr and exit with code 2.
281
+
282
+ | Option | Short | Default | Description |
283
+ |--------|-------|---------|-------------|
284
+ | `MESSAGE` | - | required | Text to synthesize |
285
+ | `--output` | `-o` | automatic timestamp | Output WAV file name or path |
286
+ | `--output-dir` | - | `output` | Directory for WAV files when `--output` is a basename; can also be set with `VV_SYNTH_OUTPUT_DIR` |
287
+ | `--speaker` | `-s` | `2` | VOICEVOX speaker style ID |
288
+ | `--speed` | - | `1.0` | Finite speech speed from `0.01` to `10.0` (`1.0` is normal; larger is faster) |
289
+ | `--engine-url` | - | `http://127.0.0.1:50021` | Engine URL; can also be set with `VOICEVOX_ENGINE_URL` |
290
+
291
+ Examples:
292
+
293
+ ```shell
294
+ vv-synth "テストです" -o hello.wav
295
+ vv-synth "テストです" --output-dir artifacts
296
+ vv-synth "テストです" -s 3 --speed 1.5
297
+ ```
298
+
299
+ Speaker style IDs: [http://127.0.0.1:50021/docs](http://127.0.0.1:50021/docs), `/speakers`
300
+
301
+ ## Architecture
302
+
303
+ ### Component Layout
304
+
305
+ Keep this diagram aligned with the implementation when module boundaries change.
306
+
307
+ ```mermaid
308
+ flowchart TB
309
+ subgraph cli["vv-synth CLI"]
310
+ main["main.py\nTyper"]
311
+ paths["vv_synth/output_paths.py"]
312
+ client["vv_synth/engine_client.py"]
313
+ end
314
+
315
+ subgraph external["External"]
316
+ engine["VOICEVOX Engine\nDocker or binary / :50021"]
317
+ end
318
+
319
+ subgraph fs["Filesystem"]
320
+ outdir["./output/\n(current working directory)"]
321
+ end
322
+
323
+ user(["User / Agent"]) --> main
324
+ main --> paths
325
+ main --> client
326
+ paths --> outdir
327
+ client -->|"POST /audio_query"| engine
328
+ client -->|"POST /synthesis"| engine
329
+ client --> outdir
330
+ ```
331
+
332
+ | Path | Responsibility |
333
+ |------|----------------|
334
+ | `main.py` | Typer CLI and `vv-synth` entry point |
335
+ | `vv_synth/output_paths.py` | Output path resolution (`output/`, `-o`, timestamp names) |
336
+ | `vv_synth/engine_client.py` | Engine HTTP calls, speech rate, WAV saving |
337
+
338
+ ### Synthesis Flow
339
+
340
+ Update this sequence if the Engine API call order changes.
341
+
342
+ ```mermaid
343
+ sequenceDiagram
344
+ actor U as User
345
+ participant M as main.py
346
+ participant P as output_paths
347
+ participant C as engine_client
348
+ participant E as VOICEVOX Engine
349
+
350
+ U->>M: vv-synth MESSAGE [options]
351
+ M->>P: resolve_output_file()
352
+ P-->>M: Path
353
+ M->>C: synthesize_text_to_file()
354
+ C->>E: POST /audio_query?text&speaker
355
+ E-->>C: AudioQuery (JSON)
356
+ C->>C: apply_speed_scale()
357
+ C->>E: POST /synthesis?speaker
358
+ E-->>C: WAV bytes
359
+ C->>C: save_wav()
360
+ C-->>M: Path
361
+ M-->>U: INFO wrote path (stdout)
362
+ ```
363
+
364
+ ### Project Layout
365
+
366
+ ```text
367
+ vv-synth/
368
+ ├── main.py
369
+ ├── vv_synth/
370
+ │ ├── engine_client.py
371
+ │ └── output_paths.py
372
+ ├── tests/ # pytest suite (Engine mocked; no running Engine needed)
373
+ ├── output/ # Git-ignored WAV output, except .gitkeep
374
+ ├── docs/ # Japanese project overview (OVERVIEW.ja.md)
375
+ ├── skills/ # Agent Skills (vv-synth, vv-synth-dev)
376
+ ├── .claude/rules/ # Agent rules referenced by AGENTS.md
377
+ ├── .github/ # CI workflow, Dependabot, issue / PR templates, and Code of Conduct
378
+ ├── pyproject.toml # vv-synth package and [project.scripts]
379
+ ├── AGENTS.md # Shared agent guidance
380
+ ├── CLAUDE.md # Claude Code summary
381
+ ├── CONTRIBUTING.md # Contribution guide
382
+ ├── SECURITY.md # Security policy
383
+ ├── README.md # English main README
384
+ ├── README.ja.md # Japanese README
385
+ ├── LICENSE # MIT (code only; generated audio follows VOICEVOX terms)
386
+ └── uv.lock
387
+ ```
388
+
389
+ ## Output Directory
390
+
391
+ By default, WAV files are written under `output/` relative to the command's current working directory.
392
+
393
+ | Invocation | Output example |
394
+ |------------|----------------|
395
+ | no `-o` | `output/20260521-143052.wav` |
396
+ | `-o hello.wav` | `output/hello.wav` |
397
+ | `-o path/to/a.wav` | `path/to/a.wav` |
398
+ | `--output-dir artifacts` | `artifacts/...` |
399
+
400
+ ## VOICEVOX CORE
401
+
402
+ `vv-synth` only uses the VOICEVOX Engine HTTP API. VOICEVOX CORE, voice libraries, and model files are not bundled with this CLI and are not part of its setup flow.
403
+
404
+ ## Maintenance
405
+
406
+ ### Environment
407
+
408
+ ```shell
409
+ uv sync --group dev
410
+ ```
411
+
412
+ Start VOICEVOX Engine with Docker or an official binary, then confirm:
413
+
414
+ ```shell
415
+ curl -sSf http://127.0.0.1:50021/version
416
+ ```
417
+
418
+ ### Change Map
419
+
420
+ | Change | Primary files | Also update |
421
+ |--------|---------------|-------------|
422
+ | CLI options/help | `main.py` | [Usage](#usage), `vv-synth --help` wording |
423
+ | Output path rules | `vv_synth/output_paths.py` | [Output Directory](#output-directory) |
424
+ | Engine API/speed/errors | `vv_synth/engine_client.py` | Mermaid diagrams |
425
+ | Global command name | `pyproject.toml` `[project.scripts]` | install instructions |
426
+ | Dependency versions | `pyproject.toml` | `uv lock` / `uv.lock` |
427
+ | Agent rules/boundaries | `AGENTS.md`, `CLAUDE.md`, `.claude/rules/*.md` | Architecture tables |
428
+ | Engine setup or VOICEVOX terms guidance | README files, `skills/*/SKILL.md` | official links and Docker/binary parity |
429
+
430
+ ### Mermaid Updates
431
+
432
+ The canonical architecture diagrams live in this README. Update them when:
433
+
434
+ - modules are added, renamed, or change responsibility;
435
+ - Engine API endpoints or call order change;
436
+ - CLI dependencies on the filesystem or external services change.
437
+
438
+ Keep the copies in `README.ja.md` and `docs/OVERVIEW.ja.md` aligned when diagram content changes.
439
+
440
+ ### Quality Checks
441
+
442
+ ```shell
443
+ uv run ruff check .
444
+ uv run ruff format .
445
+ uv run ty check
446
+ uv run pytest
447
+ ```
448
+
449
+ `pytest` runs the unit test suite in `tests/`. Tests mock the Engine over `urllib`, so no
450
+ running VOICEVOX Engine is required.
451
+
452
+ Manual smoke test with Engine running:
453
+
454
+ ```shell
455
+ uv run vv-synth "maintenance smoke test"
456
+ ls -la output/
457
+ ```
458
+
459
+ If package entry points changed:
460
+
461
+ ```shell
462
+ uv tool install --editable .
463
+ vv-synth --help
464
+ ```
465
+
466
+ ### Documentation Sync Checklist
467
+
468
+ - [ ] `README.md` and `README.ja.md` describe the same user-facing behavior.
469
+ - [ ] Mermaid diagrams match the implementation.
470
+ - [ ] `AGENTS.md` / `CLAUDE.md` and Agent Skills are current.
471
+ - [ ] CLI option tables match `vv-synth --help`.
472
+ - [ ] VOICEVOX Engine / voice library terms links and the "do not vendor external assets" policy remain intact.
473
+ - [ ] `LICENSE` and the `pyproject.toml` license metadata still match.
474
+
475
+ ### Commit Policy
476
+
477
+ - Commit only when explicitly requested.
478
+ - Use an English one-line message: `prefix: message`.
479
+ - Examples: `feat: add pitch option`, `docs: update engine setup`.
480
+ - Do not commit `output/*.wav`, `voicevox_core/`, `download`, VOICEVOX Engine binaries, models, voice libraries, or virtual environments.
481
+
482
+ ## Troubleshooting
483
+
484
+ | Symptom | Check |
485
+ |---------|-------|
486
+ | `command not found: vv-synth` | `uv tool install --editable .` and PATH (`~/.local/bin`) |
487
+ | `Connection refused` | Start Engine with Docker or an official binary; confirm port 50021 |
488
+ | HTTP 4xx | Check the `--speaker` style ID |
489
+ | WAV is not saved | Current working directory, `--output-dir`, and write permissions |
490
+ | Mermaid does not render | Markdown fence must be ` ```mermaid ` |
@@ -0,0 +1,9 @@
1
+ main.py,sha256=68cx4UWUocoOkZj_yF0SuCe1HHNmgpfg2jlzDlS69YI,3222
2
+ vv_synth/__init__.py,sha256=GpvEr-mYuJFc8JxtpNasMWPXzwO0LSMIVTZRpKAAN3E,84
3
+ vv_synth/engine_client.py,sha256=78JhYgmW_1vtgGvSAKyQ44fIYcnRoK75eHoezsdYnnU,5011
4
+ vv_synth/output_paths.py,sha256=b_I5VnZIobXJLtK29nYO19zYSSOWiPoSDK2g6JXxW90,1410
5
+ vv_synth-0.1.0.dist-info/METADATA,sha256=SX_LtVMfm_lkO_7yO1V1lKCyZRNWQaIROnIpoPAkFpw,17757
6
+ vv_synth-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
7
+ vv_synth-0.1.0.dist-info/entry_points.txt,sha256=2ukr3LSEXamUrbaoIplkZi76jixSZ46uBjWmLTo9Ivo,55
8
+ vv_synth-0.1.0.dist-info/licenses/LICENSE,sha256=jOtBfB9-XJqaXtC-pRxzw27_z3A25ped9F9clL9gndo,1063
9
+ vv_synth-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,3 @@
1
+ [console_scripts]
2
+ vv-synth = main:main
3
+ vvs = main:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 ru-461
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.