vocalize-cli 0.1.1__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/PKG-INFO +59 -5
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/README.md +57 -4
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/pyproject.toml +1 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/tests/test_cli.py +63 -5
- vocalize_cli-0.2.1/tests/test_config.py +167 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/tests/test_tts.py +36 -1
- vocalize_cli-0.2.1/tests/test_wizard.py +382 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/vocalize/__init__.py +1 -1
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/vocalize/cli.py +23 -9
- vocalize_cli-0.2.1/vocalize/config.py +157 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/vocalize/exceptions.py +4 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/vocalize/tts.py +15 -2
- vocalize_cli-0.2.1/vocalize/wizard.py +419 -0
- vocalize_cli-0.1.1/tests/test_config.py +0 -39
- vocalize_cli-0.1.1/vocalize/config.py +0 -55
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/.env.example +0 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/.github/workflows/ci.yml +0 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/.gitignore +0 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/LICENSE +0 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/hooks/claude_stop_hook.py +0 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/hooks/install_hook.py +0 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/tests/conftest.py +0 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/tests/test_audio.py +0 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/tests/test_claude_stop_hook.py +0 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/tests/test_install_hook.py +0 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/tests/test_preprocess.py +0 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/vocalize/__main__.py +0 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/vocalize/audio.py +0 -0
- {vocalize_cli-0.1.1 → vocalize_cli-0.2.1}/vocalize/preprocess.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: vocalize-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: A CLI that turns text, markdown, or piped stdin into speech via the ElevenLabs API, with markdown-table-aware preprocessing.
|
|
5
5
|
Project-URL: Homepage, https://github.com/matthager12-collab/vocalize
|
|
6
6
|
Project-URL: Repository, https://github.com/matthager12-collab/vocalize
|
|
@@ -20,6 +20,7 @@ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
|
20
20
|
Requires-Python: >=3.10
|
|
21
21
|
Requires-Dist: click>=8.1
|
|
22
22
|
Requires-Dist: elevenlabs>=2.0
|
|
23
|
+
Requires-Dist: tomli>=2.0; python_version < '3.11'
|
|
23
24
|
Provides-Extra: dev
|
|
24
25
|
Requires-Dist: build; extra == 'dev'
|
|
25
26
|
Requires-Dist: pytest-cov>=4.0; extra == 'dev'
|
|
@@ -109,13 +110,65 @@ vocalize speak-file report.md --voice <voice-id> --model eleven_flash_v2_5 \
|
|
|
109
110
|
# Cap how much gets sent (handy for free-tier character budgets)
|
|
110
111
|
vocalize speak-file long-report.md --max-chars 2000
|
|
111
112
|
|
|
113
|
+
# Slow it down a little
|
|
114
|
+
vocalize speak-file report.md --speed 0.9
|
|
115
|
+
|
|
112
116
|
# Skip the markdown flattening entirely
|
|
113
117
|
vocalize speak "raw **markdown** stays raw" --raw
|
|
114
118
|
```
|
|
115
119
|
|
|
116
120
|
Every synthesis result is cached on disk under `~/.cache/vocalize/`, keyed
|
|
117
|
-
by a hash of (text, voice, model, format) — re-running the same
|
|
118
|
-
twice doesn't burn API quota twice.
|
|
121
|
+
by a hash of (text, voice, model, format, speed) — re-running the same
|
|
122
|
+
command twice doesn't burn API quota twice.
|
|
123
|
+
|
|
124
|
+
## Configuration
|
|
125
|
+
|
|
126
|
+
Each setting is resolved on its own, taking the first source that supplies
|
|
127
|
+
it: CLI flag, then environment variable, then config file, then the
|
|
128
|
+
built-in default.
|
|
129
|
+
|
|
130
|
+
There's an interactive way to set the file up, if you'd rather not write
|
|
131
|
+
TOML by hand. It walks through three lists — voice (with a live preview of
|
|
132
|
+
the highlighted one), model, and speed — shows you a summary, and writes the
|
|
133
|
+
config file below. Unrecognised top-level keys already in that file are
|
|
134
|
+
carried through; comments and layout are not preserved. A file containing a
|
|
135
|
+
TOML table or array is left alone entirely, with a message saying to edit it
|
|
136
|
+
by hand. The wizard paints on the controlling terminal rather than on stdout,
|
|
137
|
+
so it still works under output-capturing wrappers like `op run`.
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
vocalize config
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
Hotkeys: `↑`/`↓` or `k`/`j` move, `Enter` selects, `p` previews the
|
|
144
|
+
highlighted voice, `m` types a value by hand, `q` or `Esc` cancels without
|
|
145
|
+
writing anything.
|
|
146
|
+
|
|
147
|
+
| Setting | Flag | Env var | Config file key | Default |
|
|
148
|
+
|---|---|---|---|---|
|
|
149
|
+
| Voice ID | `--voice` | `VOCALIZE_VOICE` | `voice` | `21m00Tcm4TlvDq8ikWAM` ("Rachel") |
|
|
150
|
+
| Model ID | `--model` | `VOCALIZE_MODEL` | `model` | `eleven_multilingual_v2` |
|
|
151
|
+
| Speed | `--speed` | `VOCALIZE_SPEED` | `speed` | unset — the API's own 1.0 |
|
|
152
|
+
| Max characters | `--max-chars` | `VOCALIZE_MAX_CHARS` (hook only) | not read from the config file | unset on the CLI; 500 in the hook |
|
|
153
|
+
| Hook binary | — | `VOCALIZE_BIN` | not read from the config file | `vocalize` as found on `PATH` |
|
|
154
|
+
|
|
155
|
+
The config file is TOML at `$XDG_CONFIG_HOME/vocalize/config.toml`, falling
|
|
156
|
+
back to `~/.config/vocalize/config.toml`. Flat keys, no sections:
|
|
157
|
+
|
|
158
|
+
```toml
|
|
159
|
+
voice = "21m00Tcm4TlvDq8ikWAM"
|
|
160
|
+
model = "eleven_flash_v2_5"
|
|
161
|
+
speed = 0.95
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
Not having a config file is normal and silent. A file that isn't valid TOML
|
|
165
|
+
is an error naming the file; a key that isn't recognised is a warning on
|
|
166
|
+
stderr, so a typo doesn't pass unnoticed but doesn't stop the run either.
|
|
167
|
+
`speed` must be a number between 0.7 and 1.2 — anything else is a one-line
|
|
168
|
+
error naming the source it came from.
|
|
169
|
+
|
|
170
|
+
The API key is separate and never read from this file: use `--api-key`,
|
|
171
|
+
`ELEVENLABS_API_KEY`, or a `.env` file.
|
|
119
172
|
|
|
120
173
|
## Claude Code integration
|
|
121
174
|
|
|
@@ -194,7 +247,7 @@ vocalize/
|
|
|
194
247
|
__init__.py # package version
|
|
195
248
|
__main__.py # python -m vocalize entry point
|
|
196
249
|
preprocess.py # markdown -> speakable text (pure function, fully unit tested)
|
|
197
|
-
config.py # API key resolution:
|
|
250
|
+
config.py # API key resolution + settings: flag > env > config.toml > default
|
|
198
251
|
exceptions.py # VocalizeError / TTSRequestError
|
|
199
252
|
tts.py # ElevenLabs API wrapper + disk cache (client is injected, so
|
|
200
253
|
# it's mockable in tests without hitting the network)
|
|
@@ -235,7 +288,8 @@ All tests run offline: the ElevenLabs client is dependency-injected into
|
|
|
235
288
|
only plays WAV, so the mp3 files this tool generates likely won't play
|
|
236
289
|
there. Use `--no-play` and open the saved file with whatever's on hand.
|
|
237
290
|
- **The disk cache under `~/.cache/vocalize` grows unbounded.** It's
|
|
238
|
-
content-addressed (keyed by a hash of text, voice, model,
|
|
291
|
+
content-addressed (keyed by a hash of text, voice, model, format, and
|
|
292
|
+
speed),
|
|
239
293
|
so it's always safe to delete some or all of it — nothing will break,
|
|
240
294
|
you'll just re-pay for a re-synthesized clip.
|
|
241
295
|
- **`--api-key` on the command line is visible to other local processes**
|
|
@@ -77,13 +77,65 @@ vocalize speak-file report.md --voice <voice-id> --model eleven_flash_v2_5 \
|
|
|
77
77
|
# Cap how much gets sent (handy for free-tier character budgets)
|
|
78
78
|
vocalize speak-file long-report.md --max-chars 2000
|
|
79
79
|
|
|
80
|
+
# Slow it down a little
|
|
81
|
+
vocalize speak-file report.md --speed 0.9
|
|
82
|
+
|
|
80
83
|
# Skip the markdown flattening entirely
|
|
81
84
|
vocalize speak "raw **markdown** stays raw" --raw
|
|
82
85
|
```
|
|
83
86
|
|
|
84
87
|
Every synthesis result is cached on disk under `~/.cache/vocalize/`, keyed
|
|
85
|
-
by a hash of (text, voice, model, format) — re-running the same
|
|
86
|
-
twice doesn't burn API quota twice.
|
|
88
|
+
by a hash of (text, voice, model, format, speed) — re-running the same
|
|
89
|
+
command twice doesn't burn API quota twice.
|
|
90
|
+
|
|
91
|
+
## Configuration
|
|
92
|
+
|
|
93
|
+
Each setting is resolved on its own, taking the first source that supplies
|
|
94
|
+
it: CLI flag, then environment variable, then config file, then the
|
|
95
|
+
built-in default.
|
|
96
|
+
|
|
97
|
+
There's an interactive way to set the file up, if you'd rather not write
|
|
98
|
+
TOML by hand. It walks through three lists — voice (with a live preview of
|
|
99
|
+
the highlighted one), model, and speed — shows you a summary, and writes the
|
|
100
|
+
config file below. Unrecognised top-level keys already in that file are
|
|
101
|
+
carried through; comments and layout are not preserved. A file containing a
|
|
102
|
+
TOML table or array is left alone entirely, with a message saying to edit it
|
|
103
|
+
by hand. The wizard paints on the controlling terminal rather than on stdout,
|
|
104
|
+
so it still works under output-capturing wrappers like `op run`.
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
vocalize config
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Hotkeys: `↑`/`↓` or `k`/`j` move, `Enter` selects, `p` previews the
|
|
111
|
+
highlighted voice, `m` types a value by hand, `q` or `Esc` cancels without
|
|
112
|
+
writing anything.
|
|
113
|
+
|
|
114
|
+
| Setting | Flag | Env var | Config file key | Default |
|
|
115
|
+
|---|---|---|---|---|
|
|
116
|
+
| Voice ID | `--voice` | `VOCALIZE_VOICE` | `voice` | `21m00Tcm4TlvDq8ikWAM` ("Rachel") |
|
|
117
|
+
| Model ID | `--model` | `VOCALIZE_MODEL` | `model` | `eleven_multilingual_v2` |
|
|
118
|
+
| Speed | `--speed` | `VOCALIZE_SPEED` | `speed` | unset — the API's own 1.0 |
|
|
119
|
+
| Max characters | `--max-chars` | `VOCALIZE_MAX_CHARS` (hook only) | not read from the config file | unset on the CLI; 500 in the hook |
|
|
120
|
+
| Hook binary | — | `VOCALIZE_BIN` | not read from the config file | `vocalize` as found on `PATH` |
|
|
121
|
+
|
|
122
|
+
The config file is TOML at `$XDG_CONFIG_HOME/vocalize/config.toml`, falling
|
|
123
|
+
back to `~/.config/vocalize/config.toml`. Flat keys, no sections:
|
|
124
|
+
|
|
125
|
+
```toml
|
|
126
|
+
voice = "21m00Tcm4TlvDq8ikWAM"
|
|
127
|
+
model = "eleven_flash_v2_5"
|
|
128
|
+
speed = 0.95
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Not having a config file is normal and silent. A file that isn't valid TOML
|
|
132
|
+
is an error naming the file; a key that isn't recognised is a warning on
|
|
133
|
+
stderr, so a typo doesn't pass unnoticed but doesn't stop the run either.
|
|
134
|
+
`speed` must be a number between 0.7 and 1.2 — anything else is a one-line
|
|
135
|
+
error naming the source it came from.
|
|
136
|
+
|
|
137
|
+
The API key is separate and never read from this file: use `--api-key`,
|
|
138
|
+
`ELEVENLABS_API_KEY`, or a `.env` file.
|
|
87
139
|
|
|
88
140
|
## Claude Code integration
|
|
89
141
|
|
|
@@ -162,7 +214,7 @@ vocalize/
|
|
|
162
214
|
__init__.py # package version
|
|
163
215
|
__main__.py # python -m vocalize entry point
|
|
164
216
|
preprocess.py # markdown -> speakable text (pure function, fully unit tested)
|
|
165
|
-
config.py # API key resolution:
|
|
217
|
+
config.py # API key resolution + settings: flag > env > config.toml > default
|
|
166
218
|
exceptions.py # VocalizeError / TTSRequestError
|
|
167
219
|
tts.py # ElevenLabs API wrapper + disk cache (client is injected, so
|
|
168
220
|
# it's mockable in tests without hitting the network)
|
|
@@ -203,7 +255,8 @@ All tests run offline: the ElevenLabs client is dependency-injected into
|
|
|
203
255
|
only plays WAV, so the mp3 files this tool generates likely won't play
|
|
204
256
|
there. Use `--no-play` and open the saved file with whatever's on hand.
|
|
205
257
|
- **The disk cache under `~/.cache/vocalize` grows unbounded.** It's
|
|
206
|
-
content-addressed (keyed by a hash of text, voice, model,
|
|
258
|
+
content-addressed (keyed by a hash of text, voice, model, format, and
|
|
259
|
+
speed),
|
|
207
260
|
so it's always safe to delete some or all of it — nothing will break,
|
|
208
261
|
you'll just re-pay for a re-synthesized clip.
|
|
209
262
|
- **`--api-key` on the command line is visible to other local processes**
|
|
@@ -8,15 +8,17 @@ from vocalize.cli import main
|
|
|
8
8
|
def _patch_tts(monkeypatch, audio=b"fake-mp3-bytes"):
|
|
9
9
|
monkeypatch.setattr(cli_module, "build_client", lambda key: object())
|
|
10
10
|
captured_text = []
|
|
11
|
+
captured_settings = []
|
|
11
12
|
|
|
12
13
|
def fake_synthesize(client, text, settings):
|
|
13
14
|
captured_text.append(text)
|
|
15
|
+
captured_settings.append(settings)
|
|
14
16
|
return audio
|
|
15
17
|
|
|
16
18
|
monkeypatch.setattr(cli_module, "synthesize", fake_synthesize)
|
|
17
19
|
played = {}
|
|
18
20
|
monkeypatch.setattr(cli_module, "play_audio", lambda path: played.setdefault("path", path))
|
|
19
|
-
return played, captured_text
|
|
21
|
+
return played, captured_text, captured_settings
|
|
20
22
|
|
|
21
23
|
|
|
22
24
|
def test_speak_writes_audio_file(monkeypatch, tmp_path):
|
|
@@ -34,7 +36,7 @@ def test_speak_writes_audio_file(monkeypatch, tmp_path):
|
|
|
34
36
|
|
|
35
37
|
|
|
36
38
|
def test_speak_plays_by_default(monkeypatch, tmp_path):
|
|
37
|
-
played, _captured_text = _patch_tts(monkeypatch)
|
|
39
|
+
played, _captured_text, _captured_settings = _patch_tts(monkeypatch)
|
|
38
40
|
out_file = tmp_path / "out.mp3"
|
|
39
41
|
runner = CliRunner()
|
|
40
42
|
|
|
@@ -62,7 +64,7 @@ def test_speak_file_reads_from_stdin(monkeypatch, tmp_path):
|
|
|
62
64
|
|
|
63
65
|
|
|
64
66
|
def test_speak_file_reads_from_path(monkeypatch, tmp_path):
|
|
65
|
-
_played, captured_text = _patch_tts(monkeypatch)
|
|
67
|
+
_played, captured_text, _captured_settings = _patch_tts(monkeypatch)
|
|
66
68
|
src = tmp_path / "notes.md"
|
|
67
69
|
src.write_text("| a | b |\n|---|---|\n| 1 | 2 |\n")
|
|
68
70
|
out_file = tmp_path / "out.mp3"
|
|
@@ -80,7 +82,7 @@ def test_speak_file_reads_from_path(monkeypatch, tmp_path):
|
|
|
80
82
|
|
|
81
83
|
|
|
82
84
|
def test_raw_flag_skips_flattening(monkeypatch, tmp_path):
|
|
83
|
-
_played, captured_text = _patch_tts(monkeypatch)
|
|
85
|
+
_played, captured_text, _captured_settings = _patch_tts(monkeypatch)
|
|
84
86
|
src = tmp_path / "notes.md"
|
|
85
87
|
src.write_text("| a | b |\n|---|---|\n| 1 | 2 |\n")
|
|
86
88
|
out_file = tmp_path / "out.mp3"
|
|
@@ -100,7 +102,7 @@ def test_raw_flag_skips_flattening(monkeypatch, tmp_path):
|
|
|
100
102
|
|
|
101
103
|
|
|
102
104
|
def test_max_chars_truncates_and_notes(monkeypatch, tmp_path):
|
|
103
|
-
_played, captured_text = _patch_tts(monkeypatch)
|
|
105
|
+
_played, captured_text, _captured_settings = _patch_tts(monkeypatch)
|
|
104
106
|
out_file = tmp_path / "out.mp3"
|
|
105
107
|
runner = CliRunner()
|
|
106
108
|
|
|
@@ -212,3 +214,59 @@ def test_default_output_lands_in_cache_dir(monkeypatch, tmp_path):
|
|
|
212
214
|
|
|
213
215
|
assert result.exit_code == 0, result.output
|
|
214
216
|
assert (tmp_path / "last.mp3").exists()
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _isolate_config(monkeypatch, tmp_path):
|
|
220
|
+
monkeypatch.setenv("XDG_CONFIG_HOME", str(tmp_path / "empty-config"))
|
|
221
|
+
for var in ("VOCALIZE_VOICE", "VOCALIZE_MODEL", "VOCALIZE_SPEED"):
|
|
222
|
+
monkeypatch.delenv(var, raising=False)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def test_speed_flag_reaches_the_settings(monkeypatch, tmp_path):
|
|
226
|
+
_isolate_config(monkeypatch, tmp_path)
|
|
227
|
+
_played, _captured_text, captured_settings = _patch_tts(monkeypatch)
|
|
228
|
+
out_file = tmp_path / "out.mp3"
|
|
229
|
+
runner = CliRunner()
|
|
230
|
+
|
|
231
|
+
result = runner.invoke(
|
|
232
|
+
main,
|
|
233
|
+
[
|
|
234
|
+
"speak", "hello", "--api-key", "fake-key",
|
|
235
|
+
"--output", str(out_file), "--no-play", "--speed", "1.1",
|
|
236
|
+
],
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
assert result.exit_code == 0, result.output
|
|
240
|
+
assert captured_settings[0].speed == 1.1
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def test_no_speed_flag_leaves_speed_unset(monkeypatch, tmp_path):
|
|
244
|
+
_isolate_config(monkeypatch, tmp_path)
|
|
245
|
+
_played, _captured_text, captured_settings = _patch_tts(monkeypatch)
|
|
246
|
+
out_file = tmp_path / "out.mp3"
|
|
247
|
+
runner = CliRunner()
|
|
248
|
+
|
|
249
|
+
result = runner.invoke(
|
|
250
|
+
main,
|
|
251
|
+
["speak", "hello", "--api-key", "fake-key", "--output", str(out_file), "--no-play"],
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
assert result.exit_code == 0, result.output
|
|
255
|
+
assert captured_settings[0].speed is None
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def test_invalid_speed_gives_a_clean_error_not_a_traceback(monkeypatch, tmp_path, capsys):
|
|
259
|
+
_isolate_config(monkeypatch, tmp_path)
|
|
260
|
+
monkeypatch.setattr(
|
|
261
|
+
"sys.argv",
|
|
262
|
+
["vocalize", "speak", "hello", "--api-key", "fake-key", "--no-play", "--speed", "5"],
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
with pytest.raises(SystemExit) as excinfo:
|
|
266
|
+
cli_module.run()
|
|
267
|
+
|
|
268
|
+
assert excinfo.value.code == 1
|
|
269
|
+
captured = capsys.readouterr()
|
|
270
|
+
assert captured.err.startswith("Error: ")
|
|
271
|
+
assert "--speed" in captured.err
|
|
272
|
+
assert "Traceback" not in captured.err
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
import os
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from vocalize.config import (
|
|
6
|
+
DEFAULT_MODEL,
|
|
7
|
+
DEFAULT_VOICE,
|
|
8
|
+
_load_dotenv_if_present,
|
|
9
|
+
config_path,
|
|
10
|
+
resolve_api_key,
|
|
11
|
+
resolve_settings,
|
|
12
|
+
)
|
|
13
|
+
from vocalize.exceptions import ConfigError, MissingAPIKeyError
|
|
14
|
+
|
|
15
|
+
# Bound at import time on purpose: conftest's autouse fixture replaces the
|
|
16
|
+
# module attribute, so this reference is the only way to reach the real one.
|
|
17
|
+
real_load_dotenv = _load_dotenv_if_present
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def test_explicit_key_wins_over_everything(monkeypatch):
|
|
21
|
+
monkeypatch.setenv("ELEVENLABS_API_KEY", "env-key")
|
|
22
|
+
assert resolve_api_key("explicit-key") == "explicit-key"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def test_falls_back_to_env_var(monkeypatch):
|
|
26
|
+
monkeypatch.setenv("ELEVENLABS_API_KEY", "env-key")
|
|
27
|
+
assert resolve_api_key(None) == "env-key"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def test_raises_clear_error_when_nothing_found(monkeypatch):
|
|
31
|
+
monkeypatch.delenv("ELEVENLABS_API_KEY", raising=False)
|
|
32
|
+
with pytest.raises(MissingAPIKeyError):
|
|
33
|
+
resolve_api_key(None)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_dotenv_loader_reads_the_env_file_in_the_cwd(monkeypatch, tmp_path):
|
|
37
|
+
monkeypatch.chdir(tmp_path)
|
|
38
|
+
(tmp_path / ".env").write_text("ELEVENLABS_API_KEY=from-cwd-file\n", encoding="utf-8")
|
|
39
|
+
# setenv before delenv so monkeypatch restores the var whether or not it
|
|
40
|
+
# was set beforehand — the real loader writes straight to os.environ.
|
|
41
|
+
monkeypatch.setenv("ELEVENLABS_API_KEY", "placeholder")
|
|
42
|
+
monkeypatch.delenv("ELEVENLABS_API_KEY")
|
|
43
|
+
|
|
44
|
+
real_load_dotenv()
|
|
45
|
+
|
|
46
|
+
assert os.environ["ELEVENLABS_API_KEY"] == "from-cwd-file"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _isolate(monkeypatch, tmp_path, body=None):
|
|
50
|
+
"""Point the config loader at tmp_path and clear the setting env vars."""
|
|
51
|
+
monkeypatch.setenv("XDG_CONFIG_HOME", str(tmp_path))
|
|
52
|
+
for var in ("VOCALIZE_VOICE", "VOCALIZE_MODEL", "VOCALIZE_SPEED"):
|
|
53
|
+
monkeypatch.delenv(var, raising=False)
|
|
54
|
+
path = tmp_path / "vocalize" / "config.toml"
|
|
55
|
+
if body is not None:
|
|
56
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
57
|
+
path.write_text(body, encoding="utf-8")
|
|
58
|
+
return path
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def test_config_path_honours_xdg_config_home(monkeypatch, tmp_path):
|
|
62
|
+
monkeypatch.setenv("XDG_CONFIG_HOME", str(tmp_path))
|
|
63
|
+
assert config_path() == tmp_path / "vocalize" / "config.toml"
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def test_config_path_falls_back_to_dot_config(monkeypatch, tmp_path):
|
|
67
|
+
monkeypatch.delenv("XDG_CONFIG_HOME", raising=False)
|
|
68
|
+
monkeypatch.setattr("pathlib.Path.home", classmethod(lambda cls: tmp_path))
|
|
69
|
+
assert config_path() == tmp_path / ".config" / "vocalize" / "config.toml"
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def test_missing_config_file_gives_the_built_in_defaults(monkeypatch, tmp_path):
|
|
73
|
+
_isolate(monkeypatch, tmp_path)
|
|
74
|
+
|
|
75
|
+
settings = resolve_settings()
|
|
76
|
+
|
|
77
|
+
assert settings.voice_id == DEFAULT_VOICE
|
|
78
|
+
assert settings.model_id == DEFAULT_MODEL
|
|
79
|
+
assert settings.speed is None
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def test_config_file_beats_the_default(monkeypatch, tmp_path):
|
|
83
|
+
_isolate(monkeypatch, tmp_path, 'voice = "file-voice"\nmodel = "file-model"\nspeed = 0.9\n')
|
|
84
|
+
|
|
85
|
+
settings = resolve_settings()
|
|
86
|
+
|
|
87
|
+
assert settings.voice_id == "file-voice"
|
|
88
|
+
assert settings.model_id == "file-model"
|
|
89
|
+
assert settings.speed == 0.9
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def test_env_var_beats_the_config_file(monkeypatch, tmp_path):
|
|
93
|
+
_isolate(monkeypatch, tmp_path, 'voice = "file-voice"\nmodel = "file-model"\nspeed = 0.9\n')
|
|
94
|
+
monkeypatch.setenv("VOCALIZE_VOICE", "env-voice")
|
|
95
|
+
monkeypatch.setenv("VOCALIZE_MODEL", "env-model")
|
|
96
|
+
monkeypatch.setenv("VOCALIZE_SPEED", "1.1")
|
|
97
|
+
|
|
98
|
+
settings = resolve_settings()
|
|
99
|
+
|
|
100
|
+
assert settings.voice_id == "env-voice"
|
|
101
|
+
assert settings.model_id == "env-model"
|
|
102
|
+
assert settings.speed == 1.1
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def test_flag_beats_env_var_and_config_file(monkeypatch, tmp_path):
|
|
106
|
+
_isolate(monkeypatch, tmp_path, 'voice = "file-voice"\nmodel = "file-model"\nspeed = 0.9\n')
|
|
107
|
+
monkeypatch.setenv("VOCALIZE_VOICE", "env-voice")
|
|
108
|
+
monkeypatch.setenv("VOCALIZE_MODEL", "env-model")
|
|
109
|
+
monkeypatch.setenv("VOCALIZE_SPEED", "1.1")
|
|
110
|
+
|
|
111
|
+
settings = resolve_settings(voice_id="flag-voice", model_id="flag-model", speed=0.8)
|
|
112
|
+
|
|
113
|
+
assert settings.voice_id == "flag-voice"
|
|
114
|
+
assert settings.model_id == "flag-model"
|
|
115
|
+
assert settings.speed == 0.8
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def test_unknown_config_key_warns_but_still_loads_the_known_ones(monkeypatch, tmp_path, capsys):
|
|
119
|
+
_isolate(monkeypatch, tmp_path, 'voice = "file-voice"\nvoise = "typo"\n')
|
|
120
|
+
|
|
121
|
+
settings = resolve_settings()
|
|
122
|
+
|
|
123
|
+
assert settings.voice_id == "file-voice"
|
|
124
|
+
captured = capsys.readouterr()
|
|
125
|
+
assert "vocalize: unknown config key 'voise'" in captured.err
|
|
126
|
+
assert captured.err.count("unknown config key") == 1
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def test_malformed_config_file_gives_a_clean_error(monkeypatch, tmp_path):
|
|
130
|
+
path = _isolate(monkeypatch, tmp_path, "voice = \n")
|
|
131
|
+
|
|
132
|
+
with pytest.raises(ConfigError) as excinfo:
|
|
133
|
+
resolve_settings()
|
|
134
|
+
|
|
135
|
+
assert str(path) in str(excinfo.value)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
@pytest.mark.parametrize("value", ["fast", "0.2", "5"])
|
|
139
|
+
def test_invalid_speed_from_the_env_var_is_rejected(monkeypatch, tmp_path, value):
|
|
140
|
+
_isolate(monkeypatch, tmp_path)
|
|
141
|
+
monkeypatch.setenv("VOCALIZE_SPEED", value)
|
|
142
|
+
|
|
143
|
+
with pytest.raises(ConfigError) as excinfo:
|
|
144
|
+
resolve_settings()
|
|
145
|
+
|
|
146
|
+
assert "VOCALIZE_SPEED" in str(excinfo.value)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
@pytest.mark.parametrize("literal", ['"fast"', "0.2", "5"])
|
|
150
|
+
def test_invalid_speed_from_the_config_file_is_rejected(monkeypatch, tmp_path, literal):
|
|
151
|
+
path = _isolate(monkeypatch, tmp_path, f"speed = {literal}\n")
|
|
152
|
+
|
|
153
|
+
with pytest.raises(ConfigError) as excinfo:
|
|
154
|
+
resolve_settings()
|
|
155
|
+
|
|
156
|
+
assert "'speed'" in str(excinfo.value)
|
|
157
|
+
assert str(path) in str(excinfo.value)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
@pytest.mark.parametrize("value", ["fast", 0.2, 5])
|
|
161
|
+
def test_invalid_speed_from_the_flag_is_rejected(monkeypatch, tmp_path, value):
|
|
162
|
+
_isolate(monkeypatch, tmp_path)
|
|
163
|
+
|
|
164
|
+
with pytest.raises(ConfigError) as excinfo:
|
|
165
|
+
resolve_settings(speed=value)
|
|
166
|
+
|
|
167
|
+
assert "--speed" in str(excinfo.value)
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import hashlib
|
|
1
2
|
from pathlib import Path
|
|
2
3
|
from types import SimpleNamespace
|
|
3
4
|
|
|
@@ -5,7 +6,7 @@ import pytest
|
|
|
5
6
|
|
|
6
7
|
from vocalize.config import Settings
|
|
7
8
|
from vocalize.exceptions import TTSRequestError
|
|
8
|
-
from vocalize.tts import list_voices, synthesize
|
|
9
|
+
from vocalize.tts import _cache_key, list_voices, synthesize
|
|
9
10
|
|
|
10
11
|
|
|
11
12
|
class FakeTTSNamespace:
|
|
@@ -109,3 +110,37 @@ def test_list_voices_returns_id_and_name():
|
|
|
109
110
|
result = list_voices(client)
|
|
110
111
|
|
|
111
112
|
assert result == [{"id": "abc123", "name": "Rachel"}]
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def test_speed_is_passed_through_as_voice_settings(tmp_path):
|
|
116
|
+
client = FakeClient()
|
|
117
|
+
settings = Settings(voice_id="v1", model_id="m1", speed=1.1)
|
|
118
|
+
|
|
119
|
+
synthesize(client, "hi", settings, cache_dir=tmp_path)
|
|
120
|
+
|
|
121
|
+
assert client.text_to_speech.calls[0]["voice_settings"].speed == 1.1
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def test_unset_speed_sends_no_voice_settings_kwarg(tmp_path):
|
|
125
|
+
client = FakeClient()
|
|
126
|
+
settings = Settings(voice_id="v1", model_id="m1")
|
|
127
|
+
|
|
128
|
+
synthesize(client, "hi", settings, cache_dir=tmp_path)
|
|
129
|
+
|
|
130
|
+
assert "voice_settings" not in client.text_to_speech.calls[0]
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def test_cache_key_differs_by_speed():
|
|
134
|
+
unset = Settings(voice_id="v1", model_id="m1")
|
|
135
|
+
faster = Settings(voice_id="v1", model_id="m1", speed=1.1)
|
|
136
|
+
|
|
137
|
+
assert _cache_key("hi", unset) != _cache_key("hi", faster)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def test_cache_key_is_unchanged_when_speed_is_unset():
|
|
141
|
+
# Pins the pre-speed payload scheme so caches written by older
|
|
142
|
+
# versions keep hitting.
|
|
143
|
+
settings = Settings(voice_id="v1", model_id="m1", output_format="f1")
|
|
144
|
+
old_payload = f"{settings.voice_id}|{settings.model_id}|{settings.output_format}|hi"
|
|
145
|
+
|
|
146
|
+
assert _cache_key("hi", settings) == hashlib.sha256(old_payload.encode("utf-8")).hexdigest()
|