speak-cli 1.2.0__tar.gz → 1.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {speak_cli-1.2.0 → speak_cli-1.3.0}/PKG-INFO +18 -1
- {speak_cli-1.2.0 → speak_cli-1.3.0}/README.md +17 -0
- speak_cli-1.3.0/src/speak_cli/__init__.py +20 -0
- speak_cli-1.3.0/src/speak_cli/api.py +124 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/src/speak_cli/cli.py +2 -23
- {speak_cli-1.2.0 → speak_cli-1.3.0}/src/speak_cli/playback.py +30 -0
- speak_cli-1.3.0/tests/test_api.py +57 -0
- speak_cli-1.2.0/src/speak_cli/__init__.py +0 -6
- {speak_cli-1.2.0 → speak_cli-1.3.0}/.github/workflows/ci.yml +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/.github/workflows/publish.yml +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/.gitignore +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/.python-version +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/pyproject.toml +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/src/speak_cli/chunker.py +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/src/speak_cli/config.py +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/src/speak_cli/daemon.py +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/src/speak_cli/engine.py +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/src/speak_cli/ipc.py +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/src/speak_cli/langdetect.py +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/src/speak_cli/voices.py +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/tests/test_chunker.py +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/tests/test_engine_filter.py +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/tests/test_ipc_paths.py +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/tests/test_langdetect.py +0 -0
- {speak_cli-1.2.0 → speak_cli-1.3.0}/tests/test_voices.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: speak-cli
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.3.0
|
|
4
4
|
Summary: Speak text out loud from the command line using Supertonic 3 (local, offline TTS)
|
|
5
5
|
Project-URL: Repository, https://github.com/MohamedAliRashad/tts-cli
|
|
6
6
|
Project-URL: Issues, https://github.com/MohamedAliRashad/tts-cli/issues
|
|
@@ -51,6 +51,23 @@ ollama run llama3 "tell a story" | speak --live # voice for your LLM
|
|
|
51
51
|
```
|
|
52
52
|
**Note:** `say` also works as an alias.
|
|
53
53
|
|
|
54
|
+
## Use from Python
|
|
55
|
+
|
|
56
|
+
The same engine is available as a library — add `speak-cli` to your project (`uv add speak-cli` / `pip install speak-cli`) and:
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
from speak_cli import Speaker, say
|
|
60
|
+
|
|
61
|
+
say("hello world") # one-liner: synthesize and play
|
|
62
|
+
|
|
63
|
+
speaker = Speaker(voice="noah", speed=1.1) # model loads once, reuse it
|
|
64
|
+
speaker.say("long texts stream, so they start speaking instantly")
|
|
65
|
+
wav_bytes = speaker.synthesize("raw 44.1 kHz WAV bytes")
|
|
66
|
+
speaker.save("مرحبا بالعالم", "clip.wav") # language auto-detected per call
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Per-call overrides work everywhere: `speaker.say("bonjour", lang="fr", speed=0.9)`.
|
|
70
|
+
|
|
54
71
|
## Voices
|
|
55
72
|
|
|
56
73
|
Ten voices, picked by name (`speak --list-voices`):
|
|
@@ -28,6 +28,23 @@ ollama run llama3 "tell a story" | speak --live # voice for your LLM
|
|
|
28
28
|
```
|
|
29
29
|
**Note:** `say` also works as an alias.
|
|
30
30
|
|
|
31
|
+
## Use from Python
|
|
32
|
+
|
|
33
|
+
The same engine is available as a library — add `speak-cli` to your project (`uv add speak-cli` / `pip install speak-cli`) and:
|
|
34
|
+
|
|
35
|
+
```python
|
|
36
|
+
from speak_cli import Speaker, say
|
|
37
|
+
|
|
38
|
+
say("hello world") # one-liner: synthesize and play
|
|
39
|
+
|
|
40
|
+
speaker = Speaker(voice="noah", speed=1.1) # model loads once, reuse it
|
|
41
|
+
speaker.say("long texts stream, so they start speaking instantly")
|
|
42
|
+
wav_bytes = speaker.synthesize("raw 44.1 kHz WAV bytes")
|
|
43
|
+
speaker.save("مرحبا بالعالم", "clip.wav") # language auto-detected per call
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Per-call overrides work everywhere: `speaker.say("bonjour", lang="fr", speed=0.9)`.
|
|
47
|
+
|
|
31
48
|
## Voices
|
|
32
49
|
|
|
33
50
|
Ten voices, picked by name (`speak --list-voices`):
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
2
|
+
|
|
3
|
+
try:
|
|
4
|
+
__version__ = version("speak-cli")
|
|
5
|
+
except PackageNotFoundError: # running from a source tree without install
|
|
6
|
+
__version__ = "0.0.0"
|
|
7
|
+
|
|
8
|
+
__all__ = ["Speaker", "say", "synthesize", "save", "list_voices", "__version__"]
|
|
9
|
+
|
|
10
|
+
_API_NAMES = {"Speaker", "say", "synthesize", "save", "list_voices"}
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def __getattr__(name: str):
|
|
14
|
+
# PEP 562 lazy exports: `import speak_cli` stays instant; the API module
|
|
15
|
+
# (and eventually onnxruntime) loads only when actually used.
|
|
16
|
+
if name in _API_NAMES:
|
|
17
|
+
from . import api
|
|
18
|
+
|
|
19
|
+
return getattr(api, name)
|
|
20
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""Public Python API for speak-cli.
|
|
2
|
+
|
|
3
|
+
from speak_cli import Speaker, say
|
|
4
|
+
|
|
5
|
+
say("hello world") # one-liner: synthesize and play
|
|
6
|
+
|
|
7
|
+
s = Speaker(voice="noah", speed=1.1) # reusable: model loads once
|
|
8
|
+
s.say("fast repeat calls")
|
|
9
|
+
wav = s.synthesize("raw 44.1kHz WAV bytes")
|
|
10
|
+
s.save("save to disk", "clip.wav")
|
|
11
|
+
|
|
12
|
+
Everything runs in-process (no daemon); create one Speaker and reuse it to
|
|
13
|
+
keep the models warm. All heavy imports happen on first synthesis, so
|
|
14
|
+
importing this module is instant.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from .langdetect import SUPPORTED_LANGS, detect_lang
|
|
22
|
+
from .voices import VOICES, resolve
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def list_voices() -> list[str]:
|
|
26
|
+
"""Names of the available voices."""
|
|
27
|
+
return list(VOICES)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class Speaker:
|
|
31
|
+
"""Reusable text-to-speech engine with a fixed default voice and pace.
|
|
32
|
+
|
|
33
|
+
Per-call keyword arguments override the constructor defaults.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
def __init__(self, voice: str = "sara", speed: float = 1.0,
|
|
37
|
+
steps: int = 8, lang: str = "auto") -> None:
|
|
38
|
+
self.voice = resolve(voice)[0] # fail fast on unknown names
|
|
39
|
+
self.speed = speed
|
|
40
|
+
self.steps = steps
|
|
41
|
+
self.lang = lang
|
|
42
|
+
self._engine = None
|
|
43
|
+
|
|
44
|
+
def _resolve(self, text: str, voice: str | None, speed: float | None,
|
|
45
|
+
lang: str | None, steps: int | None) -> dict:
|
|
46
|
+
from .chunker import clean
|
|
47
|
+
|
|
48
|
+
text = clean(text).strip()
|
|
49
|
+
if not text:
|
|
50
|
+
raise ValueError("no text to speak")
|
|
51
|
+
lang = (lang or self.lang).lower()
|
|
52
|
+
if lang == "auto":
|
|
53
|
+
lang = detect_lang(text)
|
|
54
|
+
if lang not in SUPPORTED_LANGS:
|
|
55
|
+
raise ValueError(
|
|
56
|
+
f"unsupported language {lang!r}; supported: "
|
|
57
|
+
+ " ".join(sorted(SUPPORTED_LANGS))
|
|
58
|
+
)
|
|
59
|
+
return {
|
|
60
|
+
"text": text,
|
|
61
|
+
"voice": resolve(voice)[1] if voice else resolve(self.voice)[1],
|
|
62
|
+
"speed": speed if speed is not None else self.speed,
|
|
63
|
+
"lang": lang,
|
|
64
|
+
"steps": steps if steps is not None else self.steps,
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
def _synth(self, req: dict) -> bytes:
|
|
68
|
+
if self._engine is None:
|
|
69
|
+
from .engine import Engine
|
|
70
|
+
|
|
71
|
+
self._engine = Engine()
|
|
72
|
+
wav, _ = self._engine.synthesize(**req)
|
|
73
|
+
return wav
|
|
74
|
+
|
|
75
|
+
def synthesize(self, text: str, *, voice: str | None = None,
|
|
76
|
+
speed: float | None = None, lang: str | None = None,
|
|
77
|
+
steps: int | None = None) -> bytes:
|
|
78
|
+
"""Synthesize speech; returns a complete 44.1kHz 16-bit WAV as bytes."""
|
|
79
|
+
return self._synth(self._resolve(text, voice, speed, lang, steps))
|
|
80
|
+
|
|
81
|
+
def say(self, text: str, *, voice: str | None = None,
|
|
82
|
+
speed: float | None = None, lang: str | None = None,
|
|
83
|
+
steps: int | None = None) -> None:
|
|
84
|
+
"""Speak text through the speakers, streaming so long texts start fast."""
|
|
85
|
+
from .chunker import chunks
|
|
86
|
+
from .playback import stream_play
|
|
87
|
+
|
|
88
|
+
req = self._resolve(text, voice, speed, lang, steps)
|
|
89
|
+
pieces = ((p, req["lang"]) for p in chunks(req["text"]))
|
|
90
|
+
stream_play(
|
|
91
|
+
pieces,
|
|
92
|
+
lambda t, lang: self._synth({**req, "text": t, "lang": lang}),
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
def save(self, text: str, path: str | Path, **kwargs) -> Path:
|
|
96
|
+
"""Synthesize speech and write it to a WAV file; returns the path."""
|
|
97
|
+
path = Path(path)
|
|
98
|
+
path.write_bytes(self.synthesize(text, **kwargs))
|
|
99
|
+
return path
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
_default_speaker: Speaker | None = None
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _default() -> Speaker:
|
|
106
|
+
global _default_speaker
|
|
107
|
+
if _default_speaker is None:
|
|
108
|
+
_default_speaker = Speaker()
|
|
109
|
+
return _default_speaker
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def say(text: str, **kwargs) -> None:
|
|
113
|
+
"""Speak text out loud using a shared default Speaker."""
|
|
114
|
+
_default().say(text, **kwargs)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def synthesize(text: str, **kwargs) -> bytes:
|
|
118
|
+
"""Return speech for `text` as WAV bytes, using a shared default Speaker."""
|
|
119
|
+
return _default().synthesize(text, **kwargs)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def save(text: str, path: str | Path, **kwargs) -> Path:
|
|
123
|
+
"""Write speech for `text` to a WAV file, using a shared default Speaker."""
|
|
124
|
+
return _default().save(text, path, **kwargs)
|
|
@@ -3,9 +3,7 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import argparse
|
|
6
|
-
import queue
|
|
7
6
|
import sys
|
|
8
|
-
import threading
|
|
9
7
|
from collections.abc import Iterable
|
|
10
8
|
|
|
11
9
|
from . import __version__, voices
|
|
@@ -122,28 +120,9 @@ class Synth:
|
|
|
122
120
|
|
|
123
121
|
def _stream_speak(pieces: Iterable[tuple[str, str | None]], synth: Synth,
|
|
124
122
|
verbose: bool) -> None:
|
|
125
|
-
|
|
126
|
-
from .playback import play
|
|
123
|
+
from .playback import stream_play
|
|
127
124
|
|
|
128
|
-
|
|
129
|
-
errors: list[Exception] = []
|
|
130
|
-
|
|
131
|
-
def produce() -> None:
|
|
132
|
-
try:
|
|
133
|
-
for text, lang in pieces:
|
|
134
|
-
q.put(synth(text, lang))
|
|
135
|
-
except Exception as e:
|
|
136
|
-
errors.append(e)
|
|
137
|
-
finally:
|
|
138
|
-
q.put(None)
|
|
139
|
-
|
|
140
|
-
t = threading.Thread(target=produce, daemon=True)
|
|
141
|
-
t.start()
|
|
142
|
-
while (wav := q.get()) is not None:
|
|
143
|
-
play(wav, verbose=verbose)
|
|
144
|
-
t.join()
|
|
145
|
-
if errors:
|
|
146
|
-
raise errors[0]
|
|
125
|
+
stream_play(pieces, synth, verbose=verbose)
|
|
147
126
|
|
|
148
127
|
|
|
149
128
|
def main(argv: list[str] | None = None) -> int:
|
|
@@ -74,6 +74,36 @@ def _play_players(path: str, errors: list[str]) -> str | None:
|
|
|
74
74
|
return None
|
|
75
75
|
|
|
76
76
|
|
|
77
|
+
def stream_play(pieces, synth_fn, verbose: bool = False) -> None:
|
|
78
|
+
"""Play piece N while piece N+1 synthesizes.
|
|
79
|
+
|
|
80
|
+
`pieces` yields (text, lang) tuples; `synth_fn(text, lang)` returns WAV
|
|
81
|
+
bytes. Used by both the CLI and the library API for instant-start speech.
|
|
82
|
+
"""
|
|
83
|
+
import queue
|
|
84
|
+
import threading
|
|
85
|
+
|
|
86
|
+
q: queue.Queue = queue.Queue(maxsize=3)
|
|
87
|
+
errors: list[Exception] = []
|
|
88
|
+
|
|
89
|
+
def produce() -> None:
|
|
90
|
+
try:
|
|
91
|
+
for text, lang in pieces:
|
|
92
|
+
q.put(synth_fn(text, lang))
|
|
93
|
+
except Exception as e:
|
|
94
|
+
errors.append(e)
|
|
95
|
+
finally:
|
|
96
|
+
q.put(None)
|
|
97
|
+
|
|
98
|
+
t = threading.Thread(target=produce, daemon=True)
|
|
99
|
+
t.start()
|
|
100
|
+
while (wav := q.get()) is not None:
|
|
101
|
+
play(wav, verbose=verbose)
|
|
102
|
+
t.join()
|
|
103
|
+
if errors:
|
|
104
|
+
raise errors[0]
|
|
105
|
+
|
|
106
|
+
|
|
77
107
|
def play(wav_bytes: bytes, verbose: bool = False) -> None:
|
|
78
108
|
"""Play WAV bytes; raises PlaybackError if nothing works."""
|
|
79
109
|
errors: list[str] = []
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
import pytest
|
|
2
|
+
|
|
3
|
+
import speak_cli
|
|
4
|
+
from speak_cli.api import Speaker, list_voices
|
|
5
|
+
from speak_cli.engine import models_cached
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def test_lazy_exports():
|
|
9
|
+
assert callable(speak_cli.say)
|
|
10
|
+
assert callable(speak_cli.synthesize)
|
|
11
|
+
assert speak_cli.Speaker is Speaker
|
|
12
|
+
with pytest.raises(AttributeError):
|
|
13
|
+
speak_cli.nonexistent
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def test_list_voices():
|
|
17
|
+
voices = list_voices()
|
|
18
|
+
assert "sara" in voices and "noah" in voices
|
|
19
|
+
assert len(voices) == 10
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def test_unknown_voice_fails_fast():
|
|
23
|
+
with pytest.raises(ValueError, match="unknown voice"):
|
|
24
|
+
Speaker(voice="bogus")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def test_empty_text_rejected():
|
|
28
|
+
with pytest.raises(ValueError, match="no text"):
|
|
29
|
+
Speaker()._resolve(" \x1b[2K ", None, None, None, None)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def test_unsupported_lang_rejected():
|
|
33
|
+
with pytest.raises(ValueError, match="unsupported language"):
|
|
34
|
+
Speaker()._resolve("hi", None, None, "xx", None)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_resolve_defaults_and_overrides():
|
|
38
|
+
s = Speaker(voice="noah", speed=1.2)
|
|
39
|
+
req = s._resolve("hello there", None, None, None, None)
|
|
40
|
+
assert req == {"text": "hello there", "voice": "M5", "speed": 1.2,
|
|
41
|
+
"lang": "en", "steps": 8}
|
|
42
|
+
req = s._resolve("مرحبا", "emma", 0.9, None, 6)
|
|
43
|
+
assert req["voice"] == "F2" and req["speed"] == 0.9
|
|
44
|
+
assert req["lang"] == "ar" and req["steps"] == 6
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@pytest.mark.skipif(not models_cached(), reason="model assets not downloaded")
|
|
48
|
+
def test_synthesize_returns_wav():
|
|
49
|
+
wav = Speaker().synthesize("library api check", steps=2)
|
|
50
|
+
assert wav[:4] == b"RIFF" and wav[8:12] == b"WAVE"
|
|
51
|
+
assert len(wav) > 10_000
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@pytest.mark.skipif(not models_cached(), reason="model assets not downloaded")
|
|
55
|
+
def test_save_writes_file(tmp_path):
|
|
56
|
+
out = speak_cli.save("saved from the api", tmp_path / "clip.wav", steps=2)
|
|
57
|
+
assert out.exists() and out.stat().st_size > 10_000
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|