wfloat 0.0.1__tar.gz → 1.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,2 @@
1
+ include README.md
2
+ recursive-include python *.py
wfloat-1.0.1/PKG-INFO ADDED
@@ -0,0 +1,172 @@
1
+ Metadata-Version: 2.4
2
+ Name: wfloat
3
+ Version: 1.0.1
4
+ Summary: High-level Python wrapper for Wfloat TTS
5
+ Home-page: https://github.com/wfloat/wfloat-python
6
+ Author: wfloat
7
+ License: MIT
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: Operating System :: Microsoft :: Windows
10
+ Classifier: Operating System :: POSIX :: Linux
11
+ Classifier: Operating System :: MacOS :: MacOS X
12
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
13
+ Requires-Python: >=3.9
14
+ Description-Content-Type: text/markdown
15
+ Requires-Dist: wfloat-sherpa-onnx==1.12.24
16
+ Dynamic: author
17
+ Dynamic: classifier
18
+ Dynamic: description
19
+ Dynamic: description-content-type
20
+ Dynamic: home-page
21
+ Dynamic: license
22
+ Dynamic: requires-dist
23
+ Dynamic: requires-python
24
+ Dynamic: summary
25
+
26
+ # wfloat
27
+
28
+ `wfloat` is the Python package for `wfloat-tts`, Wfloat's on-device English
29
+ text-to-speech model.
30
+
31
+ It runs speech locally in Python instead of calling a hosted inference API.
32
+ The model supports 20 voices with emotion and intensity control.
33
+
34
+ If you're building for the browser, use
35
+ [`@wfloat/wfloat-web`](https://github.com/wfloat/wfloat-web). If you're
36
+ building for React Native, use
37
+ [`@wfloat/react-native-wfloat`](https://github.com/wfloat/react-native-wfloat).
38
+
39
+ Try it in the browser: https://wfloat.com/demo
40
+
41
+ <audio controls src="./sample.wav">
42
+ <a href="./sample.wav">Sample dialogue</a>
43
+ </audio>
44
+
45
+ [Sample dialogue](sample.wav)
46
+
47
+ ## Install
48
+
49
+ ```bash
50
+ pip install wfloat
51
+ ```
52
+
53
+ ## Usage
54
+
55
+ ```python
56
+ import wfloat
57
+
58
+ model = wfloat.load("wfloat/wfloat-tts")
59
+
60
+ result = model.generate(
61
+ text="No, no, that's not possible. The formula should have crystallized, but it adapted instead. Do you realize what that means for the rest of my work?",
62
+ voice_id="mad_scientist_woman",
63
+ emotion="surprise",
64
+ intensity=0.7,
65
+ )
66
+
67
+ result.audio.save("out.wav")
68
+ ```
69
+
70
+ For multi-speaker dialogue:
71
+
72
+ ```python
73
+ import wfloat
74
+
75
+ model = wfloat.load("wfloat/wfloat-tts")
76
+
77
+ result = model.generate_dialogue(
78
+ segments=[
79
+ {
80
+ "voice_id": "wise_elder_man",
81
+ "text": "Rain taps against the tavern shutters as you step inside.",
82
+ "emotion": "neutral",
83
+ "intensity": 0.5,
84
+ },
85
+ {
86
+ "voice_id": "strong_hero_man",
87
+ "text": "You're late. Two bandits stole the king's map over three hours ago.",
88
+ "emotion": "fear",
89
+ "intensity": 0.6,
90
+ },
91
+ {
92
+ "voice_id": "strong_hero_man",
93
+ "text": "They fled north, up into the woods.",
94
+ "emotion": "neutral",
95
+ "intensity": 0.5,
96
+ },
97
+ ],
98
+ silence_between_segments_sec=0.35,
99
+ )
100
+
101
+ result.audio.save("dialogue.wav")
102
+ ```
103
+
104
+ You can also generate a WAV from the command line:
105
+
106
+ ```bash
107
+ wfloat generate \
108
+ --text "Hello world!" \
109
+ --out out.wav \
110
+ --voice-id mad_scientist_woman \
111
+ --emotion surprise \
112
+ --intensity 0.7 \
113
+ --silence-padding-sec 0
114
+ ```
115
+
116
+ For the full CLI help:
117
+
118
+ ```bash
119
+ wfloat generate --help
120
+ ```
121
+
122
+ The first load downloads the model assets. After that, the package uses the
123
+ cached local copy.
124
+
125
+ ## Speaker IDs
126
+
127
+ Use `voice_id` string names or numeric `sid` values:
128
+
129
+ | Speaker | SID |
130
+ | --- | ---: |
131
+ | `skilled_hero_man` | 0 |
132
+ | `skilled_hero_woman` | 1 |
133
+ | `fun_hero_man` | 2 |
134
+ | `fun_hero_woman` | 3 |
135
+ | `strong_hero_man` | 4 |
136
+ | `strong_hero_woman` | 5 |
137
+ | `mad_scientist_man` | 6 |
138
+ | `mad_scientist_woman` | 7 |
139
+ | `clever_villain_man` | 8 |
140
+ | `clever_villain_woman` | 9 |
141
+ | `narrator_man` | 10 |
142
+ | `narrator_woman` | 11 |
143
+ | `wise_elder_man` | 12 |
144
+ | `wise_elder_woman` | 13 |
145
+ | `outgoing_anime_man` | 14 |
146
+ | `outgoing_anime_woman` | 15 |
147
+ | `scary_villain_man` | 16 |
148
+ | `scary_villain_woman` | 17 |
149
+ | `news_reporter_man` | 18 |
150
+ | `news_reporter_woman` | 19 |
151
+
152
+ ## Emotions
153
+
154
+ Supported emotion labels:
155
+
156
+ - `neutral`
157
+ - `joy`
158
+ - `sadness`
159
+ - `anger`
160
+ - `fear`
161
+ - `surprise`
162
+ - `dismissive`
163
+ - `confusion`
164
+
165
+ `intensity` must be between `0.0` and `1.0`.
166
+
167
+ ## More
168
+
169
+ - Docs: https://docs.wfloat.com
170
+ - Model card, voices, emotions, and samples: https://huggingface.co/Wfloat/wfloat-tts
171
+ - Web package: https://github.com/wfloat/wfloat-web
172
+ - React Native package: https://github.com/wfloat/react-native-wfloat
wfloat-1.0.1/README.md ADDED
@@ -0,0 +1,147 @@
1
+ # wfloat
2
+
3
+ `wfloat` is the Python package for `wfloat-tts`, Wfloat's on-device English
4
+ text-to-speech model.
5
+
6
+ It runs speech locally in Python instead of calling a hosted inference API.
7
+ The model supports 20 voices with emotion and intensity control.
8
+
9
+ If you're building for the browser, use
10
+ [`@wfloat/wfloat-web`](https://github.com/wfloat/wfloat-web). If you're
11
+ building for React Native, use
12
+ [`@wfloat/react-native-wfloat`](https://github.com/wfloat/react-native-wfloat).
13
+
14
+ Try it in the browser: https://wfloat.com/demo
15
+
16
+ <audio controls src="./sample.wav">
17
+ <a href="./sample.wav">Sample dialogue</a>
18
+ </audio>
19
+
20
+ [Sample dialogue](sample.wav)
21
+
22
+ ## Install
23
+
24
+ ```bash
25
+ pip install wfloat
26
+ ```
27
+
28
+ ## Usage
29
+
30
+ ```python
31
+ import wfloat
32
+
33
+ model = wfloat.load("wfloat/wfloat-tts")
34
+
35
+ result = model.generate(
36
+ text="No, no, that's not possible. The formula should have crystallized, but it adapted instead. Do you realize what that means for the rest of my work?",
37
+ voice_id="mad_scientist_woman",
38
+ emotion="surprise",
39
+ intensity=0.7,
40
+ )
41
+
42
+ result.audio.save("out.wav")
43
+ ```
44
+
45
+ For multi-speaker dialogue:
46
+
47
+ ```python
48
+ import wfloat
49
+
50
+ model = wfloat.load("wfloat/wfloat-tts")
51
+
52
+ result = model.generate_dialogue(
53
+ segments=[
54
+ {
55
+ "voice_id": "wise_elder_man",
56
+ "text": "Rain taps against the tavern shutters as you step inside.",
57
+ "emotion": "neutral",
58
+ "intensity": 0.5,
59
+ },
60
+ {
61
+ "voice_id": "strong_hero_man",
62
+ "text": "You're late. Two bandits stole the king's map over three hours ago.",
63
+ "emotion": "fear",
64
+ "intensity": 0.6,
65
+ },
66
+ {
67
+ "voice_id": "strong_hero_man",
68
+ "text": "They fled north, up into the woods.",
69
+ "emotion": "neutral",
70
+ "intensity": 0.5,
71
+ },
72
+ ],
73
+ silence_between_segments_sec=0.35,
74
+ )
75
+
76
+ result.audio.save("dialogue.wav")
77
+ ```
78
+
79
+ You can also generate a WAV from the command line:
80
+
81
+ ```bash
82
+ wfloat generate \
83
+ --text "Hello world!" \
84
+ --out out.wav \
85
+ --voice-id mad_scientist_woman \
86
+ --emotion surprise \
87
+ --intensity 0.7 \
88
+ --silence-padding-sec 0
89
+ ```
90
+
91
+ For the full CLI help:
92
+
93
+ ```bash
94
+ wfloat generate --help
95
+ ```
96
+
97
+ The first load downloads the model assets. After that, the package uses the
98
+ cached local copy.
99
+
100
+ ## Speaker IDs
101
+
102
+ Use `voice_id` string names or numeric `sid` values:
103
+
104
+ | Speaker | SID |
105
+ | --- | ---: |
106
+ | `skilled_hero_man` | 0 |
107
+ | `skilled_hero_woman` | 1 |
108
+ | `fun_hero_man` | 2 |
109
+ | `fun_hero_woman` | 3 |
110
+ | `strong_hero_man` | 4 |
111
+ | `strong_hero_woman` | 5 |
112
+ | `mad_scientist_man` | 6 |
113
+ | `mad_scientist_woman` | 7 |
114
+ | `clever_villain_man` | 8 |
115
+ | `clever_villain_woman` | 9 |
116
+ | `narrator_man` | 10 |
117
+ | `narrator_woman` | 11 |
118
+ | `wise_elder_man` | 12 |
119
+ | `wise_elder_woman` | 13 |
120
+ | `outgoing_anime_man` | 14 |
121
+ | `outgoing_anime_woman` | 15 |
122
+ | `scary_villain_man` | 16 |
123
+ | `scary_villain_woman` | 17 |
124
+ | `news_reporter_man` | 18 |
125
+ | `news_reporter_woman` | 19 |
126
+
127
+ ## Emotions
128
+
129
+ Supported emotion labels:
130
+
131
+ - `neutral`
132
+ - `joy`
133
+ - `sadness`
134
+ - `anger`
135
+ - `fear`
136
+ - `surprise`
137
+ - `dismissive`
138
+ - `confusion`
139
+
140
+ `intensity` must be between `0.0` and `1.0`.
141
+
142
+ ## More
143
+
144
+ - Docs: https://docs.wfloat.com
145
+ - Model card, voices, emotions, and samples: https://huggingface.co/Wfloat/wfloat-tts
146
+ - Web package: https://github.com/wfloat/wfloat-web
147
+ - React Native package: https://github.com/wfloat/react-native-wfloat
@@ -0,0 +1,3 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68", "wheel"]
3
+ build-backend = "setuptools.build_meta"
@@ -0,0 +1,47 @@
1
+ from ._constants import SPEAKER_IDS, VALID_EMOTIONS, VALID_SIDS
2
+ from ._model import Model, load
3
+ from ._results import Audio, GenerationResult, Timeline, TimelineChunk
4
+ from ._version import __version__
5
+
6
+ _LOW_LEVEL_EXPORTS = {
7
+ "GenerationConfig",
8
+ "OfflineTts",
9
+ "OfflineTtsConfig",
10
+ "OfflineTtsModelConfig",
11
+ "OfflineTtsWfloatModelConfig",
12
+ "WfloatPreparedText",
13
+ "git_date",
14
+ "git_sha1",
15
+ "prepare_wfloat_text",
16
+ "version",
17
+ "write_wave",
18
+ }
19
+
20
+
21
+ __all__ = [
22
+ "Audio",
23
+ "GenerationResult",
24
+ "Model",
25
+ "SPEAKER_IDS",
26
+ "Timeline",
27
+ "TimelineChunk",
28
+ "VALID_EMOTIONS",
29
+ "VALID_SIDS",
30
+ "load",
31
+ ]
32
+ __all__.extend(sorted(_LOW_LEVEL_EXPORTS))
33
+
34
+
35
+ def __getattr__(name):
36
+ if name not in _LOW_LEVEL_EXPORTS:
37
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
38
+
39
+ from . import _bindings
40
+
41
+ value = getattr(_bindings, name)
42
+ globals()[name] = value
43
+ return value
44
+
45
+
46
+ def __dir__():
47
+ return sorted(set(globals()) | _LOW_LEVEL_EXPORTS)
@@ -0,0 +1,5 @@
1
+ from ._cli import main
2
+
3
+
4
+ if __name__ == "__main__": # pragma: no cover
5
+ raise SystemExit(main())
@@ -0,0 +1,124 @@
1
+ import json
2
+ import os
3
+ from dataclasses import dataclass
4
+ from pathlib import Path
5
+ from typing import Dict, Optional
6
+ from urllib.parse import urlencode, urlparse
7
+ from urllib.request import Request, urlopen
8
+
9
+ from ._version import __version__ as PACKAGE_VERSION
10
+
11
+
12
+ DEFAULT_MODEL_ASSET_HOST = "https://wfloat.com"
13
+ DEFAULT_MODEL_ASSET_PATH = "/api/model-assets"
14
+
15
+
16
+ @dataclass(frozen=True)
17
+ class ModelAssets:
18
+ model_onnx: str
19
+ model_onnx_checksum: str
20
+ model_tokens: str
21
+ model_tokens_checksum: str
22
+ espeak_data: str
23
+ espeak_checksum: str
24
+ persistent_id: Optional[str] = None
25
+
26
+ @classmethod
27
+ def from_dict(cls, data: Dict[str, object]) -> "ModelAssets":
28
+ required_fields = (
29
+ "model_onnx",
30
+ "model_onnx_checksum",
31
+ "model_tokens",
32
+ "model_tokens_checksum",
33
+ "espeak_data",
34
+ "espeak_checksum",
35
+ )
36
+
37
+ missing = [
38
+ field_name
39
+ for field_name in required_fields
40
+ if not isinstance(data.get(field_name), str) or not str(data.get(field_name)).strip()
41
+ ]
42
+ if missing:
43
+ raise ValueError(
44
+ "Model asset response is missing required fields: %s"
45
+ % ", ".join(missing)
46
+ )
47
+
48
+ return cls(
49
+ model_onnx=str(data["model_onnx"]),
50
+ model_onnx_checksum=str(data["model_onnx_checksum"]),
51
+ model_tokens=str(data["model_tokens"]),
52
+ model_tokens_checksum=str(data["model_tokens_checksum"]),
53
+ espeak_data=str(data["espeak_data"]),
54
+ espeak_checksum=str(data["espeak_checksum"]),
55
+ persistent_id=str(data["persistent_id"]).strip()
56
+ if isinstance(data.get("persistent_id"), str) and str(data.get("persistent_id")).strip()
57
+ else None,
58
+ )
59
+
60
+ def to_dict(self) -> Dict[str, str]:
61
+ return {
62
+ "model_onnx": self.model_onnx,
63
+ "model_onnx_checksum": self.model_onnx_checksum,
64
+ "model_tokens": self.model_tokens,
65
+ "model_tokens_checksum": self.model_tokens_checksum,
66
+ "espeak_data": self.espeak_data,
67
+ "espeak_checksum": self.espeak_checksum,
68
+ **({"persistent_id": self.persistent_id} if self.persistent_id else {}),
69
+ }
70
+
71
+
72
+ def get_package_version(default: str = "0.0.0") -> str:
73
+ return PACKAGE_VERSION or default
74
+
75
+
76
+ def get_model_asset_host() -> str:
77
+ return os.environ.get("WFLOAT_MODEL_ASSET_HOST", DEFAULT_MODEL_ASSET_HOST).rstrip("/")
78
+
79
+
80
+ def filename_from_url(url: str, fallback: str) -> str:
81
+ parsed = urlparse(url)
82
+ filename = Path(parsed.path).name
83
+ return filename or fallback
84
+
85
+
86
+ def fetch_model_assets(
87
+ model_name: str,
88
+ *,
89
+ persistent_id: Optional[str] = None,
90
+ package_version_override: Optional[str] = None,
91
+ timeout: float = 60.0,
92
+ ) -> ModelAssets:
93
+ version = package_version_override or get_package_version()
94
+ query = {
95
+ "platform": "python",
96
+ "version": version,
97
+ "model_name": model_name,
98
+ }
99
+ if persistent_id:
100
+ query["persistent_id"] = persistent_id
101
+
102
+ params = urlencode(query)
103
+ url = "%s%s?%s" % (get_model_asset_host(), DEFAULT_MODEL_ASSET_PATH, params)
104
+ request = Request(
105
+ url,
106
+ headers={
107
+ "Accept": "application/json",
108
+ "User-Agent": "wfloat-python/%s" % version,
109
+ },
110
+ method="GET",
111
+ )
112
+
113
+ with urlopen(request, timeout=timeout) as response:
114
+ payload = response.read().decode("utf-8")
115
+
116
+ try:
117
+ data = json.loads(payload)
118
+ except json.JSONDecodeError as exc:
119
+ raise RuntimeError("Failed to decode model asset response JSON.") from exc
120
+
121
+ if not isinstance(data, dict):
122
+ raise RuntimeError("Model asset response must be a JSON object.")
123
+
124
+ return ModelAssets.from_dict(data)
@@ -0,0 +1,58 @@
1
+ try:
2
+ import sherpa_onnx
3
+ except ImportError as exc:
4
+ raise ImportError(
5
+ "Failed to import sherpa_onnx. "
6
+ "Reinstall wfloat so pip can install the matching wfloat-sherpa-onnx dependency."
7
+ ) from exc
8
+
9
+
10
+ _REQUIRED_EXPORTS = (
11
+ "GenerationConfig",
12
+ "OfflineTts",
13
+ "OfflineTtsConfig",
14
+ "OfflineTtsModelConfig",
15
+ "OfflineTtsWfloatModelConfig",
16
+ "WfloatPreparedText",
17
+ "git_date",
18
+ "git_sha1",
19
+ "prepare_wfloat_text",
20
+ "version",
21
+ "write_wave",
22
+ )
23
+
24
+ missing_exports = [name for name in _REQUIRED_EXPORTS if not hasattr(sherpa_onnx, name)]
25
+ if missing_exports:
26
+ raise ImportError(
27
+ "Installed sherpa_onnx is missing required exports: "
28
+ f"{', '.join(missing_exports)}. "
29
+ "Reinstall wfloat so pip can install a compatible wfloat-sherpa-onnx build."
30
+ )
31
+
32
+
33
+ GenerationConfig = sherpa_onnx.GenerationConfig
34
+ OfflineTts = sherpa_onnx.OfflineTts
35
+ OfflineTtsConfig = sherpa_onnx.OfflineTtsConfig
36
+ OfflineTtsModelConfig = sherpa_onnx.OfflineTtsModelConfig
37
+ OfflineTtsWfloatModelConfig = sherpa_onnx.OfflineTtsWfloatModelConfig
38
+ WfloatPreparedText = sherpa_onnx.WfloatPreparedText
39
+ git_date = sherpa_onnx.git_date
40
+ git_sha1 = sherpa_onnx.git_sha1
41
+ prepare_wfloat_text = sherpa_onnx.prepare_wfloat_text
42
+ version = sherpa_onnx.version
43
+ write_wave = sherpa_onnx.write_wave
44
+
45
+
46
+ __all__ = [
47
+ "GenerationConfig",
48
+ "OfflineTts",
49
+ "OfflineTtsConfig",
50
+ "OfflineTtsModelConfig",
51
+ "OfflineTtsWfloatModelConfig",
52
+ "WfloatPreparedText",
53
+ "git_date",
54
+ "git_sha1",
55
+ "prepare_wfloat_text",
56
+ "version",
57
+ "write_wave",
58
+ ]