wfloat 1.0.0__tar.gz → 1.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- wfloat-1.0.1/PKG-INFO +172 -0
- wfloat-1.0.1/README.md +147 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat/_cli.py +12 -12
- wfloat-1.0.1/python/wfloat/_version.py +1 -0
- wfloat-1.0.1/python/wfloat.egg-info/PKG-INFO +172 -0
- wfloat-1.0.1/python/wfloat.egg-info/requires.txt +1 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/setup.py +1 -1
- {wfloat-1.0.0 → wfloat-1.0.1}/tests/test_basic.py +1 -1
- wfloat-1.0.0/PKG-INFO +0 -71
- wfloat-1.0.0/README.md +0 -46
- wfloat-1.0.0/python/wfloat/_version.py +0 -1
- wfloat-1.0.0/python/wfloat.egg-info/PKG-INFO +0 -71
- wfloat-1.0.0/python/wfloat.egg-info/requires.txt +0 -1
- {wfloat-1.0.0 → wfloat-1.0.1}/MANIFEST.in +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/pyproject.toml +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat/__init__.py +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat/__main__.py +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat/_assets.py +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat/_bindings.py +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat/_cache.py +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat/_constants.py +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat/_download.py +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat/_model.py +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat/_native.py +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat/_results.py +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat.egg-info/SOURCES.txt +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat.egg-info/dependency_links.txt +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat.egg-info/entry_points.txt +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat.egg-info/not-zip-safe +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/python/wfloat.egg-info/top_level.txt +0 -0
- {wfloat-1.0.0 → wfloat-1.0.1}/setup.cfg +0 -0
wfloat-1.0.1/PKG-INFO
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: wfloat
|
|
3
|
+
Version: 1.0.1
|
|
4
|
+
Summary: High-level Python wrapper for Wfloat TTS
|
|
5
|
+
Home-page: https://github.com/wfloat/wfloat-python
|
|
6
|
+
Author: wfloat
|
|
7
|
+
License: MIT
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
10
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
11
|
+
Classifier: Operating System :: MacOS :: MacOS X
|
|
12
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
13
|
+
Requires-Python: >=3.9
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: wfloat-sherpa-onnx==1.12.24
|
|
16
|
+
Dynamic: author
|
|
17
|
+
Dynamic: classifier
|
|
18
|
+
Dynamic: description
|
|
19
|
+
Dynamic: description-content-type
|
|
20
|
+
Dynamic: home-page
|
|
21
|
+
Dynamic: license
|
|
22
|
+
Dynamic: requires-dist
|
|
23
|
+
Dynamic: requires-python
|
|
24
|
+
Dynamic: summary
|
|
25
|
+
|
|
26
|
+
# wfloat
|
|
27
|
+
|
|
28
|
+
`wfloat` is the Python package for `wfloat-tts`, Wfloat's on-device English
|
|
29
|
+
text-to-speech model.
|
|
30
|
+
|
|
31
|
+
It runs speech locally in Python instead of calling a hosted inference API.
|
|
32
|
+
The model supports 20 voices with emotion and intensity control.
|
|
33
|
+
|
|
34
|
+
If you're building for the browser, use
|
|
35
|
+
[`@wfloat/wfloat-web`](https://github.com/wfloat/wfloat-web). If you're
|
|
36
|
+
building for React Native, use
|
|
37
|
+
[`@wfloat/react-native-wfloat`](https://github.com/wfloat/react-native-wfloat).
|
|
38
|
+
|
|
39
|
+
Try it in the browser: https://wfloat.com/demo
|
|
40
|
+
|
|
41
|
+
<audio controls src="./sample.wav">
|
|
42
|
+
<a href="./sample.wav">Sample dialogue</a>
|
|
43
|
+
</audio>
|
|
44
|
+
|
|
45
|
+
[Sample dialogue](sample.wav)
|
|
46
|
+
|
|
47
|
+
## Install
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
pip install wfloat
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
## Usage
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
import wfloat
|
|
57
|
+
|
|
58
|
+
model = wfloat.load("wfloat/wfloat-tts")
|
|
59
|
+
|
|
60
|
+
result = model.generate(
|
|
61
|
+
text="No, no, that's not possible. The formula should have crystallized, but it adapted instead. Do you realize what that means for the rest of my work?",
|
|
62
|
+
voice_id="mad_scientist_woman",
|
|
63
|
+
emotion="surprise",
|
|
64
|
+
intensity=0.7,
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
result.audio.save("out.wav")
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
For multi-speaker dialogue:
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
import wfloat
|
|
74
|
+
|
|
75
|
+
model = wfloat.load("wfloat/wfloat-tts")
|
|
76
|
+
|
|
77
|
+
result = model.generate_dialogue(
|
|
78
|
+
segments=[
|
|
79
|
+
{
|
|
80
|
+
"voice_id": "wise_elder_man",
|
|
81
|
+
"text": "Rain taps against the tavern shutters as you step inside.",
|
|
82
|
+
"emotion": "neutral",
|
|
83
|
+
"intensity": 0.5,
|
|
84
|
+
},
|
|
85
|
+
{
|
|
86
|
+
"voice_id": "strong_hero_man",
|
|
87
|
+
"text": "You're late. Two bandits stole the king's map over three hours ago.",
|
|
88
|
+
"emotion": "fear",
|
|
89
|
+
"intensity": 0.6,
|
|
90
|
+
},
|
|
91
|
+
{
|
|
92
|
+
"voice_id": "strong_hero_man",
|
|
93
|
+
"text": "They fled north, up into the woods.",
|
|
94
|
+
"emotion": "neutral",
|
|
95
|
+
"intensity": 0.5,
|
|
96
|
+
},
|
|
97
|
+
],
|
|
98
|
+
silence_between_segments_sec=0.35,
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
result.audio.save("dialogue.wav")
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
You can also generate a WAV from the command line:
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
wfloat generate \
|
|
108
|
+
--text "Hello world!" \
|
|
109
|
+
--out out.wav \
|
|
110
|
+
--voice-id mad_scientist_woman \
|
|
111
|
+
--emotion surprise \
|
|
112
|
+
--intensity 0.7 \
|
|
113
|
+
--silence-padding-sec 0
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
For the full CLI help:
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
wfloat generate --help
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
The first load downloads the model assets. After that, the package uses the
|
|
123
|
+
cached local copy.
|
|
124
|
+
|
|
125
|
+
## Speaker IDs
|
|
126
|
+
|
|
127
|
+
Use `voice_id` string names or numeric `sid` values:
|
|
128
|
+
|
|
129
|
+
| Speaker | SID |
|
|
130
|
+
| --- | ---: |
|
|
131
|
+
| `skilled_hero_man` | 0 |
|
|
132
|
+
| `skilled_hero_woman` | 1 |
|
|
133
|
+
| `fun_hero_man` | 2 |
|
|
134
|
+
| `fun_hero_woman` | 3 |
|
|
135
|
+
| `strong_hero_man` | 4 |
|
|
136
|
+
| `strong_hero_woman` | 5 |
|
|
137
|
+
| `mad_scientist_man` | 6 |
|
|
138
|
+
| `mad_scientist_woman` | 7 |
|
|
139
|
+
| `clever_villain_man` | 8 |
|
|
140
|
+
| `clever_villain_woman` | 9 |
|
|
141
|
+
| `narrator_man` | 10 |
|
|
142
|
+
| `narrator_woman` | 11 |
|
|
143
|
+
| `wise_elder_man` | 12 |
|
|
144
|
+
| `wise_elder_woman` | 13 |
|
|
145
|
+
| `outgoing_anime_man` | 14 |
|
|
146
|
+
| `outgoing_anime_woman` | 15 |
|
|
147
|
+
| `scary_villain_man` | 16 |
|
|
148
|
+
| `scary_villain_woman` | 17 |
|
|
149
|
+
| `news_reporter_man` | 18 |
|
|
150
|
+
| `news_reporter_woman` | 19 |
|
|
151
|
+
|
|
152
|
+
## Emotions
|
|
153
|
+
|
|
154
|
+
Supported emotion labels:
|
|
155
|
+
|
|
156
|
+
- `neutral`
|
|
157
|
+
- `joy`
|
|
158
|
+
- `sadness`
|
|
159
|
+
- `anger`
|
|
160
|
+
- `fear`
|
|
161
|
+
- `surprise`
|
|
162
|
+
- `dismissive`
|
|
163
|
+
- `confusion`
|
|
164
|
+
|
|
165
|
+
`intensity` must be between `0.0` and `1.0`.
|
|
166
|
+
|
|
167
|
+
## More
|
|
168
|
+
|
|
169
|
+
- Docs: https://docs.wfloat.com
|
|
170
|
+
- Model card, voices, emotions, and samples: https://huggingface.co/Wfloat/wfloat-tts
|
|
171
|
+
- Web package: https://github.com/wfloat/wfloat-web
|
|
172
|
+
- React Native package: https://github.com/wfloat/react-native-wfloat
|
wfloat-1.0.1/README.md
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
# wfloat
|
|
2
|
+
|
|
3
|
+
`wfloat` is the Python package for `wfloat-tts`, Wfloat's on-device English
|
|
4
|
+
text-to-speech model.
|
|
5
|
+
|
|
6
|
+
It runs speech locally in Python instead of calling a hosted inference API.
|
|
7
|
+
The model supports 20 voices with emotion and intensity control.
|
|
8
|
+
|
|
9
|
+
If you're building for the browser, use
|
|
10
|
+
[`@wfloat/wfloat-web`](https://github.com/wfloat/wfloat-web). If you're
|
|
11
|
+
building for React Native, use
|
|
12
|
+
[`@wfloat/react-native-wfloat`](https://github.com/wfloat/react-native-wfloat).
|
|
13
|
+
|
|
14
|
+
Try it in the browser: https://wfloat.com/demo
|
|
15
|
+
|
|
16
|
+
<audio controls src="./sample.wav">
|
|
17
|
+
<a href="./sample.wav">Sample dialogue</a>
|
|
18
|
+
</audio>
|
|
19
|
+
|
|
20
|
+
[Sample dialogue](sample.wav)
|
|
21
|
+
|
|
22
|
+
## Install
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pip install wfloat
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## Usage
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
import wfloat
|
|
32
|
+
|
|
33
|
+
model = wfloat.load("wfloat/wfloat-tts")
|
|
34
|
+
|
|
35
|
+
result = model.generate(
|
|
36
|
+
text="No, no, that's not possible. The formula should have crystallized, but it adapted instead. Do you realize what that means for the rest of my work?",
|
|
37
|
+
voice_id="mad_scientist_woman",
|
|
38
|
+
emotion="surprise",
|
|
39
|
+
intensity=0.7,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
result.audio.save("out.wav")
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
For multi-speaker dialogue:
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
import wfloat
|
|
49
|
+
|
|
50
|
+
model = wfloat.load("wfloat/wfloat-tts")
|
|
51
|
+
|
|
52
|
+
result = model.generate_dialogue(
|
|
53
|
+
segments=[
|
|
54
|
+
{
|
|
55
|
+
"voice_id": "wise_elder_man",
|
|
56
|
+
"text": "Rain taps against the tavern shutters as you step inside.",
|
|
57
|
+
"emotion": "neutral",
|
|
58
|
+
"intensity": 0.5,
|
|
59
|
+
},
|
|
60
|
+
{
|
|
61
|
+
"voice_id": "strong_hero_man",
|
|
62
|
+
"text": "You're late. Two bandits stole the king's map over three hours ago.",
|
|
63
|
+
"emotion": "fear",
|
|
64
|
+
"intensity": 0.6,
|
|
65
|
+
},
|
|
66
|
+
{
|
|
67
|
+
"voice_id": "strong_hero_man",
|
|
68
|
+
"text": "They fled north, up into the woods.",
|
|
69
|
+
"emotion": "neutral",
|
|
70
|
+
"intensity": 0.5,
|
|
71
|
+
},
|
|
72
|
+
],
|
|
73
|
+
silence_between_segments_sec=0.35,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
result.audio.save("dialogue.wav")
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
You can also generate a WAV from the command line:
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
wfloat generate \
|
|
83
|
+
--text "Hello world!" \
|
|
84
|
+
--out out.wav \
|
|
85
|
+
--voice-id mad_scientist_woman \
|
|
86
|
+
--emotion surprise \
|
|
87
|
+
--intensity 0.7 \
|
|
88
|
+
--silence-padding-sec 0
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
For the full CLI help:
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
wfloat generate --help
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
The first load downloads the model assets. After that, the package uses the
|
|
98
|
+
cached local copy.
|
|
99
|
+
|
|
100
|
+
## Speaker IDs
|
|
101
|
+
|
|
102
|
+
Use `voice_id` string names or numeric `sid` values:
|
|
103
|
+
|
|
104
|
+
| Speaker | SID |
|
|
105
|
+
| --- | ---: |
|
|
106
|
+
| `skilled_hero_man` | 0 |
|
|
107
|
+
| `skilled_hero_woman` | 1 |
|
|
108
|
+
| `fun_hero_man` | 2 |
|
|
109
|
+
| `fun_hero_woman` | 3 |
|
|
110
|
+
| `strong_hero_man` | 4 |
|
|
111
|
+
| `strong_hero_woman` | 5 |
|
|
112
|
+
| `mad_scientist_man` | 6 |
|
|
113
|
+
| `mad_scientist_woman` | 7 |
|
|
114
|
+
| `clever_villain_man` | 8 |
|
|
115
|
+
| `clever_villain_woman` | 9 |
|
|
116
|
+
| `narrator_man` | 10 |
|
|
117
|
+
| `narrator_woman` | 11 |
|
|
118
|
+
| `wise_elder_man` | 12 |
|
|
119
|
+
| `wise_elder_woman` | 13 |
|
|
120
|
+
| `outgoing_anime_man` | 14 |
|
|
121
|
+
| `outgoing_anime_woman` | 15 |
|
|
122
|
+
| `scary_villain_man` | 16 |
|
|
123
|
+
| `scary_villain_woman` | 17 |
|
|
124
|
+
| `news_reporter_man` | 18 |
|
|
125
|
+
| `news_reporter_woman` | 19 |
|
|
126
|
+
|
|
127
|
+
## Emotions
|
|
128
|
+
|
|
129
|
+
Supported emotion labels:
|
|
130
|
+
|
|
131
|
+
- `neutral`
|
|
132
|
+
- `joy`
|
|
133
|
+
- `sadness`
|
|
134
|
+
- `anger`
|
|
135
|
+
- `fear`
|
|
136
|
+
- `surprise`
|
|
137
|
+
- `dismissive`
|
|
138
|
+
- `confusion`
|
|
139
|
+
|
|
140
|
+
`intensity` must be between `0.0` and `1.0`.
|
|
141
|
+
|
|
142
|
+
## More
|
|
143
|
+
|
|
144
|
+
- Docs: https://docs.wfloat.com
|
|
145
|
+
- Model card, voices, emotions, and samples: https://huggingface.co/Wfloat/wfloat-tts
|
|
146
|
+
- Web package: https://github.com/wfloat/wfloat-web
|
|
147
|
+
- React Native package: https://github.com/wfloat/react-native-wfloat
|
|
@@ -7,26 +7,26 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
7
7
|
parser = argparse.ArgumentParser(prog="wfloat")
|
|
8
8
|
subparsers = parser.add_subparsers(dest="command")
|
|
9
9
|
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
10
|
+
generate = subparsers.add_parser("generate", help="Generate speech and write a WAV file.")
|
|
11
|
+
generate.add_argument("--model", default="wfloat/wfloat-tts", help="Model name to load.")
|
|
12
|
+
generate.add_argument("--text", required=True, help="Text to synthesize.")
|
|
13
|
+
generate.add_argument("--out", required=True, help="Output WAV path.")
|
|
14
|
+
generate.add_argument("--voice-id", default=None, help="Voice ID name or numeric SID.")
|
|
15
|
+
generate.add_argument("--emotion", default=None, help="Emotion name.")
|
|
16
|
+
generate.add_argument("--intensity", type=float, default=None, help="Emotion intensity.")
|
|
17
|
+
generate.add_argument("--speed", type=float, default=None, help="Speech speed.")
|
|
18
|
+
generate.add_argument(
|
|
19
19
|
"--silence-padding-sec",
|
|
20
20
|
type=float,
|
|
21
21
|
default=None,
|
|
22
22
|
help="Silence padding between generated sentence chunks.",
|
|
23
23
|
)
|
|
24
|
-
|
|
24
|
+
generate.add_argument(
|
|
25
25
|
"--cache-dir",
|
|
26
26
|
default=None,
|
|
27
27
|
help="Optional override for the cache directory.",
|
|
28
28
|
)
|
|
29
|
-
|
|
29
|
+
generate.add_argument(
|
|
30
30
|
"--force-download",
|
|
31
31
|
action="store_true",
|
|
32
32
|
help="Redownload model assets even if cached copies are present.",
|
|
@@ -49,7 +49,7 @@ def main(argv=None) -> int:
|
|
|
49
49
|
parser = build_parser()
|
|
50
50
|
args = parser.parse_args(argv)
|
|
51
51
|
|
|
52
|
-
if args.command != "
|
|
52
|
+
if args.command != "generate":
|
|
53
53
|
parser.print_help()
|
|
54
54
|
return 1
|
|
55
55
|
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "1.0.1"
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: wfloat
|
|
3
|
+
Version: 1.0.1
|
|
4
|
+
Summary: High-level Python wrapper for Wfloat TTS
|
|
5
|
+
Home-page: https://github.com/wfloat/wfloat-python
|
|
6
|
+
Author: wfloat
|
|
7
|
+
License: MIT
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
10
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
11
|
+
Classifier: Operating System :: MacOS :: MacOS X
|
|
12
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
13
|
+
Requires-Python: >=3.9
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: wfloat-sherpa-onnx==1.12.24
|
|
16
|
+
Dynamic: author
|
|
17
|
+
Dynamic: classifier
|
|
18
|
+
Dynamic: description
|
|
19
|
+
Dynamic: description-content-type
|
|
20
|
+
Dynamic: home-page
|
|
21
|
+
Dynamic: license
|
|
22
|
+
Dynamic: requires-dist
|
|
23
|
+
Dynamic: requires-python
|
|
24
|
+
Dynamic: summary
|
|
25
|
+
|
|
26
|
+
# wfloat
|
|
27
|
+
|
|
28
|
+
`wfloat` is the Python package for `wfloat-tts`, Wfloat's on-device English
|
|
29
|
+
text-to-speech model.
|
|
30
|
+
|
|
31
|
+
It runs speech locally in Python instead of calling a hosted inference API.
|
|
32
|
+
The model supports 20 voices with emotion and intensity control.
|
|
33
|
+
|
|
34
|
+
If you're building for the browser, use
|
|
35
|
+
[`@wfloat/wfloat-web`](https://github.com/wfloat/wfloat-web). If you're
|
|
36
|
+
building for React Native, use
|
|
37
|
+
[`@wfloat/react-native-wfloat`](https://github.com/wfloat/react-native-wfloat).
|
|
38
|
+
|
|
39
|
+
Try it in the browser: https://wfloat.com/demo
|
|
40
|
+
|
|
41
|
+
<audio controls src="./sample.wav">
|
|
42
|
+
<a href="./sample.wav">Sample dialogue</a>
|
|
43
|
+
</audio>
|
|
44
|
+
|
|
45
|
+
[Sample dialogue](sample.wav)
|
|
46
|
+
|
|
47
|
+
## Install
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
pip install wfloat
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
## Usage
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
import wfloat
|
|
57
|
+
|
|
58
|
+
model = wfloat.load("wfloat/wfloat-tts")
|
|
59
|
+
|
|
60
|
+
result = model.generate(
|
|
61
|
+
text="No, no, that's not possible. The formula should have crystallized, but it adapted instead. Do you realize what that means for the rest of my work?",
|
|
62
|
+
voice_id="mad_scientist_woman",
|
|
63
|
+
emotion="surprise",
|
|
64
|
+
intensity=0.7,
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
result.audio.save("out.wav")
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
For multi-speaker dialogue:
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
import wfloat
|
|
74
|
+
|
|
75
|
+
model = wfloat.load("wfloat/wfloat-tts")
|
|
76
|
+
|
|
77
|
+
result = model.generate_dialogue(
|
|
78
|
+
segments=[
|
|
79
|
+
{
|
|
80
|
+
"voice_id": "wise_elder_man",
|
|
81
|
+
"text": "Rain taps against the tavern shutters as you step inside.",
|
|
82
|
+
"emotion": "neutral",
|
|
83
|
+
"intensity": 0.5,
|
|
84
|
+
},
|
|
85
|
+
{
|
|
86
|
+
"voice_id": "strong_hero_man",
|
|
87
|
+
"text": "You're late. Two bandits stole the king's map over three hours ago.",
|
|
88
|
+
"emotion": "fear",
|
|
89
|
+
"intensity": 0.6,
|
|
90
|
+
},
|
|
91
|
+
{
|
|
92
|
+
"voice_id": "strong_hero_man",
|
|
93
|
+
"text": "They fled north, up into the woods.",
|
|
94
|
+
"emotion": "neutral",
|
|
95
|
+
"intensity": 0.5,
|
|
96
|
+
},
|
|
97
|
+
],
|
|
98
|
+
silence_between_segments_sec=0.35,
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
result.audio.save("dialogue.wav")
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
You can also generate a WAV from the command line:
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
wfloat generate \
|
|
108
|
+
--text "Hello world!" \
|
|
109
|
+
--out out.wav \
|
|
110
|
+
--voice-id mad_scientist_woman \
|
|
111
|
+
--emotion surprise \
|
|
112
|
+
--intensity 0.7 \
|
|
113
|
+
--silence-padding-sec 0
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
For the full CLI help:
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
wfloat generate --help
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
The first load downloads the model assets. After that, the package uses the
|
|
123
|
+
cached local copy.
|
|
124
|
+
|
|
125
|
+
## Speaker IDs
|
|
126
|
+
|
|
127
|
+
Use `voice_id` string names or numeric `sid` values:
|
|
128
|
+
|
|
129
|
+
| Speaker | SID |
|
|
130
|
+
| --- | ---: |
|
|
131
|
+
| `skilled_hero_man` | 0 |
|
|
132
|
+
| `skilled_hero_woman` | 1 |
|
|
133
|
+
| `fun_hero_man` | 2 |
|
|
134
|
+
| `fun_hero_woman` | 3 |
|
|
135
|
+
| `strong_hero_man` | 4 |
|
|
136
|
+
| `strong_hero_woman` | 5 |
|
|
137
|
+
| `mad_scientist_man` | 6 |
|
|
138
|
+
| `mad_scientist_woman` | 7 |
|
|
139
|
+
| `clever_villain_man` | 8 |
|
|
140
|
+
| `clever_villain_woman` | 9 |
|
|
141
|
+
| `narrator_man` | 10 |
|
|
142
|
+
| `narrator_woman` | 11 |
|
|
143
|
+
| `wise_elder_man` | 12 |
|
|
144
|
+
| `wise_elder_woman` | 13 |
|
|
145
|
+
| `outgoing_anime_man` | 14 |
|
|
146
|
+
| `outgoing_anime_woman` | 15 |
|
|
147
|
+
| `scary_villain_man` | 16 |
|
|
148
|
+
| `scary_villain_woman` | 17 |
|
|
149
|
+
| `news_reporter_man` | 18 |
|
|
150
|
+
| `news_reporter_woman` | 19 |
|
|
151
|
+
|
|
152
|
+
## Emotions
|
|
153
|
+
|
|
154
|
+
Supported emotion labels:
|
|
155
|
+
|
|
156
|
+
- `neutral`
|
|
157
|
+
- `joy`
|
|
158
|
+
- `sadness`
|
|
159
|
+
- `anger`
|
|
160
|
+
- `fear`
|
|
161
|
+
- `surprise`
|
|
162
|
+
- `dismissive`
|
|
163
|
+
- `confusion`
|
|
164
|
+
|
|
165
|
+
`intensity` must be between `0.0` and `1.0`.
|
|
166
|
+
|
|
167
|
+
## More
|
|
168
|
+
|
|
169
|
+
- Docs: https://docs.wfloat.com
|
|
170
|
+
- Model card, voices, emotions, and samples: https://huggingface.co/Wfloat/wfloat-tts
|
|
171
|
+
- Web package: https://github.com/wfloat/wfloat-web
|
|
172
|
+
- React Native package: https://github.com/wfloat/react-native-wfloat
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
wfloat-sherpa-onnx==1.12.24
|
wfloat-1.0.0/PKG-INFO
DELETED
|
@@ -1,71 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: wfloat
|
|
3
|
-
Version: 1.0.0
|
|
4
|
-
Summary: High-level Python wrapper for Wfloat TTS
|
|
5
|
-
Home-page: https://github.com/wfloat/wfloat-python
|
|
6
|
-
Author: wfloat
|
|
7
|
-
License: MIT
|
|
8
|
-
Classifier: Programming Language :: Python :: 3
|
|
9
|
-
Classifier: Operating System :: Microsoft :: Windows
|
|
10
|
-
Classifier: Operating System :: POSIX :: Linux
|
|
11
|
-
Classifier: Operating System :: MacOS :: MacOS X
|
|
12
|
-
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
13
|
-
Requires-Python: >=3.9
|
|
14
|
-
Description-Content-Type: text/markdown
|
|
15
|
-
Requires-Dist: wfloat-sherpa-onnx==1.12.23
|
|
16
|
-
Dynamic: author
|
|
17
|
-
Dynamic: classifier
|
|
18
|
-
Dynamic: description
|
|
19
|
-
Dynamic: description-content-type
|
|
20
|
-
Dynamic: home-page
|
|
21
|
-
Dynamic: license
|
|
22
|
-
Dynamic: requires-dist
|
|
23
|
-
Dynamic: requires-python
|
|
24
|
-
Dynamic: summary
|
|
25
|
-
|
|
26
|
-
# wfloat
|
|
27
|
-
|
|
28
|
-
`wfloat` is a high-level Python wrapper around `sherpa-onnx` for loading
|
|
29
|
-
Wfloat-compatible speech models and generating audio files.
|
|
30
|
-
|
|
31
|
-
## Install
|
|
32
|
-
|
|
33
|
-
Install `wfloat` normally:
|
|
34
|
-
|
|
35
|
-
```bash
|
|
36
|
-
pip install wfloat
|
|
37
|
-
```
|
|
38
|
-
|
|
39
|
-
That will also install the matching `wfloat-sherpa-onnx` dependency from PyPI.
|
|
40
|
-
|
|
41
|
-
When installing from this repo locally:
|
|
42
|
-
|
|
43
|
-
```bash
|
|
44
|
-
pip install ./packages/wfloat-python
|
|
45
|
-
```
|
|
46
|
-
|
|
47
|
-
## Usage
|
|
48
|
-
|
|
49
|
-
```python
|
|
50
|
-
import wfloat
|
|
51
|
-
|
|
52
|
-
model = wfloat.load("wfloat/wfloat-tts")
|
|
53
|
-
|
|
54
|
-
result = model.generate(
|
|
55
|
-
text="The signal is clean. Start the recording.",
|
|
56
|
-
voice_id="narrator_woman",
|
|
57
|
-
emotion="neutral",
|
|
58
|
-
intensity=0.5,
|
|
59
|
-
speed=1.0,
|
|
60
|
-
)
|
|
61
|
-
|
|
62
|
-
result.audio.save("out.wav")
|
|
63
|
-
```
|
|
64
|
-
|
|
65
|
-
## Notes
|
|
66
|
-
|
|
67
|
-
- `wfloat` does not build or bundle native libraries.
|
|
68
|
-
- Low-level bindings come from the installed `wfloat-sherpa-onnx` dependency,
|
|
69
|
-
which provides `import sherpa_onnx`.
|
|
70
|
-
- The public API is intentionally high-level; low-level native config objects
|
|
71
|
-
are re-exported only for advanced use.
|
wfloat-1.0.0/README.md
DELETED
|
@@ -1,46 +0,0 @@
|
|
|
1
|
-
# wfloat
|
|
2
|
-
|
|
3
|
-
`wfloat` is a high-level Python wrapper around `sherpa-onnx` for loading
|
|
4
|
-
Wfloat-compatible speech models and generating audio files.
|
|
5
|
-
|
|
6
|
-
## Install
|
|
7
|
-
|
|
8
|
-
Install `wfloat` normally:
|
|
9
|
-
|
|
10
|
-
```bash
|
|
11
|
-
pip install wfloat
|
|
12
|
-
```
|
|
13
|
-
|
|
14
|
-
That will also install the matching `wfloat-sherpa-onnx` dependency from PyPI.
|
|
15
|
-
|
|
16
|
-
When installing from this repo locally:
|
|
17
|
-
|
|
18
|
-
```bash
|
|
19
|
-
pip install ./packages/wfloat-python
|
|
20
|
-
```
|
|
21
|
-
|
|
22
|
-
## Usage
|
|
23
|
-
|
|
24
|
-
```python
|
|
25
|
-
import wfloat
|
|
26
|
-
|
|
27
|
-
model = wfloat.load("wfloat/wfloat-tts")
|
|
28
|
-
|
|
29
|
-
result = model.generate(
|
|
30
|
-
text="The signal is clean. Start the recording.",
|
|
31
|
-
voice_id="narrator_woman",
|
|
32
|
-
emotion="neutral",
|
|
33
|
-
intensity=0.5,
|
|
34
|
-
speed=1.0,
|
|
35
|
-
)
|
|
36
|
-
|
|
37
|
-
result.audio.save("out.wav")
|
|
38
|
-
```
|
|
39
|
-
|
|
40
|
-
## Notes
|
|
41
|
-
|
|
42
|
-
- `wfloat` does not build or bundle native libraries.
|
|
43
|
-
- Low-level bindings come from the installed `wfloat-sherpa-onnx` dependency,
|
|
44
|
-
which provides `import sherpa_onnx`.
|
|
45
|
-
- The public API is intentionally high-level; low-level native config objects
|
|
46
|
-
are re-exported only for advanced use.
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
__version__ = "1.0.0"
|
|
@@ -1,71 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: wfloat
|
|
3
|
-
Version: 1.0.0
|
|
4
|
-
Summary: High-level Python wrapper for Wfloat TTS
|
|
5
|
-
Home-page: https://github.com/wfloat/wfloat-python
|
|
6
|
-
Author: wfloat
|
|
7
|
-
License: MIT
|
|
8
|
-
Classifier: Programming Language :: Python :: 3
|
|
9
|
-
Classifier: Operating System :: Microsoft :: Windows
|
|
10
|
-
Classifier: Operating System :: POSIX :: Linux
|
|
11
|
-
Classifier: Operating System :: MacOS :: MacOS X
|
|
12
|
-
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
13
|
-
Requires-Python: >=3.9
|
|
14
|
-
Description-Content-Type: text/markdown
|
|
15
|
-
Requires-Dist: wfloat-sherpa-onnx==1.12.23
|
|
16
|
-
Dynamic: author
|
|
17
|
-
Dynamic: classifier
|
|
18
|
-
Dynamic: description
|
|
19
|
-
Dynamic: description-content-type
|
|
20
|
-
Dynamic: home-page
|
|
21
|
-
Dynamic: license
|
|
22
|
-
Dynamic: requires-dist
|
|
23
|
-
Dynamic: requires-python
|
|
24
|
-
Dynamic: summary
|
|
25
|
-
|
|
26
|
-
# wfloat
|
|
27
|
-
|
|
28
|
-
`wfloat` is a high-level Python wrapper around `sherpa-onnx` for loading
|
|
29
|
-
Wfloat-compatible speech models and generating audio files.
|
|
30
|
-
|
|
31
|
-
## Install
|
|
32
|
-
|
|
33
|
-
Install `wfloat` normally:
|
|
34
|
-
|
|
35
|
-
```bash
|
|
36
|
-
pip install wfloat
|
|
37
|
-
```
|
|
38
|
-
|
|
39
|
-
That will also install the matching `wfloat-sherpa-onnx` dependency from PyPI.
|
|
40
|
-
|
|
41
|
-
When installing from this repo locally:
|
|
42
|
-
|
|
43
|
-
```bash
|
|
44
|
-
pip install ./packages/wfloat-python
|
|
45
|
-
```
|
|
46
|
-
|
|
47
|
-
## Usage
|
|
48
|
-
|
|
49
|
-
```python
|
|
50
|
-
import wfloat
|
|
51
|
-
|
|
52
|
-
model = wfloat.load("wfloat/wfloat-tts")
|
|
53
|
-
|
|
54
|
-
result = model.generate(
|
|
55
|
-
text="The signal is clean. Start the recording.",
|
|
56
|
-
voice_id="narrator_woman",
|
|
57
|
-
emotion="neutral",
|
|
58
|
-
intensity=0.5,
|
|
59
|
-
speed=1.0,
|
|
60
|
-
)
|
|
61
|
-
|
|
62
|
-
result.audio.save("out.wav")
|
|
63
|
-
```
|
|
64
|
-
|
|
65
|
-
## Notes
|
|
66
|
-
|
|
67
|
-
- `wfloat` does not build or bundle native libraries.
|
|
68
|
-
- Low-level bindings come from the installed `wfloat-sherpa-onnx` dependency,
|
|
69
|
-
which provides `import sherpa_onnx`.
|
|
70
|
-
- The public API is intentionally high-level; low-level native config objects
|
|
71
|
-
are re-exported only for advanced use.
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
wfloat-sherpa-onnx==1.12.23
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|