wfloat 2.0.0__py3-none-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- wfloat/__init__.py +57 -0
- wfloat/__main__.py +5 -0
- wfloat/_assets.py +421 -0
- wfloat/_cache.py +208 -0
- wfloat/_cli.py +74 -0
- wfloat/_constants.py +147 -0
- wfloat/_core.py +1795 -0
- wfloat/_download.py +130 -0
- wfloat/_generated_model_urls.py +46 -0
- wfloat/_llm.py +84 -0
- wfloat/_llm_assets.py +142 -0
- wfloat/_llm_load.py +61 -0
- wfloat/_model.py +394 -0
- wfloat/_native.py +17 -0
- wfloat/_results.py +163 -0
- wfloat/_stt.py +122 -0
- wfloat/_stt_assets.py +144 -0
- wfloat/_stt_load.py +75 -0
- wfloat/_vad.py +119 -0
- wfloat/_vad_assets.py +130 -0
- wfloat/_vad_load.py +97 -0
- wfloat/_version.py +1 -0
- wfloat/native/wfloat-core.dll +0 -0
- wfloat-2.0.0.dist-info/METADATA +251 -0
- wfloat-2.0.0.dist-info/RECORD +28 -0
- wfloat-2.0.0.dist-info/WHEEL +5 -0
- wfloat-2.0.0.dist-info/entry_points.txt +2 -0
- wfloat-2.0.0.dist-info/top_level.txt +1 -0
wfloat/_cli.py
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
|
|
3
|
+
from ._model import load
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
7
|
+
parser = argparse.ArgumentParser(prog="wfloat")
|
|
8
|
+
subparsers = parser.add_subparsers(dest="command")
|
|
9
|
+
|
|
10
|
+
generate = subparsers.add_parser("generate", help="Generate speech and write a WAV file.")
|
|
11
|
+
generate.add_argument("--model", default="wfloat/wfloat-tts", help="Model name to load.")
|
|
12
|
+
generate.add_argument("--text", required=True, help="Text to synthesize.")
|
|
13
|
+
generate.add_argument("--out", required=True, help="Output WAV path.")
|
|
14
|
+
generate.add_argument("--voice-id", default=None, help="Voice ID name or numeric SID.")
|
|
15
|
+
generate.add_argument("--emotion", default=None, help="Emotion name.")
|
|
16
|
+
generate.add_argument("--intensity", type=float, default=None, help="Emotion intensity.")
|
|
17
|
+
generate.add_argument("--speed", type=float, default=None, help="Speech speed.")
|
|
18
|
+
generate.add_argument(
|
|
19
|
+
"--silence-padding-sec",
|
|
20
|
+
type=float,
|
|
21
|
+
default=None,
|
|
22
|
+
help="Silence padding between generated sentence chunks.",
|
|
23
|
+
)
|
|
24
|
+
generate.add_argument(
|
|
25
|
+
"--cache-dir",
|
|
26
|
+
default=None,
|
|
27
|
+
help="Optional override for the cache directory.",
|
|
28
|
+
)
|
|
29
|
+
generate.add_argument(
|
|
30
|
+
"--force-download",
|
|
31
|
+
action="store_true",
|
|
32
|
+
help="Redownload model assets even if cached copies are present.",
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
return parser
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _maybe_parse_voice_id(value):
|
|
39
|
+
if value is None:
|
|
40
|
+
return None
|
|
41
|
+
|
|
42
|
+
try:
|
|
43
|
+
return int(value)
|
|
44
|
+
except (TypeError, ValueError):
|
|
45
|
+
return value
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def main(argv=None) -> int:
|
|
49
|
+
parser = build_parser()
|
|
50
|
+
args = parser.parse_args(argv)
|
|
51
|
+
|
|
52
|
+
if args.command != "generate":
|
|
53
|
+
parser.print_help()
|
|
54
|
+
return 1
|
|
55
|
+
|
|
56
|
+
model = load(
|
|
57
|
+
args.model,
|
|
58
|
+
cache_dir=args.cache_dir,
|
|
59
|
+
force_download=args.force_download,
|
|
60
|
+
)
|
|
61
|
+
result = model.generate(
|
|
62
|
+
text=args.text,
|
|
63
|
+
voice_id=_maybe_parse_voice_id(args.voice_id),
|
|
64
|
+
emotion=args.emotion,
|
|
65
|
+
intensity=args.intensity,
|
|
66
|
+
speed=args.speed,
|
|
67
|
+
silence_padding_sec=args.silence_padding_sec,
|
|
68
|
+
)
|
|
69
|
+
result.audio.save(args.out)
|
|
70
|
+
print(
|
|
71
|
+
"Saved %s (duration=%.2fs, sample_rate=%d)"
|
|
72
|
+
% (args.out, result.audio.duration_sec, result.audio.sample_rate)
|
|
73
|
+
)
|
|
74
|
+
return 0
|
wfloat/_constants.py
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
from typing import Optional, Union
|
|
2
|
+
|
|
3
|
+
VoiceId = Union[str, int]
|
|
4
|
+
|
|
5
|
+
VALID_EMOTIONS = [
|
|
6
|
+
"neutral",
|
|
7
|
+
"joy",
|
|
8
|
+
"sadness",
|
|
9
|
+
"anger",
|
|
10
|
+
"fear",
|
|
11
|
+
"surprise",
|
|
12
|
+
"dismissive",
|
|
13
|
+
"confusion",
|
|
14
|
+
]
|
|
15
|
+
|
|
16
|
+
SPEAKER_IDS = {
|
|
17
|
+
"skilled_hero_man": 0,
|
|
18
|
+
"skilled_hero_woman": 1,
|
|
19
|
+
"fun_hero_man": 2,
|
|
20
|
+
"fun_hero_woman": 3,
|
|
21
|
+
"strong_hero_man": 4,
|
|
22
|
+
"strong_hero_woman": 5,
|
|
23
|
+
"mad_scientist_man": 6,
|
|
24
|
+
"mad_scientist_woman": 7,
|
|
25
|
+
"clever_villain_man": 8,
|
|
26
|
+
"clever_villain_woman": 9,
|
|
27
|
+
"narrator_man": 10,
|
|
28
|
+
"narrator_woman": 11,
|
|
29
|
+
"wise_elder_man": 12,
|
|
30
|
+
"wise_elder_woman": 13,
|
|
31
|
+
"outgoing_anime_man": 14,
|
|
32
|
+
"outgoing_anime_woman": 15,
|
|
33
|
+
"scary_villain_man": 16,
|
|
34
|
+
"scary_villain_woman": 17,
|
|
35
|
+
"news_reporter_man": 18,
|
|
36
|
+
"news_reporter_woman": 19,
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
VALID_SIDS = tuple(SPEAKER_IDS.values())
|
|
40
|
+
|
|
41
|
+
DEFAULT_VOICE_ID = 0
|
|
42
|
+
DEFAULT_EMOTION = "neutral"
|
|
43
|
+
DEFAULT_INTENSITY = 0.5
|
|
44
|
+
DEFAULT_SPEED = 1.0
|
|
45
|
+
DEFAULT_SILENCE_PADDING_SEC = 0.1
|
|
46
|
+
DEFAULT_SILENCE_BETWEEN_SEGMENTS_SEC = 0.2
|
|
47
|
+
DEFAULT_NUM_THREADS = 1
|
|
48
|
+
DEFAULT_PROVIDER = "cpu"
|
|
49
|
+
DEFAULT_MODEL_NAME = "wfloat/wfloat-tts"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def normalize_text(text: str) -> str:
|
|
53
|
+
if not isinstance(text, str):
|
|
54
|
+
raise TypeError("text must be a string.")
|
|
55
|
+
|
|
56
|
+
if not text.strip():
|
|
57
|
+
raise ValueError("text is required.")
|
|
58
|
+
|
|
59
|
+
return text
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def normalize_voice_id(voice_id: Optional[VoiceId]) -> int:
|
|
63
|
+
if voice_id is None:
|
|
64
|
+
return DEFAULT_VOICE_ID
|
|
65
|
+
|
|
66
|
+
if isinstance(voice_id, int):
|
|
67
|
+
if voice_id not in VALID_SIDS:
|
|
68
|
+
raise ValueError("Invalid numeric voice_id: %s" % voice_id)
|
|
69
|
+
return voice_id
|
|
70
|
+
|
|
71
|
+
if isinstance(voice_id, str):
|
|
72
|
+
trimmed = voice_id.strip()
|
|
73
|
+
if not trimmed:
|
|
74
|
+
return DEFAULT_VOICE_ID
|
|
75
|
+
|
|
76
|
+
mapped_sid = SPEAKER_IDS.get(trimmed)
|
|
77
|
+
if mapped_sid is None:
|
|
78
|
+
raise ValueError("Invalid string voice_id: %s" % trimmed)
|
|
79
|
+
|
|
80
|
+
return mapped_sid
|
|
81
|
+
|
|
82
|
+
raise TypeError("voice_id must be a string, integer, or None.")
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def normalize_emotion(emotion: Optional[str]) -> str:
|
|
86
|
+
if emotion is None:
|
|
87
|
+
return DEFAULT_EMOTION
|
|
88
|
+
|
|
89
|
+
if not isinstance(emotion, str):
|
|
90
|
+
raise TypeError("emotion must be a string or None.")
|
|
91
|
+
|
|
92
|
+
trimmed = emotion.strip()
|
|
93
|
+
if not trimmed:
|
|
94
|
+
return DEFAULT_EMOTION
|
|
95
|
+
|
|
96
|
+
if trimmed not in VALID_EMOTIONS:
|
|
97
|
+
raise ValueError("Invalid emotion: %s" % trimmed)
|
|
98
|
+
|
|
99
|
+
return trimmed
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def normalize_intensity(intensity: Optional[float]) -> float:
|
|
103
|
+
if intensity is None:
|
|
104
|
+
return DEFAULT_INTENSITY
|
|
105
|
+
|
|
106
|
+
try:
|
|
107
|
+
value = float(intensity)
|
|
108
|
+
except (TypeError, ValueError):
|
|
109
|
+
raise TypeError("intensity must be a finite number between 0 and 1.")
|
|
110
|
+
|
|
111
|
+
if value < 0 or value > 1:
|
|
112
|
+
raise ValueError("intensity must be between 0 and 1.")
|
|
113
|
+
|
|
114
|
+
return value
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def normalize_speed(speed: Optional[float], default: float = DEFAULT_SPEED) -> float:
|
|
118
|
+
if speed is None:
|
|
119
|
+
return default
|
|
120
|
+
|
|
121
|
+
try:
|
|
122
|
+
value = float(speed)
|
|
123
|
+
except (TypeError, ValueError):
|
|
124
|
+
raise TypeError("speed must be a finite number greater than 0.")
|
|
125
|
+
|
|
126
|
+
if value <= 0:
|
|
127
|
+
raise ValueError("speed must be greater than 0.")
|
|
128
|
+
|
|
129
|
+
return value
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def normalize_silence_padding_sec(
|
|
133
|
+
silence_padding_sec: Optional[float],
|
|
134
|
+
default: float = DEFAULT_SILENCE_PADDING_SEC,
|
|
135
|
+
) -> float:
|
|
136
|
+
if silence_padding_sec is None:
|
|
137
|
+
return default
|
|
138
|
+
|
|
139
|
+
try:
|
|
140
|
+
value = float(silence_padding_sec)
|
|
141
|
+
except (TypeError, ValueError):
|
|
142
|
+
raise TypeError("silence_padding_sec must be a finite number >= 0.")
|
|
143
|
+
|
|
144
|
+
if value < 0:
|
|
145
|
+
raise ValueError("silence_padding_sec must be >= 0.")
|
|
146
|
+
|
|
147
|
+
return value
|