wfloat 3.0.0__py3-none-macosx_14_0_arm64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- wfloat/__init__.py +113 -0
- wfloat/__main__.py +5 -0
- wfloat/_activity.py +360 -0
- wfloat/_assets.py +481 -0
- wfloat/_audio.py +192 -0
- wfloat/_cache.py +208 -0
- wfloat/_cli.py +74 -0
- wfloat/_composite.py +33 -0
- wfloat/_constants.py +147 -0
- wfloat/_core.py +1879 -0
- wfloat/_download.py +130 -0
- wfloat/_generated_model_urls.py +444 -0
- wfloat/_language.py +983 -0
- wfloat/_language_load.py +43 -0
- wfloat/_language_types.py +162 -0
- wfloat/_lifecycle.py +783 -0
- wfloat/_llm.py +84 -0
- wfloat/_llm_assets.py +178 -0
- wfloat/_llm_bridge.py +298 -0
- wfloat/_llm_load.py +61 -0
- wfloat/_model.py +394 -0
- wfloat/_native.py +17 -0
- wfloat/_operations.py +72 -0
- wfloat/_pocket.py +227 -0
- wfloat/_recognition.py +600 -0
- wfloat/_results.py +163 -0
- wfloat/_schemas.py +359 -0
- wfloat/_speech.py +505 -0
- wfloat/_stt.py +128 -0
- wfloat/_stt_assets.py +151 -0
- wfloat/_stt_contracts.py +42 -0
- wfloat/_stt_load.py +90 -0
- wfloat/_tools.py +115 -0
- wfloat/_tts_bridge.py +140 -0
- wfloat/_tts_families.py +205 -0
- wfloat/_vad.py +119 -0
- wfloat/_vad_assets.py +130 -0
- wfloat/_vad_load.py +97 -0
- wfloat/_version.py +1 -0
- wfloat/_zipformer_vocabulary.py +216 -0
- wfloat/native/libwfloat-core.dylib +4 -0
- wfloat/native/libwfloat-python-llm.dylib +0 -0
- wfloat/py.typed +0 -0
- wfloat-3.0.0.dist-info/METADATA +66 -0
- wfloat-3.0.0.dist-info/RECORD +48 -0
- wfloat-3.0.0.dist-info/WHEEL +6 -0
- wfloat-3.0.0.dist-info/entry_points.txt +2 -0
- wfloat-3.0.0.dist-info/top_level.txt +1 -0
wfloat/__init__.py
ADDED
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
from ._constants import SPEAKER_IDS, VALID_EMOTIONS, VALID_SIDS
|
|
2
|
+
from ._llm import LlmModel
|
|
3
|
+
from ._llm_load import load_llm_model
|
|
4
|
+
from ._model import Model, TtsModel, load, load_tts_model
|
|
5
|
+
from ._stt import SttModel, SttSession
|
|
6
|
+
from ._stt_load import load_moonshine_tiny_en, load_stt_model, load_whisper_tiny_en
|
|
7
|
+
from ._vad import VadModel
|
|
8
|
+
from ._vad_load import load_silero_vad, load_vad_model
|
|
9
|
+
from ._results import (
|
|
10
|
+
Audio,
|
|
11
|
+
AudioResult,
|
|
12
|
+
GenerationResult,
|
|
13
|
+
LlmGenerationResult,
|
|
14
|
+
StreamingTranscriptionResult,
|
|
15
|
+
TranscriptionResult,
|
|
16
|
+
TranscriptionSegment,
|
|
17
|
+
TranscriptionToken,
|
|
18
|
+
Timeline,
|
|
19
|
+
TimelineChunk,
|
|
20
|
+
TtsSynthesisResult,
|
|
21
|
+
VadDetectionResult,
|
|
22
|
+
VadSegment,
|
|
23
|
+
)
|
|
24
|
+
from ._version import __version__
|
|
25
|
+
|
|
26
|
+
__all__ = [
|
|
27
|
+
"Audio",
|
|
28
|
+
"AudioResult",
|
|
29
|
+
"GenerationResult",
|
|
30
|
+
"LlmGenerationResult",
|
|
31
|
+
"LlmModel",
|
|
32
|
+
"Model",
|
|
33
|
+
"SPEAKER_IDS",
|
|
34
|
+
"SttModel",
|
|
35
|
+
"SttSession",
|
|
36
|
+
"StreamingTranscriptionResult",
|
|
37
|
+
"TtsModel",
|
|
38
|
+
"Timeline",
|
|
39
|
+
"TimelineChunk",
|
|
40
|
+
"TranscriptionResult",
|
|
41
|
+
"TranscriptionSegment",
|
|
42
|
+
"TranscriptionToken",
|
|
43
|
+
"TtsSynthesisResult",
|
|
44
|
+
"VALID_EMOTIONS",
|
|
45
|
+
"VALID_SIDS",
|
|
46
|
+
"VadDetectionResult",
|
|
47
|
+
"VadModel",
|
|
48
|
+
"VadSegment",
|
|
49
|
+
"load",
|
|
50
|
+
"load_llm_model",
|
|
51
|
+
"load_moonshine_tiny_en",
|
|
52
|
+
"load_silero_vad",
|
|
53
|
+
"load_stt_model",
|
|
54
|
+
"load_whisper_tiny_en",
|
|
55
|
+
"load_tts_model",
|
|
56
|
+
"load_vad_model",
|
|
57
|
+
]
|
|
58
|
+
|
|
59
|
+
# Redesigned synchronous surface. Legacy loaders above remain importable, while
|
|
60
|
+
# the value types below describe the new task-specific loaders.
|
|
61
|
+
from ._audio import Audio
|
|
62
|
+
from ._operations import OperationCancelledError
|
|
63
|
+
from ._schemas import SchemaConfigurationError, SchemaValidationError
|
|
64
|
+
from ._tools import ToolContext, ToolDefinition, define_tool
|
|
65
|
+
from ._language_types import (
|
|
66
|
+
GenerationResult, GenerationError, GenerationRound, PartialGenerationResult,
|
|
67
|
+
Usage, ContextLimit, StructuredOutput, StopContext, ToolCall,
|
|
68
|
+
TextEvent, ReasoningEvent, RoundStartEvent, ToolCallEvent, ToolStartEvent,
|
|
69
|
+
ToolResultEvent, ToolErrorEvent, ToolCancelEvent, ToolValidationErrorEvent,
|
|
70
|
+
GenerationEvent, StopReason, ToolValidationError,
|
|
71
|
+
)
|
|
72
|
+
from ._language import LanguageModel, LanguageStream, tool_result
|
|
73
|
+
from ._language_load import load_language_model
|
|
74
|
+
from ._speech import (
|
|
75
|
+
TextToSpeechModel, SpeechResult, SpeechChunk, SpeechTiming, SpeechSegment,
|
|
76
|
+
SpeechStream, load_text_to_speech,
|
|
77
|
+
)
|
|
78
|
+
from ._recognition import (
|
|
79
|
+
SpeechToTextModel, StreamingSpeechToTextModel, TranscriptionSession,
|
|
80
|
+
TranscriptionResult, TranscriptionUpdate, LiveTranscriptUpdate,
|
|
81
|
+
TranscriptTiming, TranscriptWord, TranscriptSegment, ProvisionalTranscript,
|
|
82
|
+
PartialTranscript, TranscriptionError, load_speech_to_text,
|
|
83
|
+
load_streaming_speech_to_text,
|
|
84
|
+
)
|
|
85
|
+
from ._activity import (
|
|
86
|
+
VoiceActivityDetectionModel, VadSession, DetectionResult, VadSessionResult,
|
|
87
|
+
SpeechRange, SpeechStartEvent, VadProbabilityEvent, VadError,
|
|
88
|
+
load_voice_activity_detection,
|
|
89
|
+
)
|
|
90
|
+
from ._lifecycle import (
|
|
91
|
+
download_model, delete_model_assets, ModelProgressEvent,
|
|
92
|
+
ModelAssetsDeletedError, ModelAssetsInUseError,
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
__all__ += [
|
|
96
|
+
"OperationCancelledError", "SchemaConfigurationError", "SchemaValidationError",
|
|
97
|
+
"ToolContext", "ToolDefinition", "define_tool", "tool_result", "ToolCall",
|
|
98
|
+
"LanguageModel", "LanguageStream", "GenerationError", "GenerationRound",
|
|
99
|
+
"PartialGenerationResult", "Usage", "ContextLimit", "StructuredOutput", "StopContext",
|
|
100
|
+
"GenerationEvent", "StopReason", "ToolValidationError",
|
|
101
|
+
"TextEvent", "ReasoningEvent", "RoundStartEvent", "ToolCallEvent", "ToolStartEvent",
|
|
102
|
+
"ToolResultEvent", "ToolErrorEvent", "ToolCancelEvent", "ToolValidationErrorEvent",
|
|
103
|
+
"TextToSpeechModel", "SpeechResult", "SpeechChunk", "SpeechTiming", "SpeechSegment",
|
|
104
|
+
"SpeechStream", "SpeechToTextModel", "StreamingSpeechToTextModel", "TranscriptionSession",
|
|
105
|
+
"TranscriptionUpdate", "LiveTranscriptUpdate", "TranscriptTiming", "TranscriptWord",
|
|
106
|
+
"TranscriptSegment", "ProvisionalTranscript", "PartialTranscript", "TranscriptionError",
|
|
107
|
+
"VoiceActivityDetectionModel", "VadSession", "DetectionResult", "VadSessionResult",
|
|
108
|
+
"SpeechRange", "SpeechStartEvent", "VadProbabilityEvent", "VadError",
|
|
109
|
+
"load_language_model", "load_text_to_speech", "load_speech_to_text",
|
|
110
|
+
"load_streaming_speech_to_text", "load_voice_activity_detection",
|
|
111
|
+
"download_model", "delete_model_assets", "ModelProgressEvent",
|
|
112
|
+
"ModelAssetsDeletedError", "ModelAssetsInUseError",
|
|
113
|
+
]
|
wfloat/__main__.py
ADDED
wfloat/_activity.py
ADDED
|
@@ -0,0 +1,360 @@
|
|
|
1
|
+
"""Synchronous VAD with Wfloat-owned confirmation, padding, and clip retention.
|
|
2
|
+
|
|
3
|
+
Native integration uses the additive score_frame(samples)->probability API and
|
|
4
|
+
reset() for recurrent/left context. Older native libraries reject explicitly.
|
|
5
|
+
The legacy segmented detector is never fed, avoiding its maximum speech split.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import math
|
|
10
|
+
import threading
|
|
11
|
+
from contextlib import contextmanager
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Optional, Union
|
|
15
|
+
|
|
16
|
+
import numpy as np
|
|
17
|
+
|
|
18
|
+
from ._audio import Audio, normalize_audio, _StreamingResampler
|
|
19
|
+
from ._operations import CancellationEvent
|
|
20
|
+
from ._recognition import _TaskModel, _callback
|
|
21
|
+
from ._vad_load import load_vad_model
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True)
|
|
25
|
+
class SpeechRange:
|
|
26
|
+
id: str
|
|
27
|
+
start_ms: float
|
|
28
|
+
end_ms: float
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True)
|
|
32
|
+
class SpeechSegment(SpeechRange):
|
|
33
|
+
audio: Optional[Audio] = None
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass(frozen=True)
|
|
37
|
+
class SpeechStartEvent:
|
|
38
|
+
id: str
|
|
39
|
+
start_ms: float
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass(frozen=True)
|
|
43
|
+
class VadProbabilityEvent:
|
|
44
|
+
probability: float
|
|
45
|
+
start_ms: float
|
|
46
|
+
end_ms: float
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass(frozen=True)
|
|
50
|
+
class DetectionData:
|
|
51
|
+
segments: list[SpeechSegment]
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass(frozen=True)
|
|
55
|
+
class DetectionResult(DetectionData):
|
|
56
|
+
stop_reason: str = 'complete'
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass(frozen=True)
|
|
60
|
+
class VadSessionData:
|
|
61
|
+
segments: list[SpeechRange]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@dataclass(frozen=True)
|
|
65
|
+
class VadSessionResult(VadSessionData):
|
|
66
|
+
stop_reason: str = 'complete'
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class VadError(RuntimeError):
|
|
70
|
+
def __init__(self, message, partial_result):
|
|
71
|
+
super().__init__(message)
|
|
72
|
+
self.partial_result = partial_result
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass(frozen=True)
|
|
76
|
+
class _Options:
|
|
77
|
+
speech_threshold: float
|
|
78
|
+
silence_threshold: float
|
|
79
|
+
min_speech_duration_ms: float
|
|
80
|
+
min_silence_duration_ms: float
|
|
81
|
+
speech_padding_ms: float
|
|
82
|
+
return_audio: bool
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _options(speech_threshold=.5, silence_threshold=None, min_speech_duration_ms=250,
|
|
86
|
+
min_silence_duration_ms=500, speech_padding_ms=30, return_audio=False):
|
|
87
|
+
def number(value, name, upper=None):
|
|
88
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
89
|
+
raise TypeError(f'{name} must be a number')
|
|
90
|
+
if not math.isfinite(value) or value < 0 or (upper is not None and value > upper):
|
|
91
|
+
raise ValueError(f'{name} is out of range')
|
|
92
|
+
return value
|
|
93
|
+
speech_threshold = number(speech_threshold, 'speech_threshold', 1)
|
|
94
|
+
silence_threshold = max(0, speech_threshold - .15) if silence_threshold is None else number(silence_threshold, 'silence_threshold', 1)
|
|
95
|
+
if silence_threshold > speech_threshold:
|
|
96
|
+
raise ValueError('silence_threshold must not exceed speech_threshold')
|
|
97
|
+
for name, value in [('min_speech_duration_ms', min_speech_duration_ms),
|
|
98
|
+
('min_silence_duration_ms', min_silence_duration_ms),
|
|
99
|
+
('speech_padding_ms', speech_padding_ms)]:
|
|
100
|
+
number(value, name, 2 ** 48)
|
|
101
|
+
if not isinstance(return_audio, bool):
|
|
102
|
+
raise TypeError('return_audio must be boolean')
|
|
103
|
+
return _Options(speech_threshold, silence_threshold, min_speech_duration_ms,
|
|
104
|
+
min_silence_duration_ms, speech_padding_ms, return_audio)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
class VoiceActivityDetectionModel(_TaskModel):
|
|
108
|
+
def __init__(self, model_id, native):
|
|
109
|
+
if not callable(getattr(native, 'score_frame', None)):
|
|
110
|
+
raise NotImplementedError('VAD requires a native score_frame binding with recurrent/context reset; the legacy segmented detector cannot supply probabilities or unsplit speech')
|
|
111
|
+
if not getattr(native, 'supports_probabilities', True):
|
|
112
|
+
raise NotImplementedError('Native VAD probability scoring unavailable; rebuild the speech core with the score-frame ABI')
|
|
113
|
+
super().__init__(native)
|
|
114
|
+
self.model_id = model_id
|
|
115
|
+
self.sample_rate = native.sample_rate
|
|
116
|
+
self.window_size = native.window_size
|
|
117
|
+
if self.sample_rate <= 0 or self.window_size <= 0:
|
|
118
|
+
raise ValueError('Invalid native VAD frame geometry')
|
|
119
|
+
|
|
120
|
+
def detect(self, audio, *, sample_rate=None, speech_threshold=.5, silence_threshold=None,
|
|
121
|
+
min_speech_duration_ms=250, min_silence_duration_ms=500, speech_padding_ms=30,
|
|
122
|
+
return_audio=False, on_probability=None, cancel_event=None) -> DetectionResult:
|
|
123
|
+
config = _options(speech_threshold, silence_threshold, min_speech_duration_ms,
|
|
124
|
+
min_silence_duration_ms, speech_padding_ms, return_audio)
|
|
125
|
+
_callback(on_probability, 'on_probability')
|
|
126
|
+
event = CancellationEvent(cancel_event)
|
|
127
|
+
pcm = normalize_audio(audio, sample_rate, self.sample_rate)
|
|
128
|
+
if not len(pcm.samples):
|
|
129
|
+
raise ValueError('Audio must not be empty')
|
|
130
|
+
with self._file_operation(event):
|
|
131
|
+
if event.is_set():
|
|
132
|
+
return DetectionResult([], 'cancelled')
|
|
133
|
+
session = VadSession(self, config, event, on_probability, None, None, None, file=True)
|
|
134
|
+
try:
|
|
135
|
+
session.push(pcm)
|
|
136
|
+
return session.finish()
|
|
137
|
+
finally:
|
|
138
|
+
session.cancel()
|
|
139
|
+
|
|
140
|
+
def create_session(self, *, speech_threshold=.5, silence_threshold=None,
|
|
141
|
+
min_speech_duration_ms=250, min_silence_duration_ms=500,
|
|
142
|
+
speech_padding_ms=30, return_audio=False, on_probability=None,
|
|
143
|
+
on_speech_start=None, on_speech_end=None, on_error=None, cancel_event=None) -> VadSession:
|
|
144
|
+
config = _options(speech_threshold, silence_threshold, min_speech_duration_ms,
|
|
145
|
+
min_silence_duration_ms, speech_padding_ms, return_audio)
|
|
146
|
+
for callback, name in [(on_probability, 'on_probability'), (on_speech_start, 'on_speech_start'),
|
|
147
|
+
(on_speech_end, 'on_speech_end'), (on_error, 'on_error')]:
|
|
148
|
+
_callback(callback, name)
|
|
149
|
+
event = CancellationEvent(cancel_event)
|
|
150
|
+
with self._condition:
|
|
151
|
+
self._ensure_open()
|
|
152
|
+
if self._unloading or self._owner or self._queue or self._session:
|
|
153
|
+
raise RuntimeError('Model already has active or queued work')
|
|
154
|
+
session = VadSession(self, config, event, on_probability, on_speech_start, on_speech_end, on_error)
|
|
155
|
+
self._session = session
|
|
156
|
+
return session
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class VadSession:
|
|
160
|
+
def __init__(self, model, config, event, on_probability, on_start, on_end, on_error, file=False):
|
|
161
|
+
self._model, self._config, self._event, self._file = model, config, event, file
|
|
162
|
+
self._on_probability, self._on_start, self._on_end, self._on_error = on_probability, on_start, on_end, on_error
|
|
163
|
+
self._lock = threading.RLock()
|
|
164
|
+
self._thread = None
|
|
165
|
+
self._done = threading.Event()
|
|
166
|
+
self._error = self._result = self._callback_exception = None
|
|
167
|
+
self._segments = []
|
|
168
|
+
self._audio = np.empty(0, np.float32)
|
|
169
|
+
self._retained = np.empty(0, np.float32)
|
|
170
|
+
self._retained_start = self._position = self._previous_end = self._input = 0
|
|
171
|
+
self._candidate = self._silence = self._active = None
|
|
172
|
+
self._resampler = _StreamingResampler(model.sample_rate)
|
|
173
|
+
try:
|
|
174
|
+
model._native.reset()
|
|
175
|
+
except Exception as cause:
|
|
176
|
+
raise VadError(str(cause), DetectionData([]) if file else VadSessionData([])) from cause
|
|
177
|
+
|
|
178
|
+
def _snapshot(self):
|
|
179
|
+
return DetectionData(list(self._segments)) if self._file else VadSessionData(list(self._segments))
|
|
180
|
+
|
|
181
|
+
def _cleanup(self):
|
|
182
|
+
self._audio = self._retained = np.empty(0, np.float32)
|
|
183
|
+
self._resampler = None
|
|
184
|
+
self._candidate = self._silence = self._active = None
|
|
185
|
+
self._done.set()
|
|
186
|
+
if not self._file:
|
|
187
|
+
self._model._release_session(self)
|
|
188
|
+
|
|
189
|
+
def _terminal(self, reason):
|
|
190
|
+
if not self._done.is_set():
|
|
191
|
+
cls = DetectionResult if self._file else VadSessionResult
|
|
192
|
+
self._result = cls(list(self._segments), reason)
|
|
193
|
+
self._cleanup()
|
|
194
|
+
|
|
195
|
+
def _notify(self, callback, value):
|
|
196
|
+
if callback is not None and not self._event.is_set():
|
|
197
|
+
try:
|
|
198
|
+
callback(value)
|
|
199
|
+
except BaseException as error:
|
|
200
|
+
self._callback_exception = error
|
|
201
|
+
raise
|
|
202
|
+
|
|
203
|
+
@contextmanager
|
|
204
|
+
def _drive(self):
|
|
205
|
+
with self._lock:
|
|
206
|
+
if self._thread is not None:
|
|
207
|
+
raise RuntimeError('Reentrant session processing is unsupported')
|
|
208
|
+
self._thread = threading.get_ident()
|
|
209
|
+
try:
|
|
210
|
+
yield
|
|
211
|
+
except BaseException as cause:
|
|
212
|
+
if not self._done.is_set():
|
|
213
|
+
if cause is self._callback_exception or not isinstance(cause, Exception):
|
|
214
|
+
self._error = cause
|
|
215
|
+
else:
|
|
216
|
+
self._error = VadError(str(cause), self._snapshot())
|
|
217
|
+
self._error.__cause__ = cause
|
|
218
|
+
self._cleanup()
|
|
219
|
+
if isinstance(self._error, VadError):
|
|
220
|
+
self._notify(self._on_error, self._error)
|
|
221
|
+
raise self._error
|
|
222
|
+
raise
|
|
223
|
+
finally:
|
|
224
|
+
self._thread = None
|
|
225
|
+
if self._event.is_set() and not self._done.is_set():
|
|
226
|
+
self._terminal('cancelled')
|
|
227
|
+
|
|
228
|
+
def push(self, audio, *, sample_rate=None) -> None:
|
|
229
|
+
if self._done.is_set():
|
|
230
|
+
raise RuntimeError('Session no longer accepts audio')
|
|
231
|
+
if isinstance(audio, (str, Path)):
|
|
232
|
+
raise TypeError('Live push accepts PCM audio, not file paths')
|
|
233
|
+
pcm = normalize_audio(audio, sample_rate)
|
|
234
|
+
if not len(pcm.samples):
|
|
235
|
+
raise ValueError('Audio must not be empty')
|
|
236
|
+
with self._drive():
|
|
237
|
+
if self._done.is_set():
|
|
238
|
+
raise RuntimeError('Session no longer accepts audio')
|
|
239
|
+
if self._event.is_set():
|
|
240
|
+
return
|
|
241
|
+
self._input += len(pcm.samples)
|
|
242
|
+
self._audio = np.concatenate((self._audio, self._resampler.push(pcm)))
|
|
243
|
+
self._pump(False)
|
|
244
|
+
|
|
245
|
+
def _samples(self, ms):
|
|
246
|
+
return math.ceil(ms * self._model.sample_rate / 1000)
|
|
247
|
+
|
|
248
|
+
def _ms(self, sample):
|
|
249
|
+
return sample * 1000 / self._model.sample_rate
|
|
250
|
+
|
|
251
|
+
def _pump(self, final):
|
|
252
|
+
window = self._model.window_size
|
|
253
|
+
while len(self._audio) and (len(self._audio) >= window or final) and not self._event.is_set():
|
|
254
|
+
count = min(window, len(self._audio))
|
|
255
|
+
audio = self._audio[:count].copy()
|
|
256
|
+
self._audio = self._audio[count:]
|
|
257
|
+
frame = np.pad(audio, (0, window - count)) if count < window else audio
|
|
258
|
+
probability = float(self._model._native.score_frame(frame))
|
|
259
|
+
if not math.isfinite(probability) or not 0 <= probability <= 1:
|
|
260
|
+
raise ValueError('Native VAD returned an invalid probability')
|
|
261
|
+
if self._event.is_set():
|
|
262
|
+
break
|
|
263
|
+
self._feed(audio, probability)
|
|
264
|
+
|
|
265
|
+
def _feed(self, audio, probability):
|
|
266
|
+
config = self._config
|
|
267
|
+
start, end = self._position, self._position + len(audio)
|
|
268
|
+
if config.return_audio:
|
|
269
|
+
self._retained = np.concatenate((self._retained, audio))
|
|
270
|
+
self._position = end
|
|
271
|
+
self._notify(self._on_probability, VadProbabilityEvent(probability, self._ms(start), self._ms(end)))
|
|
272
|
+
if self._event.is_set():
|
|
273
|
+
return
|
|
274
|
+
if self._active is None:
|
|
275
|
+
if probability >= config.speech_threshold:
|
|
276
|
+
if self._candidate is None:
|
|
277
|
+
self._candidate = start
|
|
278
|
+
if end - self._candidate >= self._samples(config.min_speech_duration_ms):
|
|
279
|
+
self._active = max(self._previous_end, 0, self._candidate - self._samples(config.speech_padding_ms))
|
|
280
|
+
self._notify(self._on_start, SpeechStartEvent(str(len(self._segments)), self._ms(self._active)))
|
|
281
|
+
else:
|
|
282
|
+
self._candidate = None
|
|
283
|
+
elif probability < config.silence_threshold:
|
|
284
|
+
if self._silence is None:
|
|
285
|
+
self._silence = start
|
|
286
|
+
if end - self._silence >= self._samples(config.min_silence_duration_ms):
|
|
287
|
+
self._close(min(end, self._silence + self._samples(config.speech_padding_ms)))
|
|
288
|
+
else:
|
|
289
|
+
self._silence = None
|
|
290
|
+
if config.return_audio:
|
|
291
|
+
keep = self._active if self._active is not None else max(self._previous_end, 0,
|
|
292
|
+
(self._candidate if self._candidate is not None else self._position) - self._samples(config.speech_padding_ms))
|
|
293
|
+
self._retained = self._retained[max(0, keep - self._retained_start):].copy()
|
|
294
|
+
self._retained_start = keep
|
|
295
|
+
|
|
296
|
+
def _close(self, end):
|
|
297
|
+
if self._active is None or self._event.is_set():
|
|
298
|
+
return
|
|
299
|
+
end = max(self._active, min(end, self._position))
|
|
300
|
+
ident, start = str(len(self._segments)), self._active
|
|
301
|
+
audio = None
|
|
302
|
+
if self._config.return_audio:
|
|
303
|
+
audio = Audio(self._retained[start - self._retained_start:end - self._retained_start], self._model.sample_rate)
|
|
304
|
+
segment = SpeechSegment(ident, self._ms(start), self._ms(end), audio)
|
|
305
|
+
self._segments.append(segment if self._file else SpeechRange(ident, segment.start_ms, segment.end_ms))
|
|
306
|
+
self._previous_end = end
|
|
307
|
+
self._active = self._candidate = self._silence = None
|
|
308
|
+
self._notify(self._on_end, segment)
|
|
309
|
+
|
|
310
|
+
def finish(self) -> Union[DetectionResult, VadSessionResult]:
|
|
311
|
+
if self._done.is_set():
|
|
312
|
+
return self.result()
|
|
313
|
+
if not self._input and not self._event.is_set():
|
|
314
|
+
raise ValueError('Cannot finish a session without audio')
|
|
315
|
+
with self._drive():
|
|
316
|
+
if not self._done.is_set() and not self._event.is_set():
|
|
317
|
+
self._audio = np.concatenate((self._audio, self._resampler.finish()))
|
|
318
|
+
self._pump(True)
|
|
319
|
+
if self._active is not None:
|
|
320
|
+
end = self._position if self._silence is None else min(self._position, self._silence + self._samples(self._config.speech_padding_ms))
|
|
321
|
+
self._close(end)
|
|
322
|
+
self._terminal('cancelled' if self._event.is_set() else 'complete')
|
|
323
|
+
return self.result()
|
|
324
|
+
|
|
325
|
+
def cancel(self) -> None:
|
|
326
|
+
self._event.set()
|
|
327
|
+
if self._lock.acquire(blocking=False):
|
|
328
|
+
try:
|
|
329
|
+
if self._thread is None:
|
|
330
|
+
self._terminal('cancelled')
|
|
331
|
+
finally:
|
|
332
|
+
self._lock.release()
|
|
333
|
+
|
|
334
|
+
def result(self) -> Union[DetectionResult, VadSessionResult]:
|
|
335
|
+
"""Wait for another thread to finish/cancel; does not end input."""
|
|
336
|
+
if self._thread == threading.get_ident() and not self._done.is_set():
|
|
337
|
+
raise RuntimeError('Cannot wait for a session result from its callback')
|
|
338
|
+
while not self._done.wait(.02):
|
|
339
|
+
if self._event.is_set():
|
|
340
|
+
self.cancel()
|
|
341
|
+
if self._error is not None:
|
|
342
|
+
raise self._error
|
|
343
|
+
return self._result
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def load_voice_activity_detection(model_id, *, cache_dir=None, on_progress=None, cancel_event=None) -> VoiceActivityDetectionModel:
|
|
347
|
+
from ._lifecycle import load_with_lifecycle
|
|
348
|
+
if model_id != 'snakers4/silero-vad':
|
|
349
|
+
raise ValueError('The probability VAD surface currently supports snakers4/silero-vad')
|
|
350
|
+
def initialize(lease):
|
|
351
|
+
legacy = load_vad_model(model_id, cache_dir=cache_dir, force_download=False)
|
|
352
|
+
try:
|
|
353
|
+
model = VoiceActivityDetectionModel(model_id, legacy._native_vad)
|
|
354
|
+
model._asset_lease = lease
|
|
355
|
+
return model
|
|
356
|
+
except BaseException:
|
|
357
|
+
legacy._native_vad.close()
|
|
358
|
+
raise
|
|
359
|
+
return load_with_lifecycle(model_id, initialize, cache_dir=cache_dir,
|
|
360
|
+
on_progress=on_progress, cancel_event=cancel_event)
|