wfloat 3.0.0__py3-none-macosx_14_0_arm64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
wfloat/__init__.py ADDED
@@ -0,0 +1,113 @@
1
+ from ._constants import SPEAKER_IDS, VALID_EMOTIONS, VALID_SIDS
2
+ from ._llm import LlmModel
3
+ from ._llm_load import load_llm_model
4
+ from ._model import Model, TtsModel, load, load_tts_model
5
+ from ._stt import SttModel, SttSession
6
+ from ._stt_load import load_moonshine_tiny_en, load_stt_model, load_whisper_tiny_en
7
+ from ._vad import VadModel
8
+ from ._vad_load import load_silero_vad, load_vad_model
9
+ from ._results import (
10
+ Audio,
11
+ AudioResult,
12
+ GenerationResult,
13
+ LlmGenerationResult,
14
+ StreamingTranscriptionResult,
15
+ TranscriptionResult,
16
+ TranscriptionSegment,
17
+ TranscriptionToken,
18
+ Timeline,
19
+ TimelineChunk,
20
+ TtsSynthesisResult,
21
+ VadDetectionResult,
22
+ VadSegment,
23
+ )
24
+ from ._version import __version__
25
+
26
+ __all__ = [
27
+ "Audio",
28
+ "AudioResult",
29
+ "GenerationResult",
30
+ "LlmGenerationResult",
31
+ "LlmModel",
32
+ "Model",
33
+ "SPEAKER_IDS",
34
+ "SttModel",
35
+ "SttSession",
36
+ "StreamingTranscriptionResult",
37
+ "TtsModel",
38
+ "Timeline",
39
+ "TimelineChunk",
40
+ "TranscriptionResult",
41
+ "TranscriptionSegment",
42
+ "TranscriptionToken",
43
+ "TtsSynthesisResult",
44
+ "VALID_EMOTIONS",
45
+ "VALID_SIDS",
46
+ "VadDetectionResult",
47
+ "VadModel",
48
+ "VadSegment",
49
+ "load",
50
+ "load_llm_model",
51
+ "load_moonshine_tiny_en",
52
+ "load_silero_vad",
53
+ "load_stt_model",
54
+ "load_whisper_tiny_en",
55
+ "load_tts_model",
56
+ "load_vad_model",
57
+ ]
58
+
59
+ # Redesigned synchronous surface. Legacy loaders above remain importable, while
60
+ # the value types below describe the new task-specific loaders.
61
+ from ._audio import Audio
62
+ from ._operations import OperationCancelledError
63
+ from ._schemas import SchemaConfigurationError, SchemaValidationError
64
+ from ._tools import ToolContext, ToolDefinition, define_tool
65
+ from ._language_types import (
66
+ GenerationResult, GenerationError, GenerationRound, PartialGenerationResult,
67
+ Usage, ContextLimit, StructuredOutput, StopContext, ToolCall,
68
+ TextEvent, ReasoningEvent, RoundStartEvent, ToolCallEvent, ToolStartEvent,
69
+ ToolResultEvent, ToolErrorEvent, ToolCancelEvent, ToolValidationErrorEvent,
70
+ GenerationEvent, StopReason, ToolValidationError,
71
+ )
72
+ from ._language import LanguageModel, LanguageStream, tool_result
73
+ from ._language_load import load_language_model
74
+ from ._speech import (
75
+ TextToSpeechModel, SpeechResult, SpeechChunk, SpeechTiming, SpeechSegment,
76
+ SpeechStream, load_text_to_speech,
77
+ )
78
+ from ._recognition import (
79
+ SpeechToTextModel, StreamingSpeechToTextModel, TranscriptionSession,
80
+ TranscriptionResult, TranscriptionUpdate, LiveTranscriptUpdate,
81
+ TranscriptTiming, TranscriptWord, TranscriptSegment, ProvisionalTranscript,
82
+ PartialTranscript, TranscriptionError, load_speech_to_text,
83
+ load_streaming_speech_to_text,
84
+ )
85
+ from ._activity import (
86
+ VoiceActivityDetectionModel, VadSession, DetectionResult, VadSessionResult,
87
+ SpeechRange, SpeechStartEvent, VadProbabilityEvent, VadError,
88
+ load_voice_activity_detection,
89
+ )
90
+ from ._lifecycle import (
91
+ download_model, delete_model_assets, ModelProgressEvent,
92
+ ModelAssetsDeletedError, ModelAssetsInUseError,
93
+ )
94
+
95
+ __all__ += [
96
+ "OperationCancelledError", "SchemaConfigurationError", "SchemaValidationError",
97
+ "ToolContext", "ToolDefinition", "define_tool", "tool_result", "ToolCall",
98
+ "LanguageModel", "LanguageStream", "GenerationError", "GenerationRound",
99
+ "PartialGenerationResult", "Usage", "ContextLimit", "StructuredOutput", "StopContext",
100
+ "GenerationEvent", "StopReason", "ToolValidationError",
101
+ "TextEvent", "ReasoningEvent", "RoundStartEvent", "ToolCallEvent", "ToolStartEvent",
102
+ "ToolResultEvent", "ToolErrorEvent", "ToolCancelEvent", "ToolValidationErrorEvent",
103
+ "TextToSpeechModel", "SpeechResult", "SpeechChunk", "SpeechTiming", "SpeechSegment",
104
+ "SpeechStream", "SpeechToTextModel", "StreamingSpeechToTextModel", "TranscriptionSession",
105
+ "TranscriptionUpdate", "LiveTranscriptUpdate", "TranscriptTiming", "TranscriptWord",
106
+ "TranscriptSegment", "ProvisionalTranscript", "PartialTranscript", "TranscriptionError",
107
+ "VoiceActivityDetectionModel", "VadSession", "DetectionResult", "VadSessionResult",
108
+ "SpeechRange", "SpeechStartEvent", "VadProbabilityEvent", "VadError",
109
+ "load_language_model", "load_text_to_speech", "load_speech_to_text",
110
+ "load_streaming_speech_to_text", "load_voice_activity_detection",
111
+ "download_model", "delete_model_assets", "ModelProgressEvent",
112
+ "ModelAssetsDeletedError", "ModelAssetsInUseError",
113
+ ]
wfloat/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ from ._cli import main
2
+
3
+
4
+ if __name__ == "__main__": # pragma: no cover
5
+ raise SystemExit(main())
wfloat/_activity.py ADDED
@@ -0,0 +1,360 @@
1
+ """Synchronous VAD with Wfloat-owned confirmation, padding, and clip retention.
2
+
3
+ Native integration uses the additive score_frame(samples)->probability API and
4
+ reset() for recurrent/left context. Older native libraries reject explicitly.
5
+ The legacy segmented detector is never fed, avoiding its maximum speech split.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import math
10
+ import threading
11
+ from contextlib import contextmanager
12
+ from dataclasses import dataclass
13
+ from pathlib import Path
14
+ from typing import Optional, Union
15
+
16
+ import numpy as np
17
+
18
+ from ._audio import Audio, normalize_audio, _StreamingResampler
19
+ from ._operations import CancellationEvent
20
+ from ._recognition import _TaskModel, _callback
21
+ from ._vad_load import load_vad_model
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class SpeechRange:
26
+ id: str
27
+ start_ms: float
28
+ end_ms: float
29
+
30
+
31
+ @dataclass(frozen=True)
32
+ class SpeechSegment(SpeechRange):
33
+ audio: Optional[Audio] = None
34
+
35
+
36
+ @dataclass(frozen=True)
37
+ class SpeechStartEvent:
38
+ id: str
39
+ start_ms: float
40
+
41
+
42
+ @dataclass(frozen=True)
43
+ class VadProbabilityEvent:
44
+ probability: float
45
+ start_ms: float
46
+ end_ms: float
47
+
48
+
49
+ @dataclass(frozen=True)
50
+ class DetectionData:
51
+ segments: list[SpeechSegment]
52
+
53
+
54
+ @dataclass(frozen=True)
55
+ class DetectionResult(DetectionData):
56
+ stop_reason: str = 'complete'
57
+
58
+
59
+ @dataclass(frozen=True)
60
+ class VadSessionData:
61
+ segments: list[SpeechRange]
62
+
63
+
64
+ @dataclass(frozen=True)
65
+ class VadSessionResult(VadSessionData):
66
+ stop_reason: str = 'complete'
67
+
68
+
69
+ class VadError(RuntimeError):
70
+ def __init__(self, message, partial_result):
71
+ super().__init__(message)
72
+ self.partial_result = partial_result
73
+
74
+
75
+ @dataclass(frozen=True)
76
+ class _Options:
77
+ speech_threshold: float
78
+ silence_threshold: float
79
+ min_speech_duration_ms: float
80
+ min_silence_duration_ms: float
81
+ speech_padding_ms: float
82
+ return_audio: bool
83
+
84
+
85
+ def _options(speech_threshold=.5, silence_threshold=None, min_speech_duration_ms=250,
86
+ min_silence_duration_ms=500, speech_padding_ms=30, return_audio=False):
87
+ def number(value, name, upper=None):
88
+ if isinstance(value, bool) or not isinstance(value, (int, float)):
89
+ raise TypeError(f'{name} must be a number')
90
+ if not math.isfinite(value) or value < 0 or (upper is not None and value > upper):
91
+ raise ValueError(f'{name} is out of range')
92
+ return value
93
+ speech_threshold = number(speech_threshold, 'speech_threshold', 1)
94
+ silence_threshold = max(0, speech_threshold - .15) if silence_threshold is None else number(silence_threshold, 'silence_threshold', 1)
95
+ if silence_threshold > speech_threshold:
96
+ raise ValueError('silence_threshold must not exceed speech_threshold')
97
+ for name, value in [('min_speech_duration_ms', min_speech_duration_ms),
98
+ ('min_silence_duration_ms', min_silence_duration_ms),
99
+ ('speech_padding_ms', speech_padding_ms)]:
100
+ number(value, name, 2 ** 48)
101
+ if not isinstance(return_audio, bool):
102
+ raise TypeError('return_audio must be boolean')
103
+ return _Options(speech_threshold, silence_threshold, min_speech_duration_ms,
104
+ min_silence_duration_ms, speech_padding_ms, return_audio)
105
+
106
+
107
+ class VoiceActivityDetectionModel(_TaskModel):
108
+ def __init__(self, model_id, native):
109
+ if not callable(getattr(native, 'score_frame', None)):
110
+ raise NotImplementedError('VAD requires a native score_frame binding with recurrent/context reset; the legacy segmented detector cannot supply probabilities or unsplit speech')
111
+ if not getattr(native, 'supports_probabilities', True):
112
+ raise NotImplementedError('Native VAD probability scoring unavailable; rebuild the speech core with the score-frame ABI')
113
+ super().__init__(native)
114
+ self.model_id = model_id
115
+ self.sample_rate = native.sample_rate
116
+ self.window_size = native.window_size
117
+ if self.sample_rate <= 0 or self.window_size <= 0:
118
+ raise ValueError('Invalid native VAD frame geometry')
119
+
120
+ def detect(self, audio, *, sample_rate=None, speech_threshold=.5, silence_threshold=None,
121
+ min_speech_duration_ms=250, min_silence_duration_ms=500, speech_padding_ms=30,
122
+ return_audio=False, on_probability=None, cancel_event=None) -> DetectionResult:
123
+ config = _options(speech_threshold, silence_threshold, min_speech_duration_ms,
124
+ min_silence_duration_ms, speech_padding_ms, return_audio)
125
+ _callback(on_probability, 'on_probability')
126
+ event = CancellationEvent(cancel_event)
127
+ pcm = normalize_audio(audio, sample_rate, self.sample_rate)
128
+ if not len(pcm.samples):
129
+ raise ValueError('Audio must not be empty')
130
+ with self._file_operation(event):
131
+ if event.is_set():
132
+ return DetectionResult([], 'cancelled')
133
+ session = VadSession(self, config, event, on_probability, None, None, None, file=True)
134
+ try:
135
+ session.push(pcm)
136
+ return session.finish()
137
+ finally:
138
+ session.cancel()
139
+
140
+ def create_session(self, *, speech_threshold=.5, silence_threshold=None,
141
+ min_speech_duration_ms=250, min_silence_duration_ms=500,
142
+ speech_padding_ms=30, return_audio=False, on_probability=None,
143
+ on_speech_start=None, on_speech_end=None, on_error=None, cancel_event=None) -> VadSession:
144
+ config = _options(speech_threshold, silence_threshold, min_speech_duration_ms,
145
+ min_silence_duration_ms, speech_padding_ms, return_audio)
146
+ for callback, name in [(on_probability, 'on_probability'), (on_speech_start, 'on_speech_start'),
147
+ (on_speech_end, 'on_speech_end'), (on_error, 'on_error')]:
148
+ _callback(callback, name)
149
+ event = CancellationEvent(cancel_event)
150
+ with self._condition:
151
+ self._ensure_open()
152
+ if self._unloading or self._owner or self._queue or self._session:
153
+ raise RuntimeError('Model already has active or queued work')
154
+ session = VadSession(self, config, event, on_probability, on_speech_start, on_speech_end, on_error)
155
+ self._session = session
156
+ return session
157
+
158
+
159
+ class VadSession:
160
+ def __init__(self, model, config, event, on_probability, on_start, on_end, on_error, file=False):
161
+ self._model, self._config, self._event, self._file = model, config, event, file
162
+ self._on_probability, self._on_start, self._on_end, self._on_error = on_probability, on_start, on_end, on_error
163
+ self._lock = threading.RLock()
164
+ self._thread = None
165
+ self._done = threading.Event()
166
+ self._error = self._result = self._callback_exception = None
167
+ self._segments = []
168
+ self._audio = np.empty(0, np.float32)
169
+ self._retained = np.empty(0, np.float32)
170
+ self._retained_start = self._position = self._previous_end = self._input = 0
171
+ self._candidate = self._silence = self._active = None
172
+ self._resampler = _StreamingResampler(model.sample_rate)
173
+ try:
174
+ model._native.reset()
175
+ except Exception as cause:
176
+ raise VadError(str(cause), DetectionData([]) if file else VadSessionData([])) from cause
177
+
178
+ def _snapshot(self):
179
+ return DetectionData(list(self._segments)) if self._file else VadSessionData(list(self._segments))
180
+
181
+ def _cleanup(self):
182
+ self._audio = self._retained = np.empty(0, np.float32)
183
+ self._resampler = None
184
+ self._candidate = self._silence = self._active = None
185
+ self._done.set()
186
+ if not self._file:
187
+ self._model._release_session(self)
188
+
189
+ def _terminal(self, reason):
190
+ if not self._done.is_set():
191
+ cls = DetectionResult if self._file else VadSessionResult
192
+ self._result = cls(list(self._segments), reason)
193
+ self._cleanup()
194
+
195
+ def _notify(self, callback, value):
196
+ if callback is not None and not self._event.is_set():
197
+ try:
198
+ callback(value)
199
+ except BaseException as error:
200
+ self._callback_exception = error
201
+ raise
202
+
203
+ @contextmanager
204
+ def _drive(self):
205
+ with self._lock:
206
+ if self._thread is not None:
207
+ raise RuntimeError('Reentrant session processing is unsupported')
208
+ self._thread = threading.get_ident()
209
+ try:
210
+ yield
211
+ except BaseException as cause:
212
+ if not self._done.is_set():
213
+ if cause is self._callback_exception or not isinstance(cause, Exception):
214
+ self._error = cause
215
+ else:
216
+ self._error = VadError(str(cause), self._snapshot())
217
+ self._error.__cause__ = cause
218
+ self._cleanup()
219
+ if isinstance(self._error, VadError):
220
+ self._notify(self._on_error, self._error)
221
+ raise self._error
222
+ raise
223
+ finally:
224
+ self._thread = None
225
+ if self._event.is_set() and not self._done.is_set():
226
+ self._terminal('cancelled')
227
+
228
+ def push(self, audio, *, sample_rate=None) -> None:
229
+ if self._done.is_set():
230
+ raise RuntimeError('Session no longer accepts audio')
231
+ if isinstance(audio, (str, Path)):
232
+ raise TypeError('Live push accepts PCM audio, not file paths')
233
+ pcm = normalize_audio(audio, sample_rate)
234
+ if not len(pcm.samples):
235
+ raise ValueError('Audio must not be empty')
236
+ with self._drive():
237
+ if self._done.is_set():
238
+ raise RuntimeError('Session no longer accepts audio')
239
+ if self._event.is_set():
240
+ return
241
+ self._input += len(pcm.samples)
242
+ self._audio = np.concatenate((self._audio, self._resampler.push(pcm)))
243
+ self._pump(False)
244
+
245
+ def _samples(self, ms):
246
+ return math.ceil(ms * self._model.sample_rate / 1000)
247
+
248
+ def _ms(self, sample):
249
+ return sample * 1000 / self._model.sample_rate
250
+
251
+ def _pump(self, final):
252
+ window = self._model.window_size
253
+ while len(self._audio) and (len(self._audio) >= window or final) and not self._event.is_set():
254
+ count = min(window, len(self._audio))
255
+ audio = self._audio[:count].copy()
256
+ self._audio = self._audio[count:]
257
+ frame = np.pad(audio, (0, window - count)) if count < window else audio
258
+ probability = float(self._model._native.score_frame(frame))
259
+ if not math.isfinite(probability) or not 0 <= probability <= 1:
260
+ raise ValueError('Native VAD returned an invalid probability')
261
+ if self._event.is_set():
262
+ break
263
+ self._feed(audio, probability)
264
+
265
+ def _feed(self, audio, probability):
266
+ config = self._config
267
+ start, end = self._position, self._position + len(audio)
268
+ if config.return_audio:
269
+ self._retained = np.concatenate((self._retained, audio))
270
+ self._position = end
271
+ self._notify(self._on_probability, VadProbabilityEvent(probability, self._ms(start), self._ms(end)))
272
+ if self._event.is_set():
273
+ return
274
+ if self._active is None:
275
+ if probability >= config.speech_threshold:
276
+ if self._candidate is None:
277
+ self._candidate = start
278
+ if end - self._candidate >= self._samples(config.min_speech_duration_ms):
279
+ self._active = max(self._previous_end, 0, self._candidate - self._samples(config.speech_padding_ms))
280
+ self._notify(self._on_start, SpeechStartEvent(str(len(self._segments)), self._ms(self._active)))
281
+ else:
282
+ self._candidate = None
283
+ elif probability < config.silence_threshold:
284
+ if self._silence is None:
285
+ self._silence = start
286
+ if end - self._silence >= self._samples(config.min_silence_duration_ms):
287
+ self._close(min(end, self._silence + self._samples(config.speech_padding_ms)))
288
+ else:
289
+ self._silence = None
290
+ if config.return_audio:
291
+ keep = self._active if self._active is not None else max(self._previous_end, 0,
292
+ (self._candidate if self._candidate is not None else self._position) - self._samples(config.speech_padding_ms))
293
+ self._retained = self._retained[max(0, keep - self._retained_start):].copy()
294
+ self._retained_start = keep
295
+
296
+ def _close(self, end):
297
+ if self._active is None or self._event.is_set():
298
+ return
299
+ end = max(self._active, min(end, self._position))
300
+ ident, start = str(len(self._segments)), self._active
301
+ audio = None
302
+ if self._config.return_audio:
303
+ audio = Audio(self._retained[start - self._retained_start:end - self._retained_start], self._model.sample_rate)
304
+ segment = SpeechSegment(ident, self._ms(start), self._ms(end), audio)
305
+ self._segments.append(segment if self._file else SpeechRange(ident, segment.start_ms, segment.end_ms))
306
+ self._previous_end = end
307
+ self._active = self._candidate = self._silence = None
308
+ self._notify(self._on_end, segment)
309
+
310
+ def finish(self) -> Union[DetectionResult, VadSessionResult]:
311
+ if self._done.is_set():
312
+ return self.result()
313
+ if not self._input and not self._event.is_set():
314
+ raise ValueError('Cannot finish a session without audio')
315
+ with self._drive():
316
+ if not self._done.is_set() and not self._event.is_set():
317
+ self._audio = np.concatenate((self._audio, self._resampler.finish()))
318
+ self._pump(True)
319
+ if self._active is not None:
320
+ end = self._position if self._silence is None else min(self._position, self._silence + self._samples(self._config.speech_padding_ms))
321
+ self._close(end)
322
+ self._terminal('cancelled' if self._event.is_set() else 'complete')
323
+ return self.result()
324
+
325
+ def cancel(self) -> None:
326
+ self._event.set()
327
+ if self._lock.acquire(blocking=False):
328
+ try:
329
+ if self._thread is None:
330
+ self._terminal('cancelled')
331
+ finally:
332
+ self._lock.release()
333
+
334
+ def result(self) -> Union[DetectionResult, VadSessionResult]:
335
+ """Wait for another thread to finish/cancel; does not end input."""
336
+ if self._thread == threading.get_ident() and not self._done.is_set():
337
+ raise RuntimeError('Cannot wait for a session result from its callback')
338
+ while not self._done.wait(.02):
339
+ if self._event.is_set():
340
+ self.cancel()
341
+ if self._error is not None:
342
+ raise self._error
343
+ return self._result
344
+
345
+
346
+ def load_voice_activity_detection(model_id, *, cache_dir=None, on_progress=None, cancel_event=None) -> VoiceActivityDetectionModel:
347
+ from ._lifecycle import load_with_lifecycle
348
+ if model_id != 'snakers4/silero-vad':
349
+ raise ValueError('The probability VAD surface currently supports snakers4/silero-vad')
350
+ def initialize(lease):
351
+ legacy = load_vad_model(model_id, cache_dir=cache_dir, force_download=False)
352
+ try:
353
+ model = VoiceActivityDetectionModel(model_id, legacy._native_vad)
354
+ model._asset_lease = lease
355
+ return model
356
+ except BaseException:
357
+ legacy._native_vad.close()
358
+ raise
359
+ return load_with_lifecycle(model_id, initialize, cache_dir=cache_dir,
360
+ on_progress=on_progress, cancel_event=cancel_event)