friday-framework-speech 0.1.0a0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- friday_framework_speech-0.1.0a0.dist-info/METADATA +56 -0
- friday_framework_speech-0.1.0a0.dist-info/RECORD +15 -0
- friday_framework_speech-0.1.0a0.dist-info/WHEEL +4 -0
- friday_speech/__init__.py +44 -0
- friday_speech/adapters/__init__.py +43 -0
- friday_speech/adapters/elevenlabs.py +183 -0
- friday_speech/adapters/picovoice.py +277 -0
- friday_speech/chunking.py +75 -0
- friday_speech/config.py +364 -0
- friday_speech/errors.py +34 -0
- friday_speech/harness.py +100 -0
- friday_speech/interfaces.py +71 -0
- friday_speech/playback.py +245 -0
- friday_speech/selection.py +149 -0
- friday_speech/service.py +123 -0
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: friday-framework-speech
|
|
3
|
+
Version: 0.1.0a0
|
|
4
|
+
Summary: Speech services (TTS, wake word) for Friday
|
|
5
|
+
Project-URL: Homepage, https://github.com/CIChuck/agent-framework
|
|
6
|
+
Project-URL: Repository, https://github.com/CIChuck/agent-framework
|
|
7
|
+
Project-URL: Issues, https://github.com/CIChuck/agent-framework/issues
|
|
8
|
+
Project-URL: Documentation, https://github.com/CIChuck/agent-framework/tree/main/docs
|
|
9
|
+
Project-URL: Source, https://github.com/CIChuck/agent-framework/tree/main/packages/friday-speech
|
|
10
|
+
Author: Friday Team
|
|
11
|
+
License-Expression: MIT
|
|
12
|
+
Keywords: assistant,audio,speech,tts,wakeword
|
|
13
|
+
Requires-Python: >=3.10
|
|
14
|
+
Requires-Dist: friday-framework-core==0.1.0a0
|
|
15
|
+
Requires-Dist: pydantic>=2.0.0
|
|
16
|
+
Requires-Dist: pyyaml>=6.0.0
|
|
17
|
+
Provides-Extra: dev
|
|
18
|
+
Requires-Dist: pytest-asyncio>=0.23.0; extra == 'dev'
|
|
19
|
+
Requires-Dist: pytest>=8.0.0; extra == 'dev'
|
|
20
|
+
Provides-Extra: elevenlabs
|
|
21
|
+
Requires-Dist: elevenlabs>=1.0.0; extra == 'elevenlabs'
|
|
22
|
+
Provides-Extra: picovoice
|
|
23
|
+
Requires-Dist: pvcobra>=2.0.0; extra == 'picovoice'
|
|
24
|
+
Requires-Dist: pvleopard>=2.0.0; extra == 'picovoice'
|
|
25
|
+
Requires-Dist: pvorca>=0.1.0; extra == 'picovoice'
|
|
26
|
+
Requires-Dist: pvporcupine>=3.0.0; extra == 'picovoice'
|
|
27
|
+
Requires-Dist: pvrecorder>=1.2.0; extra == 'picovoice'
|
|
28
|
+
Requires-Dist: pvrhino>=2.0.0; extra == 'picovoice'
|
|
29
|
+
Requires-Dist: pvspeaker>=0.1.0; extra == 'picovoice'
|
|
30
|
+
Description-Content-Type: text/markdown
|
|
31
|
+
|
|
32
|
+
# Friday Framework Speech
|
|
33
|
+
|
|
34
|
+
Speech service package for the Friday Framework.
|
|
35
|
+
|
|
36
|
+
This package provides speech service abstractions and optional text-to-speech,
|
|
37
|
+
wake-word, recording, playback, and speech-provider integrations.
|
|
38
|
+
|
|
39
|
+
## Install
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
pip install friday-framework-speech
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
Optional speech providers are available through extras such as `elevenlabs` and
|
|
46
|
+
`picovoice`.
|
|
47
|
+
|
|
48
|
+
## Import Package
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
import friday_speech
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
## License
|
|
55
|
+
|
|
56
|
+
MIT
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
friday_speech/__init__.py,sha256=Bf8Z-ly2faSZB2zpSNteFqflVffHDuMNvJpf4NYcb9c,830
|
|
2
|
+
friday_speech/chunking.py,sha256=lgf2JlvV_l00R6J31T-GPvJ5GA5UfrwCMIgO5ih_8EY,2041
|
|
3
|
+
friday_speech/config.py,sha256=2yr7JxrCmsAjd_eL98v9zl0SEovFzmM21UM0DjXtBNw,11064
|
|
4
|
+
friday_speech/errors.py,sha256=82s6pYLJ53b_f_VkPOkY_13xg63hXuCDzyCig4jvFgA,778
|
|
5
|
+
friday_speech/harness.py,sha256=K5JPGBXf2mPbIR7DFbooWMJfCvftOHWJbxO__7WGaYY,3055
|
|
6
|
+
friday_speech/interfaces.py,sha256=MqGiBmxYiPIhKGKBhIsrPRpv6bcp62clAOuKxs11bMg,1700
|
|
7
|
+
friday_speech/playback.py,sha256=R712SwwNAHNIdfyDID53hCmMNZIj_7X2APigO6vbAG0,8167
|
|
8
|
+
friday_speech/selection.py,sha256=mX2kvxRY8-O-jpZbqWH5DLWg4ht9-Ju3fjkab3125-o,4562
|
|
9
|
+
friday_speech/service.py,sha256=9_RPqk2-jN3iJA3-I4krDJJcO6JzJry2FcBTsyBtE1s,4456
|
|
10
|
+
friday_speech/adapters/__init__.py,sha256=IEbOWJ04XR5XpRlbeYNhi4B5MMPsRRJv58KRvmH_Ias,1644
|
|
11
|
+
friday_speech/adapters/elevenlabs.py,sha256=WPG5ffdOyV8TgJd3I2qcPi3qwJnY0w5M2KCcgSRyyRg,6315
|
|
12
|
+
friday_speech/adapters/picovoice.py,sha256=doFCdrutVcGzaTHOFik2jp3Qp4JZbHIJw0rUBr4LW_I,9166
|
|
13
|
+
friday_framework_speech-0.1.0a0.dist-info/METADATA,sha256=ne6dfM8afdV7dB5djS6pju0sZMYObWvBGsACS5JYtIE,1822
|
|
14
|
+
friday_framework_speech-0.1.0a0.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
15
|
+
friday_framework_speech-0.1.0a0.dist-info/RECORD,,
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# mypy: ignore-errors
|
|
2
|
+
"""Friday speech package."""
|
|
3
|
+
|
|
4
|
+
from .config import SpeechConfig
|
|
5
|
+
from .errors import (
|
|
6
|
+
AdapterError,
|
|
7
|
+
DependencyError,
|
|
8
|
+
PlaybackError,
|
|
9
|
+
SelectionError,
|
|
10
|
+
SpeechConfigError,
|
|
11
|
+
SpeechError,
|
|
12
|
+
STTError,
|
|
13
|
+
WakeWordError,
|
|
14
|
+
)
|
|
15
|
+
from .harness import SpeechHarness
|
|
16
|
+
from .interfaces import (
|
|
17
|
+
AdapterRegistry,
|
|
18
|
+
AudioFormat,
|
|
19
|
+
AudioStream,
|
|
20
|
+
STTAdapter,
|
|
21
|
+
TTSAdapter,
|
|
22
|
+
WakeWordAdapter,
|
|
23
|
+
)
|
|
24
|
+
from .service import SpeechService
|
|
25
|
+
|
|
26
|
+
__all__ = [
|
|
27
|
+
"AdapterError",
|
|
28
|
+
"AdapterRegistry",
|
|
29
|
+
"AudioFormat",
|
|
30
|
+
"AudioStream",
|
|
31
|
+
"DependencyError",
|
|
32
|
+
"PlaybackError",
|
|
33
|
+
"SelectionError",
|
|
34
|
+
"SpeechConfig",
|
|
35
|
+
"SpeechConfigError",
|
|
36
|
+
"SpeechError",
|
|
37
|
+
"SpeechHarness",
|
|
38
|
+
"SpeechService",
|
|
39
|
+
"STTAdapter",
|
|
40
|
+
"STTError",
|
|
41
|
+
"TTSAdapter",
|
|
42
|
+
"WakeWordAdapter",
|
|
43
|
+
"WakeWordError",
|
|
44
|
+
]
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# mypy: ignore-errors
|
|
2
|
+
"""Adapter factory helpers for friday-speech."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
from ..config import SpeechConfig, TTSProvider, WakeWordProvider
|
|
7
|
+
from ..errors import AdapterError
|
|
8
|
+
from ..interfaces import STTAdapter, TTSAdapter, WakeWordAdapter
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def build_tts_adapter(config: SpeechConfig) -> TTSAdapter:
|
|
12
|
+
"""Create a TTS adapter for the configured provider."""
|
|
13
|
+
if config.tts.provider == TTSProvider.ELEVENLABS:
|
|
14
|
+
from .elevenlabs import ElevenLabsAdapter
|
|
15
|
+
|
|
16
|
+
return ElevenLabsAdapter(config.elevenlabs)
|
|
17
|
+
if config.tts.provider == TTSProvider.PICOVOICE:
|
|
18
|
+
from .picovoice import PicovoiceOrcaAdapter
|
|
19
|
+
|
|
20
|
+
return PicovoiceOrcaAdapter(config.picovoice)
|
|
21
|
+
raise AdapterError(f"Unsupported TTS provider: {config.tts.provider}")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def build_wakeword_adapter(config: SpeechConfig) -> WakeWordAdapter | None:
|
|
25
|
+
"""Create a wake word adapter if enabled."""
|
|
26
|
+
if not config.wakeword.enabled:
|
|
27
|
+
return None
|
|
28
|
+
if config.wakeword.provider == WakeWordProvider.PICOVOICE:
|
|
29
|
+
from .picovoice import PicovoicePorcupineAdapter
|
|
30
|
+
|
|
31
|
+
return PicovoicePorcupineAdapter(config.picovoice)
|
|
32
|
+
raise AdapterError(f"Unsupported wake word provider: {config.wakeword.provider}")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def build_stt_adapter(config: SpeechConfig) -> STTAdapter | None:
|
|
36
|
+
"""Create a speech-to-text adapter if enabled."""
|
|
37
|
+
if not config.stt.enabled:
|
|
38
|
+
return None
|
|
39
|
+
if config.stt.provider.value == "picovoice":
|
|
40
|
+
from .picovoice import PicovoiceLeopardAdapter
|
|
41
|
+
|
|
42
|
+
return PicovoiceLeopardAdapter(config.picovoice)
|
|
43
|
+
raise AdapterError(f"Unsupported STT provider: {config.stt.provider}")
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
# mypy: ignore-errors
|
|
2
|
+
"""ElevenLabs TTS adapter (best-effort streaming)."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import asyncio
|
|
7
|
+
import re
|
|
8
|
+
import threading
|
|
9
|
+
from collections.abc import AsyncIterator, Iterable
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from ..config import ElevenLabsConfig, VoiceSettingsConfig
|
|
13
|
+
from ..errors import AdapterError, DependencyError
|
|
14
|
+
from ..interfaces import AudioFormat, AudioStream, TTSAdapter
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class ElevenLabsAdapter(TTSAdapter):
|
|
18
|
+
"""ElevenLabs adapter using the official Python SDK when available."""
|
|
19
|
+
|
|
20
|
+
def __init__(self, config: ElevenLabsConfig) -> None:
|
|
21
|
+
self._config = config
|
|
22
|
+
self.output_format = _infer_audio_format(config.output_format)
|
|
23
|
+
self.sample_rate = _infer_sample_rate(config.output_format)
|
|
24
|
+
self.channels = None
|
|
25
|
+
|
|
26
|
+
async def stream(
|
|
27
|
+
self,
|
|
28
|
+
text: str,
|
|
29
|
+
voice_id: str | None,
|
|
30
|
+
settings: VoiceSettingsConfig,
|
|
31
|
+
) -> AudioStream:
|
|
32
|
+
if not self._config.api_key:
|
|
33
|
+
raise AdapterError("ElevenLabs API key is required")
|
|
34
|
+
|
|
35
|
+
elevenlabs = _import_elevenlabs()
|
|
36
|
+
client = _build_client(elevenlabs, self._config.api_key)
|
|
37
|
+
payload = _build_voice_settings(settings)
|
|
38
|
+
|
|
39
|
+
text_to_speech = getattr(client, "text_to_speech", None)
|
|
40
|
+
if text_to_speech is None:
|
|
41
|
+
raise AdapterError("ElevenLabs client missing text_to_speech interface")
|
|
42
|
+
|
|
43
|
+
if hasattr(text_to_speech, "convert_as_stream"):
|
|
44
|
+
sync_stream = text_to_speech.convert_as_stream(
|
|
45
|
+
text=text,
|
|
46
|
+
voice_id=voice_id or self._config.voice_id,
|
|
47
|
+
model_id=self._config.model_id,
|
|
48
|
+
output_format=self._config.output_format,
|
|
49
|
+
voice_settings=payload or None,
|
|
50
|
+
)
|
|
51
|
+
chunks = _async_from_sync_iterator(sync_stream)
|
|
52
|
+
return AudioStream(
|
|
53
|
+
format=self.output_format,
|
|
54
|
+
chunks=chunks,
|
|
55
|
+
sample_rate=self.sample_rate,
|
|
56
|
+
channels=self.channels,
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
if hasattr(text_to_speech, "convert"):
|
|
60
|
+
audio = await asyncio.to_thread(
|
|
61
|
+
text_to_speech.convert,
|
|
62
|
+
text=text,
|
|
63
|
+
voice_id=voice_id or self._config.voice_id,
|
|
64
|
+
model_id=self._config.model_id,
|
|
65
|
+
output_format=self._config.output_format,
|
|
66
|
+
voice_settings=payload or None,
|
|
67
|
+
)
|
|
68
|
+
chunks = _async_single_chunk(audio)
|
|
69
|
+
return AudioStream(
|
|
70
|
+
format=self.output_format,
|
|
71
|
+
chunks=chunks,
|
|
72
|
+
sample_rate=self.sample_rate,
|
|
73
|
+
channels=self.channels,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
raise AdapterError(
|
|
77
|
+
"Unsupported ElevenLabs SDK: expected text_to_speech.convert[_as_stream]"
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _import_elevenlabs() -> Any:
|
|
82
|
+
try:
|
|
83
|
+
import elevenlabs # type: ignore
|
|
84
|
+
|
|
85
|
+
return elevenlabs
|
|
86
|
+
except Exception as exc: # pragma: no cover - dependency gate
|
|
87
|
+
raise DependencyError(
|
|
88
|
+
"ElevenLabs SDK not installed. Install with: pip install elevenlabs"
|
|
89
|
+
) from exc
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _build_client(module: Any, api_key: str) -> Any:
|
|
93
|
+
for path in ("client", None):
|
|
94
|
+
try:
|
|
95
|
+
if path:
|
|
96
|
+
client_mod = __import__(f"elevenlabs.{path}", fromlist=["ElevenLabs"])
|
|
97
|
+
client_cls = getattr(client_mod, "ElevenLabs", None)
|
|
98
|
+
else:
|
|
99
|
+
client_cls = getattr(module, "ElevenLabs", None)
|
|
100
|
+
if client_cls:
|
|
101
|
+
return client_cls(api_key=api_key)
|
|
102
|
+
except Exception:
|
|
103
|
+
continue
|
|
104
|
+
raise AdapterError("Unable to initialize ElevenLabs client")
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _build_voice_settings(settings: VoiceSettingsConfig) -> dict[str, Any]:
|
|
108
|
+
payload: dict[str, Any] = {}
|
|
109
|
+
if settings.stability is not None:
|
|
110
|
+
payload["stability"] = settings.stability
|
|
111
|
+
if settings.similarity_boost is not None:
|
|
112
|
+
payload["similarity_boost"] = settings.similarity_boost
|
|
113
|
+
if settings.style is not None:
|
|
114
|
+
payload["style"] = settings.style
|
|
115
|
+
if settings.speed is not None:
|
|
116
|
+
payload["speed"] = settings.speed
|
|
117
|
+
return payload
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _infer_audio_format(output_format: str) -> AudioFormat:
|
|
121
|
+
value = output_format.lower()
|
|
122
|
+
if value.startswith("mp3"):
|
|
123
|
+
return AudioFormat.MP3
|
|
124
|
+
if value.startswith("pcm"):
|
|
125
|
+
return AudioFormat.PCM
|
|
126
|
+
raise AdapterError(f"Unsupported ElevenLabs output_format: {output_format}")
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _infer_sample_rate(output_format: str) -> int | None:
|
|
130
|
+
match = re.search(r"(\d{4,6})", output_format)
|
|
131
|
+
if not match:
|
|
132
|
+
return None
|
|
133
|
+
try:
|
|
134
|
+
return int(match.group(1))
|
|
135
|
+
except ValueError:
|
|
136
|
+
return None
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _async_single_chunk(audio: Any) -> AsyncIterator[bytes]:
|
|
140
|
+
async def _generator() -> AsyncIterator[bytes]:
|
|
141
|
+
if isinstance(audio, (bytes, bytearray)):
|
|
142
|
+
yield bytes(audio)
|
|
143
|
+
return
|
|
144
|
+
if isinstance(audio, Iterable):
|
|
145
|
+
for chunk in audio:
|
|
146
|
+
if isinstance(chunk, (bytes, bytearray)):
|
|
147
|
+
yield bytes(chunk)
|
|
148
|
+
else:
|
|
149
|
+
raise AdapterError("ElevenLabs stream yielded non-bytes chunk")
|
|
150
|
+
return
|
|
151
|
+
raise AdapterError("ElevenLabs audio payload is not bytes or iterable")
|
|
152
|
+
|
|
153
|
+
return _generator()
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _async_from_sync_iterator(stream: Iterable[Any]) -> AsyncIterator[bytes]:
|
|
157
|
+
async def _generator() -> AsyncIterator[bytes]:
|
|
158
|
+
queue: asyncio.Queue[Any] = asyncio.Queue()
|
|
159
|
+
sentinel = object()
|
|
160
|
+
loop = asyncio.get_running_loop()
|
|
161
|
+
|
|
162
|
+
def _worker() -> None:
|
|
163
|
+
try:
|
|
164
|
+
for chunk in stream:
|
|
165
|
+
loop.call_soon_threadsafe(queue.put_nowait, chunk)
|
|
166
|
+
except Exception as exc:
|
|
167
|
+
loop.call_soon_threadsafe(queue.put_nowait, exc)
|
|
168
|
+
finally:
|
|
169
|
+
loop.call_soon_threadsafe(queue.put_nowait, sentinel)
|
|
170
|
+
|
|
171
|
+
threading.Thread(target=_worker, daemon=True).start()
|
|
172
|
+
|
|
173
|
+
while True:
|
|
174
|
+
item = await queue.get()
|
|
175
|
+
if item is sentinel:
|
|
176
|
+
break
|
|
177
|
+
if isinstance(item, Exception):
|
|
178
|
+
raise item
|
|
179
|
+
if not isinstance(item, (bytes, bytearray)):
|
|
180
|
+
raise AdapterError("ElevenLabs stream yielded non-bytes chunk")
|
|
181
|
+
yield bytes(item)
|
|
182
|
+
|
|
183
|
+
return _generator()
|
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
# mypy: ignore-errors
|
|
2
|
+
"""Picovoice adapters for wake word and TTS."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import asyncio
|
|
7
|
+
import unicodedata
|
|
8
|
+
from array import array
|
|
9
|
+
from collections.abc import AsyncIterator
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from friday_core.logging import get_logger
|
|
13
|
+
|
|
14
|
+
from ..config import PicovoiceConfig, VoiceSettingsConfig
|
|
15
|
+
from ..errors import AdapterError, DependencyError, STTError, WakeWordError
|
|
16
|
+
from ..interfaces import (
|
|
17
|
+
AudioFormat,
|
|
18
|
+
AudioStream,
|
|
19
|
+
STTAdapter,
|
|
20
|
+
TTSAdapter,
|
|
21
|
+
WakeWordAdapter,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
logger = get_logger(__name__)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class PicovoiceOrcaAdapter(TTSAdapter):
|
|
28
|
+
"""Picovoice Orca TTS adapter."""
|
|
29
|
+
|
|
30
|
+
def __init__(self, config: PicovoiceConfig) -> None:
|
|
31
|
+
self._config = config
|
|
32
|
+
self._logger = get_logger(__name__)
|
|
33
|
+
self.output_format = AudioFormat.PCM
|
|
34
|
+
self.sample_rate = None
|
|
35
|
+
self.channels = 1
|
|
36
|
+
|
|
37
|
+
async def stream(
|
|
38
|
+
self,
|
|
39
|
+
text: str,
|
|
40
|
+
voice_id: str | None,
|
|
41
|
+
settings: VoiceSettingsConfig,
|
|
42
|
+
) -> AudioStream:
|
|
43
|
+
if not self._config.access_key:
|
|
44
|
+
raise AdapterError("Picovoice access_key is required")
|
|
45
|
+
|
|
46
|
+
try:
|
|
47
|
+
import pvorca # type: ignore
|
|
48
|
+
except Exception as exc: # pragma: no cover - dependency gate
|
|
49
|
+
raise DependencyError(
|
|
50
|
+
"Picovoice Orca not installed. Install with: pip install pvorca"
|
|
51
|
+
) from exc
|
|
52
|
+
|
|
53
|
+
create_kwargs: dict[str, Any] = {
|
|
54
|
+
"access_key": self._config.access_key,
|
|
55
|
+
"model_path": self._config.orca.model_path,
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
try:
|
|
59
|
+
import inspect
|
|
60
|
+
|
|
61
|
+
supported = inspect.signature(pvorca.create).parameters
|
|
62
|
+
filtered = {
|
|
63
|
+
k: v
|
|
64
|
+
for k, v in create_kwargs.items()
|
|
65
|
+
if k in supported and v is not None
|
|
66
|
+
}
|
|
67
|
+
dropped = {k for k in create_kwargs if k not in filtered}
|
|
68
|
+
if dropped:
|
|
69
|
+
self._logger.warning(
|
|
70
|
+
"Picovoice Orca create() does not support parameters; ignoring.",
|
|
71
|
+
dropped=sorted(dropped),
|
|
72
|
+
)
|
|
73
|
+
orca = pvorca.create(**filtered)
|
|
74
|
+
except TypeError as exc:
|
|
75
|
+
self._logger.error(
|
|
76
|
+
"Picovoice Orca create() failed with configured parameters; retrying with minimal args.", # noqa: E501
|
|
77
|
+
error=str(exc),
|
|
78
|
+
)
|
|
79
|
+
orca = pvorca.create(
|
|
80
|
+
access_key=self._config.access_key,
|
|
81
|
+
model_path=self._config.orca.model_path,
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
try:
|
|
85
|
+
sanitized = _sanitize_for_orca(
|
|
86
|
+
text, getattr(orca, "valid_characters", None)
|
|
87
|
+
)
|
|
88
|
+
pcm_data = orca.synthesize(
|
|
89
|
+
sanitized, speech_rate=self._config.orca.speech_rate
|
|
90
|
+
)
|
|
91
|
+
if isinstance(pcm_data, tuple):
|
|
92
|
+
pcm_data = pcm_data[0] if pcm_data else None
|
|
93
|
+
pcm_bytes = _ensure_pcm_bytes(pcm_data)
|
|
94
|
+
|
|
95
|
+
self.sample_rate = getattr(orca, "sample_rate", None)
|
|
96
|
+
|
|
97
|
+
finally:
|
|
98
|
+
if hasattr(orca, "delete"):
|
|
99
|
+
orca.delete()
|
|
100
|
+
|
|
101
|
+
async def _generator() -> AsyncIterator[bytes]:
|
|
102
|
+
yield pcm_bytes
|
|
103
|
+
|
|
104
|
+
if not pcm_bytes:
|
|
105
|
+
raise AdapterError("Picovoice Orca returned no audio")
|
|
106
|
+
|
|
107
|
+
return AudioStream(
|
|
108
|
+
format=self.output_format,
|
|
109
|
+
chunks=_generator(),
|
|
110
|
+
sample_rate=self.sample_rate,
|
|
111
|
+
channels=self.channels,
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _ensure_pcm_bytes(payload: Any) -> bytes:
|
|
116
|
+
if payload is None:
|
|
117
|
+
return b""
|
|
118
|
+
if isinstance(payload, (bytes, bytearray)):
|
|
119
|
+
return bytes(payload)
|
|
120
|
+
if isinstance(payload, memoryview):
|
|
121
|
+
return payload.tobytes()
|
|
122
|
+
if isinstance(payload, array):
|
|
123
|
+
return payload.tobytes()
|
|
124
|
+
if isinstance(payload, list):
|
|
125
|
+
if not payload:
|
|
126
|
+
return b""
|
|
127
|
+
if any(isinstance(item, float) for item in payload):
|
|
128
|
+
max_abs = max(abs(float(item)) for item in payload)
|
|
129
|
+
if max_abs <= 1.0:
|
|
130
|
+
scaled = [
|
|
131
|
+
int(max(-1.0, min(1.0, float(item))) * 32767) for item in payload
|
|
132
|
+
]
|
|
133
|
+
return array("h", scaled).tobytes()
|
|
134
|
+
return array("h", payload).tobytes()
|
|
135
|
+
if isinstance(payload, tuple):
|
|
136
|
+
for item in payload:
|
|
137
|
+
try:
|
|
138
|
+
return _ensure_pcm_bytes(item)
|
|
139
|
+
except AdapterError:
|
|
140
|
+
continue
|
|
141
|
+
return array("h", payload).tobytes()
|
|
142
|
+
if isinstance(payload, dict):
|
|
143
|
+
for key in ("pcm", "pcm_data", "audio", "data"):
|
|
144
|
+
if key in payload:
|
|
145
|
+
return _ensure_pcm_bytes(payload[key])
|
|
146
|
+
if hasattr(payload, "pcm"):
|
|
147
|
+
return _ensure_pcm_bytes(payload.pcm)
|
|
148
|
+
if hasattr(payload, "tobytes"):
|
|
149
|
+
try:
|
|
150
|
+
return payload.tobytes()
|
|
151
|
+
except Exception:
|
|
152
|
+
pass
|
|
153
|
+
try:
|
|
154
|
+
import numpy as np # type: ignore
|
|
155
|
+
|
|
156
|
+
if isinstance(payload, np.ndarray):
|
|
157
|
+
if payload.dtype != np.int16:
|
|
158
|
+
payload = payload.astype(np.int16)
|
|
159
|
+
return payload.tobytes()
|
|
160
|
+
except Exception:
|
|
161
|
+
pass
|
|
162
|
+
logger.error(
|
|
163
|
+
"Unsupported PCM payload from Picovoice Orca",
|
|
164
|
+
payload_type=type(payload).__name__,
|
|
165
|
+
)
|
|
166
|
+
raise AdapterError("Unsupported PCM payload from Picovoice Orca")
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _sanitize_for_orca(text: str, valid_characters: str | None) -> str:
|
|
170
|
+
if not text:
|
|
171
|
+
return ""
|
|
172
|
+
normalized = unicodedata.normalize("NFKC", text)
|
|
173
|
+
translation = str.maketrans(
|
|
174
|
+
{
|
|
175
|
+
"“": '"',
|
|
176
|
+
"”": '"',
|
|
177
|
+
"‘": "'",
|
|
178
|
+
"’": "'",
|
|
179
|
+
"—": "-",
|
|
180
|
+
"–": "-",
|
|
181
|
+
"…": "...",
|
|
182
|
+
"\u00a0": " ",
|
|
183
|
+
}
|
|
184
|
+
)
|
|
185
|
+
normalized = normalized.translate(translation)
|
|
186
|
+
if not valid_characters:
|
|
187
|
+
return normalized
|
|
188
|
+
allowed = set(valid_characters)
|
|
189
|
+
sanitized = "".join(ch if ch in allowed else " " for ch in normalized)
|
|
190
|
+
return " ".join(sanitized.split())
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
class PicovoicePorcupineAdapter(WakeWordAdapter):
|
|
194
|
+
"""Picovoice Porcupine wake word adapter."""
|
|
195
|
+
|
|
196
|
+
def __init__(self, config: PicovoiceConfig) -> None:
|
|
197
|
+
self._config = config
|
|
198
|
+
self._logger = get_logger(__name__)
|
|
199
|
+
self._porcupine = None
|
|
200
|
+
self._recorder = None
|
|
201
|
+
self._active = False
|
|
202
|
+
|
|
203
|
+
async def wait_for_wakeword(self) -> None:
|
|
204
|
+
if not self._config.access_key:
|
|
205
|
+
raise WakeWordError("Picovoice access_key is required")
|
|
206
|
+
await asyncio.to_thread(self._blocking_wait)
|
|
207
|
+
|
|
208
|
+
def _blocking_wait(self) -> None:
|
|
209
|
+
if not self._active:
|
|
210
|
+
self._initialize()
|
|
211
|
+
if not self._porcupine or not self._recorder:
|
|
212
|
+
raise WakeWordError("Wake word adapter not initialized")
|
|
213
|
+
while True:
|
|
214
|
+
pcm = self._recorder.read()
|
|
215
|
+
result = self._porcupine.process(pcm)
|
|
216
|
+
if result >= 0:
|
|
217
|
+
return
|
|
218
|
+
|
|
219
|
+
def _initialize(self) -> None:
|
|
220
|
+
try:
|
|
221
|
+
import pvporcupine # type: ignore
|
|
222
|
+
from pvrecorder import PvRecorder # type: ignore
|
|
223
|
+
except Exception as exc: # pragma: no cover - dependency gate
|
|
224
|
+
raise DependencyError(
|
|
225
|
+
"Picovoice Porcupine not installed. Install with: pip install pvporcupine pvrecorder" # noqa: E501
|
|
226
|
+
) from exc
|
|
227
|
+
|
|
228
|
+
porcupine_config = self._config.porcupine
|
|
229
|
+
keywords = porcupine_config.keywords
|
|
230
|
+
keyword_paths = porcupine_config.keyword_paths
|
|
231
|
+
|
|
232
|
+
if not keywords and not keyword_paths:
|
|
233
|
+
raise WakeWordError("Porcupine requires keywords or keyword_paths")
|
|
234
|
+
|
|
235
|
+
self._porcupine = pvporcupine.create(
|
|
236
|
+
access_key=self._config.access_key,
|
|
237
|
+
keywords=keywords if keywords else None,
|
|
238
|
+
keyword_paths=keyword_paths if keyword_paths else None,
|
|
239
|
+
sensitivities=porcupine_config.sensitivities,
|
|
240
|
+
model_path=porcupine_config.model_path,
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
frame_length = self._porcupine.frame_length
|
|
244
|
+
recorder_config = self._config.recorder
|
|
245
|
+
self._recorder = PvRecorder(
|
|
246
|
+
device_index=recorder_config.device_index,
|
|
247
|
+
frame_length=recorder_config.frame_length or frame_length,
|
|
248
|
+
)
|
|
249
|
+
self._recorder.start()
|
|
250
|
+
self._active = True
|
|
251
|
+
|
|
252
|
+
async def close(self) -> None:
|
|
253
|
+
await asyncio.to_thread(self._shutdown)
|
|
254
|
+
|
|
255
|
+
def _shutdown(self) -> None:
|
|
256
|
+
if self._recorder:
|
|
257
|
+
self._recorder.stop()
|
|
258
|
+
if hasattr(self._recorder, "delete"):
|
|
259
|
+
self._recorder.delete()
|
|
260
|
+
if self._porcupine and hasattr(self._porcupine, "delete"):
|
|
261
|
+
self._porcupine.delete()
|
|
262
|
+
self._recorder = None
|
|
263
|
+
self._porcupine = None
|
|
264
|
+
self._active = False
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
class PicovoiceLeopardAdapter(STTAdapter):
|
|
268
|
+
"""Picovoice Leopard speech-to-text adapter (capture handled externally)."""
|
|
269
|
+
|
|
270
|
+
def __init__(self, config: PicovoiceConfig) -> None:
|
|
271
|
+
self._config = config
|
|
272
|
+
self._logger = get_logger(__name__)
|
|
273
|
+
|
|
274
|
+
async def transcribe(self) -> str:
|
|
275
|
+
raise STTError(
|
|
276
|
+
"Picovoice STT is not wired yet. Provide a custom STT adapter or implement capture." # noqa: E501
|
|
277
|
+
)
|