livekit-plugins-vakyam 1.8.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,45 @@
1
+ # Copyright 2025 LiveKit, Inc.
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """Vakyam AI plugin for LiveKit Agents
16
+
17
+ Support for text-to-speech with [Vakyam AI](https://vakyam.ai/) Raaga 1.
18
+
19
+ See https://docs.vakyam.ai/guides/realtime-websocket for protocol details.
20
+ """
21
+
22
+ from .tts import TTS, ChunkedStream, SynthesizeStream
23
+ from .version import __version__
24
+
25
+ __all__ = ["TTS", "ChunkedStream", "SynthesizeStream", "__version__"]
26
+
27
+ from livekit.agents import Plugin
28
+
29
+ from .log import logger
30
+
31
+
32
+ class VakyamPlugin(Plugin):
33
+ def __init__(self) -> None:
34
+ super().__init__(__name__, __version__, __package__, logger)
35
+
36
+
37
+ Plugin.register_plugin(VakyamPlugin())
38
+
39
+ _module = dir()
40
+ NOT_IN_ALL = [m for m in _module if m not in __all__]
41
+
42
+ __pdoc__ = {}
43
+
44
+ for n in NOT_IN_ALL:
45
+ __pdoc__[n] = False
@@ -0,0 +1,214 @@
1
+ # Copyright 2025 LiveKit, Inc.
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ from __future__ import annotations
16
+
17
+ import json
18
+ from typing import Any
19
+ from urllib.parse import urlparse
20
+
21
+ from livekit.agents import APIStatusError
22
+
23
+ from .models import (
24
+ CUSTOM_VOICE_PREFIX,
25
+ MAX_SPEED,
26
+ MAX_TEXT_CHARACTERS,
27
+ MIN_SPEED,
28
+ SUPPORTED_LANGUAGES,
29
+ SUPPORTED_MODELS,
30
+ SUPPORTED_SAMPLE_RATES,
31
+ TTS_STREAM_PATH,
32
+ TTS_WEBSOCKET_PATH,
33
+ )
34
+
35
+ _RETRYABLE_WS_CODES = {
36
+ "rate_limit_exceeded",
37
+ "concurrency_limit_exceeded",
38
+ "tts_workers_busy",
39
+ "tts_workers_unconfigured",
40
+ "tts_workers_unavailable",
41
+ "internal_error",
42
+ }
43
+
44
+
45
+ def normalize_voice(voice: str) -> str:
46
+ """Normalize a preset voice name or custom ``vc_`` voice ID."""
47
+ if not isinstance(voice, str) or not voice.strip():
48
+ raise ValueError("voice must be a non-empty string")
49
+ normalized = voice.strip()
50
+ if normalized == CUSTOM_VOICE_PREFIX:
51
+ raise ValueError("custom voice IDs must include a value after 'vc_'")
52
+ return normalized
53
+
54
+
55
+ def normalize_base_url(base_url: str, *, allow_insecure_base_url: bool = False) -> str:
56
+ """Normalize and validate a public API base URL."""
57
+ normalized = base_url.rstrip("/")
58
+ parsed = urlparse(normalized)
59
+ if parsed.scheme != "http":
60
+ return normalized
61
+
62
+ host = parsed.hostname or ""
63
+ if allow_insecure_base_url or host in {"localhost", "127.0.0.1", "::1"}:
64
+ return normalized
65
+
66
+ raise ValueError(
67
+ "base_url must use HTTPS unless it points to localhost. "
68
+ "Pass allow_insecure_base_url=True only for trusted development networks."
69
+ )
70
+
71
+
72
+ def websocket_url(base_url: str) -> str:
73
+ """Build the Vakyam TTS WebSocket URL from an HTTP(S) or WS(S) base URL."""
74
+ root = base_url.rstrip("/")
75
+ if root.startswith("https://"):
76
+ root = "wss://" + root[len("https://") :]
77
+ elif root.startswith("http://"):
78
+ root = "ws://" + root[len("http://") :]
79
+ elif not root.startswith(("ws://", "wss://")):
80
+ root = "wss://" + root
81
+ if root.endswith(TTS_WEBSOCKET_PATH):
82
+ return root
83
+ return root + TTS_WEBSOCKET_PATH
84
+
85
+
86
+ def http_stream_url(base_url: str) -> str:
87
+ """Build the Vakyam HTTP streaming TTS URL."""
88
+ return base_url.rstrip("/") + TTS_STREAM_PATH
89
+
90
+
91
+ def validate_tts_options(
92
+ *,
93
+ model: str,
94
+ language: str,
95
+ sample_rate: int,
96
+ speed: float,
97
+ voice: str,
98
+ ) -> None:
99
+ """Validate constructor / update_options values."""
100
+ if model not in SUPPORTED_MODELS:
101
+ valid = ", ".join(sorted(SUPPORTED_MODELS))
102
+ raise ValueError(f"model '{model}' is not supported. Valid values are: {valid}.")
103
+ if language not in SUPPORTED_LANGUAGES:
104
+ valid = ", ".join(sorted(SUPPORTED_LANGUAGES))
105
+ raise ValueError(f"language '{language}' is not supported. Valid values are: {valid}.")
106
+ if sample_rate not in SUPPORTED_SAMPLE_RATES:
107
+ valid = ", ".join(str(v) for v in sorted(SUPPORTED_SAMPLE_RATES))
108
+ raise ValueError(
109
+ f"sample_rate '{sample_rate}' is not supported. Valid values are: {valid}."
110
+ )
111
+ if not MIN_SPEED <= speed <= MAX_SPEED:
112
+ raise ValueError(f"speed must be between {MIN_SPEED} and {MAX_SPEED}")
113
+ normalize_voice(voice)
114
+
115
+
116
+ def validate_text(text: str) -> None:
117
+ """Validate utterance text for a single synthesis request."""
118
+ if not isinstance(text, str) or not text:
119
+ raise ValueError("text is required")
120
+ character_count = len(text)
121
+ if character_count > MAX_TEXT_CHARACTERS:
122
+ raise ValueError(
123
+ f"Input text is {character_count} characters. Maximum allowed is "
124
+ f"{MAX_TEXT_CHARACTERS} Unicode characters."
125
+ )
126
+
127
+
128
+ def split_text(text: str, *, max_characters: int = MAX_TEXT_CHARACTERS) -> list[str]:
129
+ """Split an oversized utterance at whitespace, falling back to a hard boundary."""
130
+ if max_characters <= 0:
131
+ raise ValueError("max_characters must be greater than zero")
132
+
133
+ remaining = text.strip()
134
+ chunks: list[str] = []
135
+ while len(remaining) > max_characters:
136
+ split_at = remaining.rfind(" ", 0, max_characters + 1)
137
+ if split_at <= 0:
138
+ split_at = max_characters
139
+ chunk = remaining[:split_at].strip()
140
+ if chunk:
141
+ chunks.append(chunk)
142
+ remaining = remaining[split_at:].lstrip()
143
+ if remaining:
144
+ chunks.append(remaining)
145
+ return chunks
146
+
147
+
148
+ def speech_payload(
149
+ *,
150
+ text: str,
151
+ model: str,
152
+ voice: str,
153
+ language: str,
154
+ sample_rate: int,
155
+ speed: float,
156
+ output_format: str = "pcm",
157
+ ) -> dict[str, Any]:
158
+ """JSON body for HTTP generate/stream requests."""
159
+ validate_text(text)
160
+ validate_tts_options(
161
+ model=model, language=language, sample_rate=sample_rate, speed=speed, voice=voice
162
+ )
163
+ return {
164
+ "text": text,
165
+ "model_id": model,
166
+ "voice": normalize_voice(voice),
167
+ "language": language,
168
+ "output_format": output_format,
169
+ "sample_rate": sample_rate,
170
+ "speed": speed,
171
+ }
172
+
173
+
174
+ def raise_http_error(status: int, body: str) -> None:
175
+ """Raise ``APIStatusError`` from a Vakyam HTTP error envelope."""
176
+ parsed: object | None = None
177
+ try:
178
+ parsed = json.loads(body) if body else None
179
+ except json.JSONDecodeError:
180
+ parsed = None
181
+
182
+ error_code: str | None = None
183
+ if isinstance(parsed, dict) and isinstance(parsed.get("error"), dict):
184
+ error = parsed["error"]
185
+ code = error.get("code")
186
+ if isinstance(code, (str, int)):
187
+ error_code = str(code)
188
+
189
+ message = f"Vakyam TTS request failed with status {status}"
190
+ safe_body: dict[str, object] = {"status_code": status}
191
+ if error_code is not None:
192
+ message += f" (error code: {error_code})"
193
+ safe_body["error_code"] = error_code
194
+
195
+ raise APIStatusError(message, status_code=status, body=safe_body)
196
+
197
+
198
+ def raise_ws_error(data: dict[str, Any]) -> None:
199
+ """Raise ``APIStatusError`` from a Vakyam WebSocket ``error`` frame."""
200
+ error = data.get("error") if isinstance(data.get("error"), dict) else {}
201
+ code = error.get("code") if isinstance(error, dict) else None
202
+ error_code = str(code) if isinstance(code, (str, int)) else None
203
+ retryable = error_code in _RETRYABLE_WS_CODES
204
+ message = "Vakyam TTS WebSocket request failed"
205
+ safe_body: dict[str, str] = {"type": "error"}
206
+ if error_code is not None:
207
+ message += f" (error code: {error_code})"
208
+ safe_body["code"] = error_code
209
+ raise APIStatusError(
210
+ message,
211
+ status_code=-1,
212
+ body=safe_body,
213
+ retryable=retryable,
214
+ )
@@ -0,0 +1,384 @@
1
+ # Copyright 2025 LiveKit, Inc.
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """Vakyam realtime WebSocket protocol for sentence-wise TTS.
16
+
17
+ Copied from the Vakyam Python SDK protocol (config → text → binary* →
18
+ end_of_utterance, with cancel + drain on barge-in). Implemented here directly
19
+ so this plugin does not wrap ``vakyamai``.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import asyncio
25
+ import json
26
+ from collections.abc import AsyncIterator
27
+ from contextlib import suppress
28
+ from dataclasses import dataclass
29
+ from typing import Any
30
+
31
+ from livekit.agents import (
32
+ APIConnectionError,
33
+ APIStatusError,
34
+ APITimeoutError,
35
+ __version__ as livekit_version,
36
+ )
37
+
38
+ from ._utils import (
39
+ normalize_base_url,
40
+ normalize_voice,
41
+ raise_ws_error,
42
+ validate_text,
43
+ validate_tts_options,
44
+ websocket_url,
45
+ )
46
+ from .models import DEFAULT_BASE_URL, DEFAULT_LANGUAGE, DEFAULT_MODEL, DEFAULT_SAMPLE_RATE
47
+ from .version import __version__
48
+
49
+ USER_AGENT = f"LiveKit-Agents-Vakyam/{__version__} livekit-agents/{livekit_version}"
50
+
51
+
52
+ @dataclass(frozen=True)
53
+ class TTSSessionConfig:
54
+ """Immutable synthesis settings applied once per WebSocket session."""
55
+
56
+ model: str = DEFAULT_MODEL
57
+ voice: str = "Archana"
58
+ language: str = DEFAULT_LANGUAGE
59
+ sample_rate: int = DEFAULT_SAMPLE_RATE
60
+ speed: float = 1.0
61
+ output_format: str = "pcm"
62
+
63
+ def to_wire_message(self) -> dict[str, Any]:
64
+ return {
65
+ "type": "config",
66
+ "model_id": self.model,
67
+ "voice": normalize_voice(self.voice),
68
+ "language": self.language,
69
+ "output_format": self.output_format,
70
+ "sample_rate": self.sample_rate,
71
+ "speed": self.speed,
72
+ }
73
+
74
+
75
+ @dataclass(frozen=True)
76
+ class WebSocketSpeechResult:
77
+ """Terminal result for one utterance (``end_of_utterance`` or ``cancellation``)."""
78
+
79
+ characters_used: int
80
+ duration_seconds: float
81
+ truncated: bool = False
82
+ truncation_reason: str | None = None
83
+ cancelled: bool = False
84
+
85
+
86
+ def text_message(text: str) -> dict[str, str]:
87
+ validate_text(text)
88
+ return {"type": "text", "text": text}
89
+
90
+
91
+ def cancel_message() -> dict[str, str]:
92
+ return {"type": "cancel"}
93
+
94
+
95
+ def ping_message() -> dict[str, str]:
96
+ return {"type": "ping"}
97
+
98
+
99
+ def disconnect_message() -> dict[str, str]:
100
+ return {"type": "disconnect"}
101
+
102
+
103
+ def speech_result_from_terminal(data: dict[str, Any]) -> WebSocketSpeechResult:
104
+ msg_type = data.get("type")
105
+ if msg_type == "cancellation":
106
+ return WebSocketSpeechResult(
107
+ characters_used=int(data.get("characters_used", 0)),
108
+ duration_seconds=float(data.get("duration_seconds", 0.0)),
109
+ cancelled=True,
110
+ )
111
+ if msg_type == "end_of_utterance":
112
+ return WebSocketSpeechResult(
113
+ characters_used=int(data["characters_used"]),
114
+ duration_seconds=float(data["duration_seconds"]),
115
+ truncated=bool(data.get("truncated", False)),
116
+ truncation_reason=(str(data["reason"]) if data.get("reason") is not None else None),
117
+ )
118
+ raise APIConnectionError(f"Not a terminal WebSocket message type: {msg_type}")
119
+
120
+
121
+ class AsyncStreamingTTSSession:
122
+ """Low-level async WebSocket session that streams PCM audio per utterance.
123
+
124
+ Protocol:
125
+ connected → config → configured →
126
+ (text → binary* → end_of_utterance | cancel → binary* → cancellation)* →
127
+ disconnect
128
+
129
+ After ``cancel``, keep receiving until ``cancellation`` (drain trailing
130
+ audio). Do not send the next ``text`` until then. The socket stays open.
131
+ """
132
+
133
+ def __init__(
134
+ self,
135
+ *,
136
+ api_key: str,
137
+ config: TTSSessionConfig,
138
+ base_url: str | None = None,
139
+ allow_insecure_base_url: bool = False,
140
+ ) -> None:
141
+ if not api_key:
142
+ raise ValueError("api_key is required")
143
+ validate_tts_options(
144
+ model=config.model,
145
+ language=config.language,
146
+ sample_rate=config.sample_rate,
147
+ speed=config.speed,
148
+ voice=config.voice,
149
+ )
150
+ resolved_base = normalize_base_url(
151
+ base_url or DEFAULT_BASE_URL, allow_insecure_base_url=allow_insecure_base_url
152
+ )
153
+ self._api_key = api_key
154
+ self._config = config
155
+ self._url = websocket_url(resolved_base)
156
+ self._connection: Any = None
157
+ self._utterance_active = False
158
+ self.last_result: WebSocketSpeechResult | None = None
159
+
160
+ @property
161
+ def sample_rate(self) -> int:
162
+ return int(self._config.sample_rate)
163
+
164
+ @property
165
+ def connected(self) -> bool:
166
+ return self._connection is not None
167
+
168
+ @property
169
+ def utterance_active(self) -> bool:
170
+ """True while waiting for ``end_of_utterance`` or ``cancellation``."""
171
+ return self._utterance_active
172
+
173
+ async def __aenter__(self) -> AsyncStreamingTTSSession:
174
+ await self.connect()
175
+ return self
176
+
177
+ async def __aexit__(
178
+ self, exc_type: type[BaseException] | None, exc: BaseException | None, traceback: object
179
+ ) -> None:
180
+ await self.close()
181
+
182
+ async def connect(self, *, timeout: float = 10.0) -> None:
183
+ if self._connection is not None:
184
+ return
185
+ try:
186
+ from websockets.asyncio.client import connect
187
+ except ImportError as exc:
188
+ raise APIConnectionError("websockets is required for Vakyam TTS streaming") from exc
189
+
190
+ async def _connect_and_configure() -> None:
191
+ self._connection = await connect(
192
+ self._url,
193
+ additional_headers={
194
+ "Authorization": f"Bearer {self._api_key}",
195
+ "User-Agent": USER_AGENT,
196
+ },
197
+ open_timeout=timeout,
198
+ close_timeout=timeout,
199
+ )
200
+ connected = json.loads(await self._connection.recv())
201
+ if connected.get("type") != "connected":
202
+ raise APIConnectionError("Unexpected WebSocket handshake response")
203
+
204
+ await self._connection.send(json.dumps(self._config.to_wire_message()))
205
+ configured = json.loads(await self._connection.recv())
206
+ if configured.get("type") == "error":
207
+ raise_ws_error(configured)
208
+ if configured.get("type") != "configured":
209
+ raise APIConnectionError("Unexpected WebSocket config response")
210
+
211
+ try:
212
+ await asyncio.wait_for(_connect_and_configure(), timeout=timeout)
213
+ except asyncio.TimeoutError as exc:
214
+ await self.close()
215
+ raise APITimeoutError("Vakyam TTS WebSocket connection timed out") from exc
216
+ except (APIStatusError, APIConnectionError):
217
+ await self.close()
218
+ raise
219
+ except Exception as exc:
220
+ await self.close()
221
+ raise _websocket_connection_error(
222
+ exc, message="Vakyam TTS WebSocket connection failed"
223
+ ) from exc
224
+
225
+ async def synthesize_stream(self, text: str, *, timeout: float = 10.0) -> AsyncIterator[bytes]:
226
+ """Send one utterance and yield raw audio until EOU or cancellation.
227
+
228
+ If this coroutine is cancelled mid-stream (LiveKit barge-in), it sends
229
+ ``cancel``, drains until ``cancellation``, then re-raises so the
230
+ WebSocket can be reused.
231
+ """
232
+ if self._connection is None:
233
+ raise APIConnectionError("WebSocket session is not connected")
234
+
235
+ payload = json.dumps(text_message(text))
236
+ self.last_result = None
237
+ self._utterance_active = True
238
+
239
+ try:
240
+ await asyncio.wait_for(self._connection.send(payload), timeout=timeout)
241
+ while True:
242
+ message = await asyncio.wait_for(self._connection.recv(), timeout=timeout)
243
+ if isinstance(message, bytes):
244
+ if message:
245
+ yield message
246
+ continue
247
+
248
+ data = json.loads(message)
249
+ msg_type = data.get("type")
250
+ if msg_type == "error":
251
+ raise_ws_error(data)
252
+ if msg_type == "pong":
253
+ continue
254
+ if msg_type in {"end_of_utterance", "cancellation"}:
255
+ self.last_result = speech_result_from_terminal(data)
256
+ return
257
+ raise APIConnectionError(f"Unexpected WebSocket message type: {msg_type}")
258
+ except asyncio.CancelledError:
259
+ await asyncio.shield(self._abort_and_drain(timeout=timeout))
260
+ raise
261
+ except asyncio.TimeoutError as exc:
262
+ raise APITimeoutError("Vakyam TTS WebSocket receive timed out") from exc
263
+ except (APIStatusError, APIConnectionError, APITimeoutError):
264
+ raise
265
+ except Exception as exc:
266
+ raise _websocket_connection_error(
267
+ exc, message="Vakyam TTS WebSocket synthesis failed"
268
+ ) from exc
269
+ finally:
270
+ self._utterance_active = False
271
+
272
+ async def cancel(
273
+ self, *, drain: bool = True, timeout: float = 10.0
274
+ ) -> WebSocketSpeechResult | None:
275
+ """Send barge-in ``cancel`` and optionally drain until ``cancellation``."""
276
+ if self._connection is None:
277
+ raise APIConnectionError("WebSocket session is not connected")
278
+
279
+ try:
280
+ await asyncio.wait_for(
281
+ self._connection.send(json.dumps(cancel_message())), timeout=timeout
282
+ )
283
+ except asyncio.TimeoutError as exc:
284
+ raise APITimeoutError("Vakyam TTS WebSocket cancellation timed out") from exc
285
+ if not drain:
286
+ return None
287
+ return await self._drain_until_terminal(timeout=timeout)
288
+
289
+ async def ping(self, *, timeout: float = 10.0) -> bool:
290
+ if self._connection is None:
291
+ raise APIConnectionError("WebSocket session is not connected")
292
+ try:
293
+ await asyncio.wait_for(
294
+ self._connection.send(json.dumps(ping_message())), timeout=timeout
295
+ )
296
+ response = json.loads(await asyncio.wait_for(self._connection.recv(), timeout=timeout))
297
+ except asyncio.TimeoutError as exc:
298
+ raise APITimeoutError("Vakyam TTS WebSocket ping timed out") from exc
299
+ except Exception as exc:
300
+ raise _websocket_connection_error(
301
+ exc, message="Vakyam TTS WebSocket ping failed"
302
+ ) from exc
303
+
304
+ if response.get("type") == "error":
305
+ raise_ws_error(response)
306
+ if response.get("type") != "pong":
307
+ raise APIConnectionError(
308
+ f"Unexpected Vakyam TTS WebSocket ping response: {response.get('type')}"
309
+ )
310
+ return True
311
+
312
+ async def close(self, *, timeout: float = 10.0) -> None:
313
+ if self._connection is None:
314
+ return
315
+ try:
316
+ with suppress(Exception):
317
+ await asyncio.wait_for(
318
+ self._connection.send(json.dumps(disconnect_message())), timeout=timeout
319
+ )
320
+ with suppress(Exception):
321
+ await asyncio.wait_for(self._connection.close(), timeout=timeout)
322
+ finally:
323
+ self._connection = None
324
+ self._utterance_active = False
325
+
326
+ async def _abort_and_drain(self, *, timeout: float) -> None:
327
+ """Best-effort cancel + drain after the consumer task was cancelled."""
328
+ if self._connection is None:
329
+ return
330
+ with suppress(Exception):
331
+ await asyncio.wait_for(
332
+ self._connection.send(json.dumps(cancel_message())), timeout=timeout
333
+ )
334
+ with suppress(Exception):
335
+ self.last_result = await self._drain_until_terminal(timeout=timeout)
336
+
337
+ async def _drain_until_terminal(self, *, timeout: float) -> WebSocketSpeechResult:
338
+ if self._connection is None:
339
+ raise APIConnectionError("WebSocket session is not connected")
340
+
341
+ while True:
342
+ try:
343
+ message = await asyncio.wait_for(self._connection.recv(), timeout=timeout)
344
+ except asyncio.TimeoutError as exc:
345
+ raise APITimeoutError("Vakyam TTS WebSocket cancellation timed out") from exc
346
+ if isinstance(message, bytes):
347
+ continue
348
+ data = json.loads(message)
349
+ msg_type = data.get("type")
350
+ if msg_type == "error":
351
+ raise_ws_error(data)
352
+ if msg_type == "pong":
353
+ continue
354
+ if msg_type in {"end_of_utterance", "cancellation"}:
355
+ result = speech_result_from_terminal(data)
356
+ self.last_result = result
357
+ return result
358
+ raise APIConnectionError(
359
+ f"Unexpected WebSocket message type while draining: {msg_type}"
360
+ )
361
+
362
+
363
+ def _websocket_connection_error(
364
+ exc: Exception, *, message: str
365
+ ) -> APIConnectionError | APIStatusError:
366
+ """Translate WebSocket close codes without exposing request headers."""
367
+ received = getattr(exc, "rcvd", None)
368
+ code = getattr(received, "code", None)
369
+ if code is None:
370
+ code = getattr(exc, "code", None)
371
+
372
+ if code == 4001:
373
+ return APIStatusError(
374
+ "Vakyam TTS authentication failed",
375
+ status_code=401,
376
+ retryable=False,
377
+ )
378
+ if isinstance(code, int):
379
+ return APIStatusError(
380
+ f"{message} (WebSocket closed with code {code})",
381
+ status_code=code,
382
+ retryable=code in {1011, 4002},
383
+ )
384
+ return APIConnectionError(f"{message}: {type(exc).__name__}")
@@ -0,0 +1,3 @@
1
+ import logging
2
+
3
+ logger = logging.getLogger("livekit.plugins.vakyam")
@@ -0,0 +1,42 @@
1
+ # Copyright 2025 LiveKit, Inc.
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ from typing import Literal
16
+
17
+ TTSModels = Literal["raaga-v1"]
18
+ TTSLanguages = Literal["bn-IN", "en-IN", "gu-IN", "hi-IN", "kn-IN", "mr-IN", "ta-IN", "te-IN"]
19
+ TTSSampleRates = Literal[8000, 16000, 24000, 48000]
20
+
21
+ DEFAULT_MODEL: TTSModels = "raaga-v1"
22
+ DEFAULT_VOICE = "Archana"
23
+ DEFAULT_LANGUAGE: TTSLanguages = "ta-IN"
24
+ DEFAULT_SAMPLE_RATE: TTSSampleRates = 24000
25
+ DEFAULT_SPEED = 1.0
26
+ DEFAULT_BASE_URL = "https://api.vakyam.ai"
27
+
28
+ TTS_STREAM_PATH = "/v1/tts/stream"
29
+ TTS_WEBSOCKET_PATH = "/v1/tts/websocket"
30
+ MAX_TEXT_CHARACTERS = 3000
31
+ CUSTOM_VOICE_PREFIX = "vc_"
32
+ MIN_SPEED = 0.5
33
+ MAX_SPEED = 2.0
34
+
35
+ # Application-level keepalive. The server closes idle sockets after ~60s.
36
+ KEEPALIVE_INTERVAL_SECONDS = 20.0
37
+
38
+ SUPPORTED_MODELS: frozenset[str] = frozenset({"raaga-v1"})
39
+ SUPPORTED_LANGUAGES: frozenset[str] = frozenset(
40
+ {"bn-IN", "en-IN", "gu-IN", "hi-IN", "kn-IN", "mr-IN", "ta-IN", "te-IN"}
41
+ )
42
+ SUPPORTED_SAMPLE_RATES: frozenset[int] = frozenset({8000, 16000, 24000, 48000})
File without changes
@@ -0,0 +1,512 @@
1
+ # Copyright 2025 LiveKit, Inc.
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """Text-to-Speech implementation for Vakyam AI (Raaga 1).
16
+
17
+ Streaming uses the realtime WebSocket API with sentence tokenization because
18
+ Vakyam expects one complete utterance per ``text`` message. One-shot
19
+ ``synthesize()`` uses HTTP ``POST /v1/tts/stream`` for PCM bytes.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import asyncio
25
+ import os
26
+ import weakref
27
+ from contextlib import suppress
28
+ from dataclasses import dataclass, replace
29
+
30
+ import aiohttp
31
+
32
+ from livekit.agents import (
33
+ APIConnectionError,
34
+ APIConnectOptions,
35
+ APIStatusError,
36
+ APITimeoutError,
37
+ tokenize,
38
+ tts,
39
+ utils,
40
+ )
41
+ from livekit.agents.types import DEFAULT_API_CONNECT_OPTIONS, NOT_GIVEN, NotGivenOr
42
+ from livekit.agents.utils import is_given
43
+
44
+ from ._utils import (
45
+ http_stream_url,
46
+ normalize_base_url,
47
+ raise_http_error,
48
+ speech_payload,
49
+ split_text,
50
+ validate_tts_options,
51
+ )
52
+ from ._websocket import AsyncStreamingTTSSession, TTSSessionConfig
53
+ from .log import logger
54
+ from .models import (
55
+ DEFAULT_BASE_URL,
56
+ DEFAULT_LANGUAGE,
57
+ DEFAULT_MODEL,
58
+ DEFAULT_SAMPLE_RATE,
59
+ DEFAULT_SPEED,
60
+ DEFAULT_VOICE,
61
+ KEEPALIVE_INTERVAL_SECONDS,
62
+ TTSLanguages,
63
+ TTSModels,
64
+ TTSSampleRates,
65
+ )
66
+ from .version import __version__
67
+
68
+ NUM_CHANNELS = 1
69
+
70
+
71
+ @dataclass
72
+ class _TTSOptions:
73
+ model: TTSModels | str
74
+ voice: str
75
+ language: TTSLanguages | str
76
+ sample_rate: TTSSampleRates | int
77
+ speed: float
78
+ api_key: str
79
+ base_url: str
80
+ allow_insecure_base_url: bool
81
+
82
+
83
+ class TTS(tts.TTS):
84
+ """Vakyam AI text-to-speech for LiveKit Agents.
85
+
86
+ Uses the realtime WebSocket API (``WS /v1/tts/websocket``) for low-latency
87
+ streaming. Text is sentence-tokenized because Vakyam expects one complete
88
+ utterance per synthesis request.
89
+ """
90
+
91
+ def __init__(
92
+ self,
93
+ *,
94
+ api_key: str | None = None,
95
+ model: TTSModels | str = DEFAULT_MODEL,
96
+ voice: str = DEFAULT_VOICE,
97
+ language: TTSLanguages | str = DEFAULT_LANGUAGE,
98
+ sample_rate: TTSSampleRates | int = DEFAULT_SAMPLE_RATE,
99
+ speed: float = DEFAULT_SPEED,
100
+ base_url: str | None = None,
101
+ allow_insecure_base_url: bool = False,
102
+ tokenizer: NotGivenOr[tokenize.SentenceTokenizer] = NOT_GIVEN,
103
+ http_session: aiohttp.ClientSession | None = None,
104
+ ) -> None:
105
+ """Create a Vakyam TTS instance.
106
+
107
+ Args:
108
+ api_key: Vakyam API key. Defaults to ``VAKYAM_API_KEY``.
109
+ model: TTS model id. Currently ``raaga-v1``.
110
+ voice: Voice selector — a preset name from ``GET /v1/voices``, or a
111
+ custom voice ID beginning with ``vc_``.
112
+ language: BCP-47 language code (for example ``ta-IN``).
113
+ sample_rate: Output PCM sample rate in Hz (8000, 16000, 24000, 48000).
114
+ speed: Speech rate multiplier (``0.5``–``2.0``).
115
+ base_url: API base URL. Defaults to ``https://api.vakyam.ai``.
116
+ allow_insecure_base_url: Allow non-localhost ``http://`` base URLs.
117
+ tokenizer: Sentence tokenizer used for streaming synthesis.
118
+ http_session: Optional aiohttp session for HTTP ``synthesize()``.
119
+ """
120
+ resolved_key = api_key or os.environ.get("VAKYAM_API_KEY")
121
+ if not resolved_key:
122
+ raise ValueError("Vakyam API key is required, either as api_key= or VAKYAM_API_KEY")
123
+
124
+ validate_tts_options(
125
+ model=str(model),
126
+ language=str(language),
127
+ sample_rate=int(sample_rate),
128
+ speed=speed,
129
+ voice=voice,
130
+ )
131
+
132
+ super().__init__(
133
+ capabilities=tts.TTSCapabilities(streaming=True, aligned_transcript=False),
134
+ sample_rate=int(sample_rate),
135
+ num_channels=NUM_CHANNELS,
136
+ )
137
+
138
+ self._opts = _TTSOptions(
139
+ model=model,
140
+ voice=voice,
141
+ language=language,
142
+ sample_rate=sample_rate,
143
+ speed=speed,
144
+ api_key=resolved_key,
145
+ base_url=normalize_base_url(
146
+ base_url or DEFAULT_BASE_URL, allow_insecure_base_url=allow_insecure_base_url
147
+ ),
148
+ allow_insecure_base_url=allow_insecure_base_url,
149
+ )
150
+ if is_given(tokenizer):
151
+ self._sentence_tokenizer = tokenizer
152
+ else:
153
+ try:
154
+ self._sentence_tokenizer = tokenize.blingfire.SentenceTokenizer()
155
+ except Exception:
156
+ self._sentence_tokenizer = tokenize.basic.SentenceTokenizer()
157
+ self._session = http_session
158
+ self._streams = weakref.WeakSet[SynthesizeStream]()
159
+ self._pools: dict[TTSSessionConfig, utils.ConnectionPool[AsyncStreamingTTSSession]] = {}
160
+ self._ws_keepalive_tasks: dict[AsyncStreamingTTSSession, asyncio.Task[None]] = {}
161
+
162
+ @property
163
+ def model(self) -> str:
164
+ return str(self._opts.model)
165
+
166
+ @property
167
+ def provider(self) -> str:
168
+ return "Vakyam"
169
+
170
+ def _ensure_session(self) -> aiohttp.ClientSession:
171
+ if not self._session:
172
+ self._session = utils.http_context.http_session()
173
+ return self._session
174
+
175
+ def update_options(
176
+ self,
177
+ *,
178
+ model: TTSModels | str | None = None,
179
+ voice: str | None = None,
180
+ language: TTSLanguages | str | None = None,
181
+ sample_rate: TTSSampleRates | int | None = None,
182
+ speed: float | None = None,
183
+ ) -> None:
184
+ """Update synthesis options for streams created after this call."""
185
+ next_opts = replace(self._opts)
186
+ next_sample_rate = self._sample_rate
187
+ if model is not None:
188
+ next_opts.model = model
189
+ if voice is not None:
190
+ next_opts.voice = voice
191
+ if language is not None:
192
+ next_opts.language = language
193
+ if sample_rate is not None:
194
+ next_opts.sample_rate = sample_rate
195
+ next_sample_rate = int(sample_rate)
196
+ if speed is not None:
197
+ next_opts.speed = speed
198
+
199
+ validate_tts_options(
200
+ model=str(next_opts.model),
201
+ language=str(next_opts.language),
202
+ sample_rate=int(next_opts.sample_rate),
203
+ speed=next_opts.speed,
204
+ voice=next_opts.voice,
205
+ )
206
+ self._opts = next_opts
207
+ self._sample_rate = next_sample_rate
208
+
209
+ def synthesize(
210
+ self, text: str, *, conn_options: APIConnectOptions = DEFAULT_API_CONNECT_OPTIONS
211
+ ) -> ChunkedStream:
212
+ return ChunkedStream(tts=self, input_text=text, conn_options=conn_options)
213
+
214
+ def stream(
215
+ self, *, conn_options: APIConnectOptions = DEFAULT_API_CONNECT_OPTIONS
216
+ ) -> SynthesizeStream:
217
+ stream = SynthesizeStream(tts=self, conn_options=conn_options)
218
+ self._streams.add(stream)
219
+ return stream
220
+
221
+ def prewarm(self) -> None:
222
+ self._pool_for(self._opts).prewarm()
223
+
224
+ async def aclose(self) -> None:
225
+ for stream in list(self._streams):
226
+ await stream.aclose()
227
+ self._streams.clear()
228
+ for pool in self._pools.values():
229
+ await pool.aclose()
230
+ self._pools.clear()
231
+
232
+ def _session_config(self, opts: _TTSOptions) -> TTSSessionConfig:
233
+ return TTSSessionConfig(
234
+ model=str(opts.model),
235
+ voice=opts.voice,
236
+ language=str(opts.language),
237
+ sample_rate=int(opts.sample_rate),
238
+ speed=opts.speed,
239
+ output_format="pcm",
240
+ )
241
+
242
+ def _pool_for(self, opts: _TTSOptions) -> utils.ConnectionPool[AsyncStreamingTTSSession]:
243
+ config = self._session_config(opts)
244
+ existing_pool = self._pools.get(config)
245
+ if existing_pool is not None:
246
+ return existing_pool
247
+
248
+ pool_ref: utils.ConnectionPool[AsyncStreamingTTSSession] | None = None
249
+
250
+ async def _connect(timeout: float) -> AsyncStreamingTTSSession:
251
+ session = AsyncStreamingTTSSession(
252
+ api_key=opts.api_key,
253
+ base_url=opts.base_url,
254
+ allow_insecure_base_url=opts.allow_insecure_base_url,
255
+ config=config,
256
+ )
257
+ await session.connect(timeout=timeout)
258
+ assert pool_ref is not None
259
+ self._start_keepalive(session, pool_ref)
260
+ return session
261
+
262
+ pool = utils.ConnectionPool(
263
+ connect_cb=_connect,
264
+ close_cb=self._close_ws,
265
+ max_session_duration=3600,
266
+ mark_refreshed_on_get=False,
267
+ )
268
+ pool_ref = pool
269
+ self._pools[config] = pool
270
+ return pool
271
+
272
+ async def _close_ws(self, session: AsyncStreamingTTSSession) -> None:
273
+ await self._stop_keepalive(session)
274
+ await session.close()
275
+
276
+ def _start_keepalive(
277
+ self,
278
+ session: AsyncStreamingTTSSession,
279
+ pool: utils.ConnectionPool[AsyncStreamingTTSSession],
280
+ ) -> None:
281
+ task = self._ws_keepalive_tasks.get(session)
282
+ if task is not None and not task.done():
283
+ return
284
+ self._ws_keepalive_tasks[session] = asyncio.create_task(
285
+ self._keepalive_loop(session, pool), name="vakyam-tts-ws-keepalive"
286
+ )
287
+
288
+ async def _stop_keepalive(self, session: AsyncStreamingTTSSession) -> None:
289
+ task = self._ws_keepalive_tasks.pop(session, None)
290
+ if task is None:
291
+ return
292
+ task.cancel()
293
+ with suppress(asyncio.CancelledError):
294
+ await task
295
+
296
+ async def _keepalive_loop(
297
+ self,
298
+ session: AsyncStreamingTTSSession,
299
+ pool: utils.ConnectionPool[AsyncStreamingTTSSession],
300
+ ) -> None:
301
+ try:
302
+ while True:
303
+ await asyncio.sleep(KEEPALIVE_INTERVAL_SECONDS)
304
+ if not session.connected:
305
+ return
306
+ try:
307
+ await session.ping()
308
+ except Exception as exc:
309
+ logger.debug(
310
+ "Vakyam TTS keepalive failed (%s); evicting session",
311
+ type(exc).__name__,
312
+ )
313
+ pool.remove(session)
314
+ return
315
+ except asyncio.CancelledError:
316
+ return
317
+ finally:
318
+ current = asyncio.current_task()
319
+ if self._ws_keepalive_tasks.get(session) is current:
320
+ self._ws_keepalive_tasks.pop(session, None)
321
+
322
+
323
+ class ChunkedStream(tts.ChunkedStream):
324
+ """One-shot synthesis over HTTP ``POST /v1/tts/stream``."""
325
+
326
+ def __init__(self, *, tts: TTS, input_text: str, conn_options: APIConnectOptions) -> None:
327
+ super().__init__(tts=tts, input_text=input_text, conn_options=conn_options)
328
+ self._tts: TTS = tts
329
+ self._opts = replace(tts._opts)
330
+
331
+ async def _run(self, output_emitter: tts.AudioEmitter) -> None:
332
+ request_id = utils.shortuuid()
333
+ payload = speech_payload(
334
+ text=self._input_text,
335
+ model=str(self._opts.model),
336
+ voice=self._opts.voice,
337
+ language=str(self._opts.language),
338
+ sample_rate=int(self._opts.sample_rate),
339
+ speed=self._opts.speed,
340
+ output_format="pcm",
341
+ )
342
+ headers = {
343
+ "Authorization": f"Bearer {self._opts.api_key}",
344
+ "Content-Type": "application/json",
345
+ "User-Agent": f"LiveKit-Agents-Vakyam/{__version__}",
346
+ }
347
+ try:
348
+ async with self._tts._ensure_session().post(
349
+ http_stream_url(self._opts.base_url),
350
+ json=payload,
351
+ headers=headers,
352
+ timeout=aiohttp.ClientTimeout(
353
+ total=None,
354
+ sock_connect=self._conn_options.timeout,
355
+ sock_read=self._conn_options.timeout,
356
+ ),
357
+ ) as resp:
358
+ if resp.status >= 400:
359
+ body = await resp.text()
360
+ raise_http_error(resp.status, body)
361
+
362
+ content_type = resp.headers.get("Content-Type", "").lower()
363
+ if content_type and not (
364
+ content_type.startswith("audio/")
365
+ or content_type.startswith("application/octet-stream")
366
+ ):
367
+ body = await resp.text()
368
+ raise APIStatusError(
369
+ "Vakyam TTS returned a non-audio response",
370
+ status_code=502,
371
+ body=body,
372
+ )
373
+
374
+ output_emitter.initialize(
375
+ request_id=request_id,
376
+ sample_rate=int(self._opts.sample_rate),
377
+ num_channels=NUM_CHANNELS,
378
+ mime_type="audio/pcm",
379
+ )
380
+ async for chunk, _ in resp.content.iter_chunks():
381
+ if chunk:
382
+ output_emitter.push(chunk)
383
+ output_emitter.flush()
384
+ except (APIStatusError, APIConnectionError, APITimeoutError):
385
+ raise
386
+ except asyncio.TimeoutError as exc:
387
+ raise APITimeoutError("Vakyam TTS HTTP stream timed out") from exc
388
+ except aiohttp.ClientError as exc:
389
+ raise APIConnectionError(f"Vakyam TTS HTTP stream connection error: {exc}") from exc
390
+
391
+
392
+ class SynthesizeStream(tts.SynthesizeStream):
393
+ """Streaming synthesis with sentence tokenization over a persistent WebSocket."""
394
+
395
+ def __init__(self, *, tts: TTS, conn_options: APIConnectOptions) -> None:
396
+ super().__init__(tts=tts, conn_options=conn_options)
397
+ self._tts: TTS = tts
398
+ self._opts = replace(tts._opts)
399
+
400
+ async def _run(self, output_emitter: tts.AudioEmitter) -> None:
401
+ segments_ch = utils.aio.Chan[tokenize.SentenceStream]()
402
+ request_id = utils.shortuuid()
403
+ output_emitter.initialize(
404
+ request_id=request_id,
405
+ sample_rate=int(self._opts.sample_rate),
406
+ num_channels=NUM_CHANNELS,
407
+ mime_type="audio/pcm",
408
+ stream=True,
409
+ frame_size_ms=50,
410
+ )
411
+
412
+ async def _tokenize_input() -> None:
413
+ sentence_stream: tokenize.SentenceStream | None = None
414
+ async for data in self._input_ch:
415
+ if isinstance(data, str):
416
+ if sentence_stream is None:
417
+ sentence_stream = self._tts._sentence_tokenizer.stream()
418
+ segments_ch.send_nowait(sentence_stream)
419
+ sentence_stream.push_text(data)
420
+ elif isinstance(data, self._FlushSentinel):
421
+ if sentence_stream is not None:
422
+ sentence_stream.end_input()
423
+ sentence_stream = None
424
+ if sentence_stream is not None:
425
+ sentence_stream.end_input()
426
+ segments_ch.close()
427
+
428
+ async def _process_segments() -> None:
429
+ async for sentence_stream in segments_ch:
430
+ await self._run_segment(sentence_stream, output_emitter)
431
+
432
+ tasks = [
433
+ asyncio.create_task(_tokenize_input()),
434
+ asyncio.create_task(_process_segments()),
435
+ ]
436
+ try:
437
+ await asyncio.gather(*tasks)
438
+ except (APIStatusError, APIConnectionError, APITimeoutError):
439
+ raise
440
+ except asyncio.CancelledError:
441
+ raise
442
+ except Exception as exc:
443
+ raise APIConnectionError(f"Vakyam TTS stream failed: {exc}") from exc
444
+ finally:
445
+ await utils.aio.gracefully_cancel(*tasks)
446
+ output_emitter.end_input()
447
+
448
+ async def _run_segment(
449
+ self, sentence_stream: tokenize.SentenceStream, output_emitter: tts.AudioEmitter
450
+ ) -> None:
451
+ segment_id = utils.shortuuid()
452
+ output_emitter.start_segment(segment_id=segment_id)
453
+ pool = self._tts._pool_for(self._opts)
454
+ deferred_error: APIStatusError | None = None
455
+ cancelled = False
456
+
457
+ async with pool.connection(timeout=self._conn_options.timeout) as session:
458
+ self._acquire_time = pool.last_acquire_time
459
+ self._connection_reused = pool.last_connection_reused
460
+ await self._tts._stop_keepalive(session)
461
+ reusable = False
462
+ try:
463
+ started = False
464
+ async for sentence in sentence_stream:
465
+ for text in split_text(sentence.token):
466
+ if not started:
467
+ self._mark_started()
468
+ started = True
469
+
470
+ async for chunk in session.synthesize_stream(
471
+ text, timeout=self._conn_options.timeout
472
+ ):
473
+ output_emitter.push(chunk)
474
+
475
+ if session.last_result and session.last_result.cancelled:
476
+ logger.debug("Vakyam TTS utterance cancelled")
477
+ break
478
+ if session.last_result and session.last_result.truncated:
479
+ deferred_error = APIStatusError(
480
+ "Vakyam TTS utterance was truncated",
481
+ status_code=500,
482
+ retryable=False,
483
+ )
484
+ break
485
+ if deferred_error is not None or (
486
+ session.last_result and session.last_result.cancelled
487
+ ):
488
+ break
489
+ reusable = True
490
+ except asyncio.CancelledError:
491
+ # Reuse only after Vakyam acknowledges cancel and trailing audio is drained.
492
+ reusable = bool(session.last_result and session.last_result.cancelled)
493
+ if not reusable:
494
+ raise
495
+ cancelled = True
496
+ except APIStatusError as exc:
497
+ # Only JSON per-message errors leave the WebSocket open. Close-code
498
+ # status errors represent a dead transport and must be evicted.
499
+ if isinstance(exc.body, dict) and exc.body.get("type") == "error":
500
+ reusable = True
501
+ deferred_error = exc
502
+ else:
503
+ raise
504
+ finally:
505
+ if reusable:
506
+ self._tts._start_keepalive(session, pool)
507
+
508
+ if cancelled:
509
+ raise asyncio.CancelledError
510
+ if deferred_error is not None:
511
+ raise deferred_error
512
+ output_emitter.end_segment()
@@ -0,0 +1,15 @@
1
+ # Copyright 2025 LiveKit, Inc.
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ __version__ = "1.8.0"
@@ -0,0 +1,80 @@
1
+ Metadata-Version: 2.5
2
+ Name: livekit-plugins-vakyam
3
+ Version: 1.8.0
4
+ Summary: LiveKit Agents plugin for Vakyam AI TTS (Raaga 1) — Indian-language text-to-speech
5
+ Project-URL: Documentation, https://docs.livekit.io
6
+ Project-URL: Website, https://livekit.io/
7
+ Project-URL: Source, https://github.com/livekit/agents
8
+ Author-email: LiveKit <hello@livekit.io>
9
+ License-Expression: Apache-2.0
10
+ Keywords: audio,indian-languages,livekit,raaga,realtime,text-to-speech,tts,vakyam,webrtc
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: License :: OSI Approved :: Apache Software License
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3 :: Only
15
+ Classifier: Programming Language :: Python :: 3.10
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Topic :: Multimedia :: Sound/Audio
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Requires-Python: >=3.10.0
21
+ Requires-Dist: livekit-agents[codecs]>=1.8.0
22
+ Requires-Dist: websockets<16.0,>=14.0
23
+ Description-Content-Type: text/markdown
24
+
25
+ # Vakyam AI plugin for LiveKit Agents
26
+
27
+ Support for voice synthesis with [Vakyam AI](https://vakyam.ai/) Raaga 1 —
28
+ text-to-speech for Indian languages.
29
+
30
+ See [https://docs.vakyam.ai/integrations/livekit](https://docs.vakyam.ai/integrations/livekit)
31
+ for provider docs.
32
+
33
+ ## Installation
34
+
35
+ ```bash
36
+ pip install livekit-plugins-vakyam
37
+ ```
38
+
39
+ Or with the LiveKit Agents extra:
40
+
41
+ ```bash
42
+ uv add "livekit-agents[vakyam]"
43
+ ```
44
+
45
+ ## Pre-requisites
46
+
47
+ You'll need an API key from [Vakyam](https://dashboard.vakyam.ai/api-keys).
48
+ Set it as an environment variable:
49
+
50
+ ```bash
51
+ export VAKYAM_API_KEY="vak_live_..."
52
+ ```
53
+
54
+ ## Usage
55
+
56
+ ```python
57
+ from livekit.agents import AgentSession
58
+ from livekit.plugins import vakyam
59
+
60
+ session = AgentSession(
61
+ tts=vakyam.TTS(
62
+ model="raaga-v1",
63
+ voice="Archana",
64
+ language="ta-IN",
65
+ sample_rate=24000,
66
+ ),
67
+ # ... stt, llm, vad
68
+ )
69
+ ```
70
+
71
+ `stream()` uses the realtime WebSocket API and sentence-tokenizes LLM text so
72
+ each utterance is one complete sentence (Vakyam does not accept partial
73
+ tokens). `synthesize()` uses HTTP streaming (`POST /v1/tts/stream`) and
74
+ returns PCM audio.
75
+
76
+ WebSocket connections are pooled and reused between sequential agent turns.
77
+ Each active synthesis stream has exclusive ownership of its connection, so an
78
+ overlapping stream uses a separate connection. On interruption, the plugin
79
+ sends `cancel`, drains through Vakyam's cancellation acknowledgement, and
80
+ returns the healthy connection to the pool.
@@ -0,0 +1,11 @@
1
+ livekit/plugins/vakyam/__init__.py,sha256=veB5JV1El0O2H1LRXiG9ajxFWWLkijuxHsJHT068oZQ,1294
2
+ livekit/plugins/vakyam/_utils.py,sha256=Tmo3BLnMwx6deAOq8ijjl9yHgbpDtL4d-4XI8_IaW0Q,7132
3
+ livekit/plugins/vakyam/_websocket.py,sha256=5z59N82zEoR7sQjpLTLFcH2-rByYcZliYHvr9-P_7-E,14093
4
+ livekit/plugins/vakyam/log.py,sha256=rSyzLdpYK33mxW_1ZuxZ6oCLgJs0UWQpkc12mpjEQqA,69
5
+ livekit/plugins/vakyam/models.py,sha256=Njf3AwJ1iR_PeGXE0Zk5VrBvd2C3ZRPqDfdfhmFmcCo,1539
6
+ livekit/plugins/vakyam/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
7
+ livekit/plugins/vakyam/tts.py,sha256=hdZG-9wtKuqJo3Eua8A4tbLT0N0f7YXTUe5Kj8OZGXk,19008
8
+ livekit/plugins/vakyam/version.py,sha256=IeTDs-sIIzY-W7jelf8ozYz_UfMZ153wLnSwlRVky74,600
9
+ livekit_plugins_vakyam-1.8.0.dist-info/METADATA,sha256=YBJFSzQLRFskdPH9pFT5aq48NDhRLD9Q0yPogwJpMJM,2568
10
+ livekit_plugins_vakyam-1.8.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
11
+ livekit_plugins_vakyam-1.8.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any