livekit-plugins-volcengine 1.2.2__tar.gz → 1.2.3.post0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (15) hide show
  1. {livekit_plugins_volcengine-1.2.2 → livekit_plugins_volcengine-1.2.3.post0}/PKG-INFO +2 -2
  2. {livekit_plugins_volcengine-1.2.2 → livekit_plugins_volcengine-1.2.3.post0}/livekit/plugins/volcengine/realtime.py +18 -0
  3. livekit_plugins_volcengine-1.2.3.post0/livekit/plugins/volcengine/version.py +1 -0
  4. {livekit_plugins_volcengine-1.2.2 → livekit_plugins_volcengine-1.2.3.post0}/pyproject.toml +1 -1
  5. livekit_plugins_volcengine-1.2.2/livekit/plugins/volcengine/version.py +0 -1
  6. {livekit_plugins_volcengine-1.2.2 → livekit_plugins_volcengine-1.2.3.post0}/.gitignore +0 -0
  7. {livekit_plugins_volcengine-1.2.2 → livekit_plugins_volcengine-1.2.3.post0}/README.md +0 -0
  8. {livekit_plugins_volcengine-1.2.2 → livekit_plugins_volcengine-1.2.3.post0}/livekit/plugins/volcengine/__init__.py +0 -0
  9. {livekit_plugins_volcengine-1.2.2 → livekit_plugins_volcengine-1.2.3.post0}/livekit/plugins/volcengine/bigmodel_stt.py +0 -0
  10. {livekit_plugins_volcengine-1.2.2 → livekit_plugins_volcengine-1.2.3.post0}/livekit/plugins/volcengine/llm.py +0 -0
  11. {livekit_plugins_volcengine-1.2.2 → livekit_plugins_volcengine-1.2.3.post0}/livekit/plugins/volcengine/log.py +0 -0
  12. {livekit_plugins_volcengine-1.2.2 → livekit_plugins_volcengine-1.2.3.post0}/livekit/plugins/volcengine/py.typed +0 -0
  13. {livekit_plugins_volcengine-1.2.2 → livekit_plugins_volcengine-1.2.3.post0}/livekit/plugins/volcengine/stt.py +0 -0
  14. {livekit_plugins_volcengine-1.2.2 → livekit_plugins_volcengine-1.2.3.post0}/livekit/plugins/volcengine/tts.py +0 -0
  15. {livekit_plugins_volcengine-1.2.2 → livekit_plugins_volcengine-1.2.3.post0}/livekit/plugins/volcengine/utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: livekit-plugins-volcengine
3
- Version: 1.2.2
3
+ Version: 1.2.3.post0
4
4
  Summary: LiveKit Agent Plugins for Volcengine
5
5
  Author-email: wangmengdi <790990241@qq.com>
6
6
  Keywords: audio,livekit,realtime,video,webrtc
@@ -14,7 +14,7 @@ Classifier: Topic :: Multimedia :: Sound/Audio
14
14
  Classifier: Topic :: Multimedia :: Video
15
15
  Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
16
16
  Requires-Python: >=3.9
17
- Requires-Dist: livekit-agents>=1.2.2
17
+ Requires-Dist: livekit-agents==1.2.2
18
18
  Requires-Dist: numpy
19
19
  Requires-Dist: openai>=1.75.0
20
20
  Requires-Dist: osc-data>=0.2.2
@@ -171,6 +171,7 @@ class _RealtimeOptions:
171
171
  conn_options: APIConnectOptions
172
172
  opening: str = "你好啊,今天过得怎么样?"
173
173
  speaking_style: str = "你的说话风格简洁明了,语速适中,语调自然。"
174
+ speaker: str = "zh_female_vv_jupiter_bigtts"
174
175
  sample_rate: int = 24000
175
176
  num_channels: int = 1
176
177
  format: str = "pcm"
@@ -197,6 +198,7 @@ class _RealtimeOptions:
197
198
  "format": self.format,
198
199
  "sample_rate": self.sample_rate,
199
200
  },
201
+ "speaker": self.speaker,
200
202
  },
201
203
  "dialog": {
202
204
  "bot_name": self.bot_name,
@@ -214,6 +216,7 @@ class _MessageGeneration:
214
216
  message_id: str
215
217
  text_ch: utils.aio.Chan[str]
216
218
  audio_ch: utils.aio.Chan[rtc.AudioFrame]
219
+ modalities: asyncio.Future[list[Literal["text", "audio"]]]
217
220
  audio_transcript: str = ""
218
221
 
219
222
 
@@ -236,6 +239,7 @@ class RealtimeModel(llm.RealtimeModel):
236
239
  self,
237
240
  bot_name: str = "豆包",
238
241
  speaking_style: str = "你的说话风格简洁明了,语速适中,语调自然。",
242
+ speaker: str = "zh_female_vv_jupiter_bigtts",
239
243
  opening: str | None = None,
240
244
  app_id: str | None = None,
241
245
  access_token: str | None = None,
@@ -270,11 +274,13 @@ class RealtimeModel(llm.RealtimeModel):
270
274
  conn_options=conn_options,
271
275
  bot_name=bot_name,
272
276
  system_role=system_role,
277
+ speaker=speaker,
273
278
  opening=opening,
274
279
  speaking_style=speaking_style,
275
280
  )
276
281
  self._http_session = http_session
277
282
  self._sessions = weakref.WeakSet[RealtimeSession]()
283
+ print(self.capabilities.audio_output)
278
284
 
279
285
  def update_options(
280
286
  self,
@@ -436,13 +442,18 @@ class RealtimeSession(
436
442
  message_id=item_id,
437
443
  text_ch=utils.aio.Chan(),
438
444
  audio_ch=utils.aio.Chan(),
445
+ modalities=asyncio.Future(),
439
446
  )
447
+ if not self._realtime_model.capabilities.audio_output:
448
+ self._current_item.audio_ch.close()
449
+ self._current_item.modalities.set_result(["text"])
440
450
 
441
451
  self._current_generation.message_ch.send_nowait(
442
452
  llm.MessageGeneration(
443
453
  message_id=item_id,
444
454
  text_stream=self._current_item.text_ch,
445
455
  audio_stream=self._current_item.audio_ch,
456
+ modalities=self._current_item.modalities,
446
457
  )
447
458
  )
448
459
 
@@ -494,6 +505,7 @@ class RealtimeSession(
494
505
  is_final = not response["results"][0]["is_interim"]
495
506
  if is_final:
496
507
  item_id = utils.shortuuid()
508
+ logger.info("transcription completed")
497
509
  self.emit(
498
510
  "input_audio_transcription_completed",
499
511
  llm.InputTranscriptionCompleted(
@@ -523,12 +535,17 @@ class RealtimeSession(
523
535
  message_id=item_id,
524
536
  text_ch=utils.aio.Chan(),
525
537
  audio_ch=utils.aio.Chan(),
538
+ modalities=asyncio.Future(),
526
539
  )
540
+ if not self._realtime_model.capabilities.audio_output:
541
+ self._current_item.audio_ch.close()
542
+ self._current_item.modalities.set_result(["text"])
527
543
  self._current_generation.message_ch.send_nowait(
528
544
  llm.MessageGeneration(
529
545
  message_id=item_id,
530
546
  text_stream=self._current_item.text_ch,
531
547
  audio_stream=self._current_item.audio_ch,
548
+ modalities=self._current_item.modalities,
532
549
  )
533
550
  )
534
551
 
@@ -544,6 +561,7 @@ class RealtimeSession(
544
561
  logger.info("tts start")
545
562
 
546
563
  elif event == 352: # TTSResponse
564
+ logger.info("tts response")
547
565
  if self._first_tts_response:
548
566
  logger.info("llm first sentence")
549
567
  logger.info("tts first response")
@@ -0,0 +1 @@
1
+ __version__ = "1.2.3.post0"
@@ -9,11 +9,11 @@ authors = [
9
9
  keywords = ["webrtc", "realtime", "audio", "video", "livekit"]
10
10
  requires-python = ">=3.9"
11
11
  dependencies = [
12
- "livekit-agents>=1.2.2",
13
12
  "openai>=1.75.0",
14
13
  "osc-data>=0.2.2",
15
14
  "pydantic",
16
15
  "numpy",
16
+ "livekit-agents==1.2.2",
17
17
  ]
18
18
 
19
19
  classifiers = [
@@ -1 +0,0 @@
1
- __version__ = "1.2.2"