@livekit/agents 0.7.9 → 1.0.0-next.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (627) hide show
  1. package/dist/_exceptions.cjs +109 -0
  2. package/dist/_exceptions.cjs.map +1 -0
  3. package/dist/_exceptions.d.cts +64 -0
  4. package/dist/_exceptions.d.ts +64 -0
  5. package/dist/_exceptions.d.ts.map +1 -0
  6. package/dist/_exceptions.js +80 -0
  7. package/dist/_exceptions.js.map +1 -0
  8. package/dist/audio.cjs +10 -3
  9. package/dist/audio.cjs.map +1 -1
  10. package/dist/audio.d.cts +2 -0
  11. package/dist/audio.d.ts +2 -0
  12. package/dist/audio.d.ts.map +1 -1
  13. package/dist/audio.js +8 -2
  14. package/dist/audio.js.map +1 -1
  15. package/dist/cli.cjs +25 -0
  16. package/dist/cli.cjs.map +1 -1
  17. package/dist/cli.d.ts.map +1 -1
  18. package/dist/cli.js +25 -0
  19. package/dist/cli.js.map +1 -1
  20. package/dist/constants.cjs +6 -3
  21. package/dist/constants.cjs.map +1 -1
  22. package/dist/constants.d.cts +2 -1
  23. package/dist/constants.d.ts +2 -1
  24. package/dist/constants.d.ts.map +1 -1
  25. package/dist/constants.js +4 -2
  26. package/dist/constants.js.map +1 -1
  27. package/dist/http_server.cjs.map +1 -1
  28. package/dist/http_server.d.cts +1 -0
  29. package/dist/http_server.d.ts +1 -0
  30. package/dist/http_server.d.ts.map +1 -1
  31. package/dist/http_server.js.map +1 -1
  32. package/dist/index.cjs +27 -20
  33. package/dist/index.cjs.map +1 -1
  34. package/dist/index.d.cts +13 -10
  35. package/dist/index.d.ts +13 -10
  36. package/dist/index.d.ts.map +1 -1
  37. package/dist/index.js +15 -11
  38. package/dist/index.js.map +1 -1
  39. package/dist/inference_runner.cjs +0 -1
  40. package/dist/inference_runner.cjs.map +1 -1
  41. package/dist/inference_runner.d.cts +2 -3
  42. package/dist/inference_runner.d.ts +2 -3
  43. package/dist/inference_runner.d.ts.map +1 -1
  44. package/dist/inference_runner.js +0 -1
  45. package/dist/inference_runner.js.map +1 -1
  46. package/dist/ipc/inference_proc_executor.cjs +2 -2
  47. package/dist/ipc/inference_proc_executor.cjs.map +1 -1
  48. package/dist/ipc/inference_proc_executor.js +2 -2
  49. package/dist/ipc/inference_proc_executor.js.map +1 -1
  50. package/dist/ipc/job_executor.cjs.map +1 -1
  51. package/dist/ipc/job_executor.js.map +1 -1
  52. package/dist/ipc/job_proc_executor.cjs +1 -0
  53. package/dist/ipc/job_proc_executor.cjs.map +1 -1
  54. package/dist/ipc/job_proc_executor.js +1 -0
  55. package/dist/ipc/job_proc_executor.js.map +1 -1
  56. package/dist/ipc/job_proc_lazy_main.cjs +1 -1
  57. package/dist/ipc/job_proc_lazy_main.cjs.map +1 -1
  58. package/dist/ipc/job_proc_lazy_main.js +1 -1
  59. package/dist/ipc/job_proc_lazy_main.js.map +1 -1
  60. package/dist/ipc/supervised_proc.d.cts +1 -1
  61. package/dist/ipc/supervised_proc.d.ts +1 -1
  62. package/dist/ipc/supervised_proc.d.ts.map +1 -1
  63. package/dist/job.cjs +14 -2
  64. package/dist/job.cjs.map +1 -1
  65. package/dist/job.d.cts +8 -0
  66. package/dist/job.d.ts +8 -0
  67. package/dist/job.d.ts.map +1 -1
  68. package/dist/job.js +12 -1
  69. package/dist/job.js.map +1 -1
  70. package/dist/llm/chat_context.cjs +332 -82
  71. package/dist/llm/chat_context.cjs.map +1 -1
  72. package/dist/llm/chat_context.d.cts +152 -48
  73. package/dist/llm/chat_context.d.ts +152 -48
  74. package/dist/llm/chat_context.d.ts.map +1 -1
  75. package/dist/llm/chat_context.js +327 -81
  76. package/dist/llm/chat_context.js.map +1 -1
  77. package/dist/llm/chat_context.test.cjs +380 -0
  78. package/dist/llm/chat_context.test.cjs.map +1 -0
  79. package/dist/llm/chat_context.test.js +385 -0
  80. package/dist/llm/chat_context.test.js.map +1 -0
  81. package/dist/llm/index.cjs +37 -8
  82. package/dist/llm/index.cjs.map +1 -1
  83. package/dist/llm/index.d.cts +7 -3
  84. package/dist/llm/index.d.ts +7 -3
  85. package/dist/llm/index.d.ts.map +1 -1
  86. package/dist/llm/index.js +39 -9
  87. package/dist/llm/index.js.map +1 -1
  88. package/dist/llm/llm.cjs +98 -33
  89. package/dist/llm/llm.cjs.map +1 -1
  90. package/dist/llm/llm.d.cts +50 -24
  91. package/dist/llm/llm.d.ts +50 -24
  92. package/dist/llm/llm.d.ts.map +1 -1
  93. package/dist/llm/llm.js +99 -33
  94. package/dist/llm/llm.js.map +1 -1
  95. package/dist/llm/provider_format/google.cjs +128 -0
  96. package/dist/llm/provider_format/google.cjs.map +1 -0
  97. package/dist/llm/provider_format/google.d.cts +6 -0
  98. package/dist/llm/provider_format/google.d.ts +6 -0
  99. package/dist/llm/provider_format/google.d.ts.map +1 -0
  100. package/dist/llm/provider_format/google.js +104 -0
  101. package/dist/llm/provider_format/google.js.map +1 -0
  102. package/dist/llm/provider_format/google.test.cjs +676 -0
  103. package/dist/llm/provider_format/google.test.cjs.map +1 -0
  104. package/dist/llm/provider_format/google.test.js +675 -0
  105. package/dist/llm/provider_format/google.test.js.map +1 -0
  106. package/dist/llm/provider_format/index.cjs +40 -0
  107. package/dist/llm/provider_format/index.cjs.map +1 -0
  108. package/dist/llm/provider_format/index.d.cts +4 -0
  109. package/dist/llm/provider_format/index.d.ts +4 -0
  110. package/dist/llm/provider_format/index.d.ts.map +1 -0
  111. package/dist/llm/provider_format/index.js +16 -0
  112. package/dist/llm/provider_format/index.js.map +1 -0
  113. package/dist/llm/provider_format/openai.cjs +116 -0
  114. package/dist/llm/provider_format/openai.cjs.map +1 -0
  115. package/dist/llm/provider_format/openai.d.cts +3 -0
  116. package/dist/llm/provider_format/openai.d.ts +3 -0
  117. package/dist/llm/provider_format/openai.d.ts.map +1 -0
  118. package/dist/llm/provider_format/openai.js +92 -0
  119. package/dist/llm/provider_format/openai.js.map +1 -0
  120. package/dist/llm/provider_format/openai.test.cjs +490 -0
  121. package/dist/llm/provider_format/openai.test.cjs.map +1 -0
  122. package/dist/llm/provider_format/openai.test.js +489 -0
  123. package/dist/llm/provider_format/openai.test.js.map +1 -0
  124. package/dist/llm/provider_format/utils.cjs +146 -0
  125. package/dist/llm/provider_format/utils.cjs.map +1 -0
  126. package/dist/llm/provider_format/utils.d.cts +38 -0
  127. package/dist/llm/provider_format/utils.d.ts +38 -0
  128. package/dist/llm/provider_format/utils.d.ts.map +1 -0
  129. package/dist/llm/provider_format/utils.js +122 -0
  130. package/dist/llm/provider_format/utils.js.map +1 -0
  131. package/dist/llm/realtime.cjs +77 -0
  132. package/dist/llm/realtime.cjs.map +1 -0
  133. package/dist/llm/realtime.d.cts +98 -0
  134. package/dist/llm/realtime.d.ts +98 -0
  135. package/dist/llm/realtime.d.ts.map +1 -0
  136. package/dist/llm/realtime.js +52 -0
  137. package/dist/llm/realtime.js.map +1 -0
  138. package/dist/llm/remote_chat_context.cjs +112 -0
  139. package/dist/llm/remote_chat_context.cjs.map +1 -0
  140. package/dist/llm/remote_chat_context.d.cts +23 -0
  141. package/dist/llm/remote_chat_context.d.ts +23 -0
  142. package/dist/llm/remote_chat_context.d.ts.map +1 -0
  143. package/dist/llm/remote_chat_context.js +88 -0
  144. package/dist/llm/remote_chat_context.js.map +1 -0
  145. package/dist/llm/remote_chat_context.test.cjs +225 -0
  146. package/dist/llm/remote_chat_context.test.cjs.map +1 -0
  147. package/dist/llm/remote_chat_context.test.js +224 -0
  148. package/dist/llm/remote_chat_context.test.js.map +1 -0
  149. package/dist/llm/tool_context.cjs +111 -0
  150. package/dist/llm/tool_context.cjs.map +1 -0
  151. package/dist/llm/tool_context.d.cts +125 -0
  152. package/dist/llm/tool_context.d.ts +125 -0
  153. package/dist/llm/tool_context.d.ts.map +1 -0
  154. package/dist/llm/tool_context.js +80 -0
  155. package/dist/llm/tool_context.js.map +1 -0
  156. package/dist/llm/tool_context.test.cjs +162 -0
  157. package/dist/llm/tool_context.test.cjs.map +1 -0
  158. package/dist/llm/tool_context.test.js +161 -0
  159. package/dist/llm/tool_context.test.js.map +1 -0
  160. package/dist/llm/tool_context.type.test.cjs +92 -0
  161. package/dist/llm/tool_context.type.test.cjs.map +1 -0
  162. package/dist/llm/tool_context.type.test.js +91 -0
  163. package/dist/llm/tool_context.type.test.js.map +1 -0
  164. package/dist/llm/utils.cjs +260 -0
  165. package/dist/llm/utils.cjs.map +1 -0
  166. package/dist/llm/utils.d.cts +42 -0
  167. package/dist/llm/utils.d.ts +42 -0
  168. package/dist/llm/utils.d.ts.map +1 -0
  169. package/dist/llm/utils.js +223 -0
  170. package/dist/llm/utils.js.map +1 -0
  171. package/dist/llm/utils.test.cjs +513 -0
  172. package/dist/llm/utils.test.cjs.map +1 -0
  173. package/dist/llm/utils.test.js +490 -0
  174. package/dist/llm/utils.test.js.map +1 -0
  175. package/dist/metrics/base.cjs +0 -27
  176. package/dist/metrics/base.cjs.map +1 -1
  177. package/dist/metrics/base.d.cts +105 -63
  178. package/dist/metrics/base.d.ts +105 -63
  179. package/dist/metrics/base.d.ts.map +1 -1
  180. package/dist/metrics/base.js +0 -19
  181. package/dist/metrics/base.js.map +1 -1
  182. package/dist/metrics/index.cjs +0 -3
  183. package/dist/metrics/index.cjs.map +1 -1
  184. package/dist/metrics/index.d.cts +2 -3
  185. package/dist/metrics/index.d.ts +2 -3
  186. package/dist/metrics/index.d.ts.map +1 -1
  187. package/dist/metrics/index.js +0 -2
  188. package/dist/metrics/index.js.map +1 -1
  189. package/dist/metrics/usage_collector.cjs +17 -12
  190. package/dist/metrics/usage_collector.cjs.map +1 -1
  191. package/dist/metrics/usage_collector.d.cts +3 -2
  192. package/dist/metrics/usage_collector.d.ts +3 -2
  193. package/dist/metrics/usage_collector.d.ts.map +1 -1
  194. package/dist/metrics/usage_collector.js +17 -12
  195. package/dist/metrics/usage_collector.js.map +1 -1
  196. package/dist/metrics/utils.cjs +22 -59
  197. package/dist/metrics/utils.cjs.map +1 -1
  198. package/dist/metrics/utils.d.cts +1 -8
  199. package/dist/metrics/utils.d.ts +1 -8
  200. package/dist/metrics/utils.d.ts.map +1 -1
  201. package/dist/metrics/utils.js +22 -52
  202. package/dist/metrics/utils.js.map +1 -1
  203. package/dist/multimodal/index.cjs +0 -2
  204. package/dist/multimodal/index.cjs.map +1 -1
  205. package/dist/multimodal/index.d.cts +0 -1
  206. package/dist/multimodal/index.d.ts +0 -1
  207. package/dist/multimodal/index.d.ts.map +1 -1
  208. package/dist/multimodal/index.js +0 -1
  209. package/dist/multimodal/index.js.map +1 -1
  210. package/dist/plugin.cjs +24 -8
  211. package/dist/plugin.cjs.map +1 -1
  212. package/dist/plugin.d.cts +18 -4
  213. package/dist/plugin.d.ts +18 -4
  214. package/dist/plugin.d.ts.map +1 -1
  215. package/dist/plugin.js +22 -7
  216. package/dist/plugin.js.map +1 -1
  217. package/dist/stream/deferred_stream.cjs +98 -0
  218. package/dist/stream/deferred_stream.cjs.map +1 -0
  219. package/dist/stream/deferred_stream.d.cts +27 -0
  220. package/dist/stream/deferred_stream.d.ts +27 -0
  221. package/dist/stream/deferred_stream.d.ts.map +1 -0
  222. package/dist/stream/deferred_stream.js +73 -0
  223. package/dist/stream/deferred_stream.js.map +1 -0
  224. package/dist/stream/deferred_stream.test.cjs +527 -0
  225. package/dist/stream/deferred_stream.test.cjs.map +1 -0
  226. package/dist/stream/deferred_stream.test.js +526 -0
  227. package/dist/stream/deferred_stream.test.js.map +1 -0
  228. package/dist/stream/identity_transform.cjs +42 -0
  229. package/dist/stream/identity_transform.cjs.map +1 -0
  230. package/dist/stream/identity_transform.d.cts +6 -0
  231. package/dist/stream/identity_transform.d.ts +6 -0
  232. package/dist/stream/identity_transform.d.ts.map +1 -0
  233. package/dist/stream/identity_transform.js +18 -0
  234. package/dist/stream/identity_transform.js.map +1 -0
  235. package/dist/stream/identity_transform.test.cjs +125 -0
  236. package/dist/stream/identity_transform.test.cjs.map +1 -0
  237. package/dist/stream/identity_transform.test.js +124 -0
  238. package/dist/stream/identity_transform.test.js.map +1 -0
  239. package/dist/stream/index.cjs +38 -0
  240. package/dist/stream/index.cjs.map +1 -0
  241. package/dist/stream/index.d.cts +5 -0
  242. package/dist/stream/index.d.ts +5 -0
  243. package/dist/stream/index.d.ts.map +1 -0
  244. package/dist/stream/index.js +11 -0
  245. package/dist/stream/index.js.map +1 -0
  246. package/dist/stream/merge_readable_streams.cjs +59 -0
  247. package/dist/stream/merge_readable_streams.cjs.map +1 -0
  248. package/dist/stream/merge_readable_streams.d.cts +4 -0
  249. package/dist/stream/merge_readable_streams.d.ts +4 -0
  250. package/dist/stream/merge_readable_streams.d.ts.map +1 -0
  251. package/dist/stream/merge_readable_streams.js +35 -0
  252. package/dist/stream/merge_readable_streams.js.map +1 -0
  253. package/dist/stream/stream_channel.cjs +47 -0
  254. package/dist/stream/stream_channel.cjs.map +1 -0
  255. package/dist/stream/stream_channel.d.cts +9 -0
  256. package/dist/stream/stream_channel.d.ts +9 -0
  257. package/dist/stream/stream_channel.d.ts.map +1 -0
  258. package/dist/stream/stream_channel.js +23 -0
  259. package/dist/stream/stream_channel.js.map +1 -0
  260. package/dist/stream/stream_channel.test.cjs +97 -0
  261. package/dist/stream/stream_channel.test.cjs.map +1 -0
  262. package/dist/stream/stream_channel.test.js +96 -0
  263. package/dist/stream/stream_channel.test.js.map +1 -0
  264. package/dist/stt/stream_adapter.cjs +3 -4
  265. package/dist/stt/stream_adapter.cjs.map +1 -1
  266. package/dist/stt/stream_adapter.d.cts +1 -0
  267. package/dist/stt/stream_adapter.d.ts +1 -0
  268. package/dist/stt/stream_adapter.d.ts.map +1 -1
  269. package/dist/stt/stream_adapter.js +3 -4
  270. package/dist/stt/stream_adapter.js.map +1 -1
  271. package/dist/stt/stt.cjs +101 -10
  272. package/dist/stt/stt.cjs.map +1 -1
  273. package/dist/stt/stt.d.cts +26 -5
  274. package/dist/stt/stt.d.ts +26 -5
  275. package/dist/stt/stt.d.ts.map +1 -1
  276. package/dist/stt/stt.js +102 -11
  277. package/dist/stt/stt.js.map +1 -1
  278. package/dist/tokenize/basic/basic.cjs +10 -5
  279. package/dist/tokenize/basic/basic.cjs.map +1 -1
  280. package/dist/tokenize/basic/basic.d.cts +7 -1
  281. package/dist/tokenize/basic/basic.d.ts +7 -1
  282. package/dist/tokenize/basic/basic.d.ts.map +1 -1
  283. package/dist/tokenize/basic/basic.js +10 -5
  284. package/dist/tokenize/basic/basic.js.map +1 -1
  285. package/dist/tokenize/basic/sentence.cjs +14 -6
  286. package/dist/tokenize/basic/sentence.cjs.map +1 -1
  287. package/dist/tokenize/basic/sentence.d.cts +1 -1
  288. package/dist/tokenize/basic/sentence.d.ts +1 -1
  289. package/dist/tokenize/basic/sentence.d.ts.map +1 -1
  290. package/dist/tokenize/basic/sentence.js +14 -6
  291. package/dist/tokenize/basic/sentence.js.map +1 -1
  292. package/dist/tokenize/token_stream.cjs +5 -3
  293. package/dist/tokenize/token_stream.cjs.map +1 -1
  294. package/dist/tokenize/token_stream.d.cts +1 -0
  295. package/dist/tokenize/token_stream.d.ts +1 -0
  296. package/dist/tokenize/token_stream.d.ts.map +1 -1
  297. package/dist/tokenize/token_stream.js +6 -4
  298. package/dist/tokenize/token_stream.js.map +1 -1
  299. package/dist/transcription.cjs +1 -2
  300. package/dist/transcription.cjs.map +1 -1
  301. package/dist/transcription.d.ts.map +1 -1
  302. package/dist/transcription.js +2 -3
  303. package/dist/transcription.js.map +1 -1
  304. package/dist/tts/index.cjs +2 -4
  305. package/dist/tts/index.cjs.map +1 -1
  306. package/dist/tts/index.d.cts +1 -1
  307. package/dist/tts/index.d.ts +1 -1
  308. package/dist/tts/index.d.ts.map +1 -1
  309. package/dist/tts/index.js +1 -3
  310. package/dist/tts/index.js.map +1 -1
  311. package/dist/tts/stream_adapter.cjs +26 -13
  312. package/dist/tts/stream_adapter.cjs.map +1 -1
  313. package/dist/tts/stream_adapter.d.cts +1 -1
  314. package/dist/tts/stream_adapter.d.ts +1 -1
  315. package/dist/tts/stream_adapter.d.ts.map +1 -1
  316. package/dist/tts/stream_adapter.js +27 -14
  317. package/dist/tts/stream_adapter.js.map +1 -1
  318. package/dist/tts/tts.cjs +157 -25
  319. package/dist/tts/tts.cjs.map +1 -1
  320. package/dist/tts/tts.d.cts +29 -5
  321. package/dist/tts/tts.d.ts +29 -5
  322. package/dist/tts/tts.d.ts.map +1 -1
  323. package/dist/tts/tts.js +157 -24
  324. package/dist/tts/tts.js.map +1 -1
  325. package/dist/types.cjs +60 -0
  326. package/dist/types.cjs.map +1 -0
  327. package/dist/types.d.cts +13 -0
  328. package/dist/types.d.ts +13 -0
  329. package/dist/types.d.ts.map +1 -0
  330. package/dist/types.js +35 -0
  331. package/dist/types.js.map +1 -0
  332. package/dist/utils.cjs +281 -27
  333. package/dist/utils.cjs.map +1 -1
  334. package/dist/utils.d.cts +134 -9
  335. package/dist/utils.d.ts +134 -9
  336. package/dist/utils.d.ts.map +1 -1
  337. package/dist/utils.js +265 -26
  338. package/dist/utils.js.map +1 -1
  339. package/dist/utils.test.cjs +492 -0
  340. package/dist/utils.test.cjs.map +1 -0
  341. package/dist/utils.test.js +498 -0
  342. package/dist/utils.test.js.map +1 -0
  343. package/dist/vad.cjs +76 -20
  344. package/dist/vad.cjs.map +1 -1
  345. package/dist/vad.d.cts +25 -5
  346. package/dist/vad.d.ts +25 -5
  347. package/dist/vad.d.ts.map +1 -1
  348. package/dist/vad.js +76 -20
  349. package/dist/vad.js.map +1 -1
  350. package/dist/voice/agent.cjs +245 -0
  351. package/dist/voice/agent.cjs.map +1 -0
  352. package/dist/voice/agent.d.cts +78 -0
  353. package/dist/voice/agent.d.ts +78 -0
  354. package/dist/voice/agent.d.ts.map +1 -0
  355. package/dist/voice/agent.js +220 -0
  356. package/dist/voice/agent.js.map +1 -0
  357. package/dist/voice/agent.test.cjs +61 -0
  358. package/dist/voice/agent.test.cjs.map +1 -0
  359. package/dist/voice/agent.test.js +60 -0
  360. package/dist/voice/agent.test.js.map +1 -0
  361. package/dist/voice/agent_activity.cjs +1453 -0
  362. package/dist/voice/agent_activity.cjs.map +1 -0
  363. package/dist/voice/agent_activity.d.cts +94 -0
  364. package/dist/voice/agent_activity.d.ts +94 -0
  365. package/dist/voice/agent_activity.d.ts.map +1 -0
  366. package/dist/voice/agent_activity.js +1449 -0
  367. package/dist/voice/agent_activity.js.map +1 -0
  368. package/dist/voice/agent_session.cjs +312 -0
  369. package/dist/voice/agent_session.cjs.map +1 -0
  370. package/dist/voice/agent_session.d.cts +121 -0
  371. package/dist/voice/agent_session.d.ts +121 -0
  372. package/dist/voice/agent_session.d.ts.map +1 -0
  373. package/dist/voice/agent_session.js +295 -0
  374. package/dist/voice/agent_session.js.map +1 -0
  375. package/dist/voice/audio_recognition.cjs +375 -0
  376. package/dist/voice/audio_recognition.cjs.map +1 -0
  377. package/dist/voice/audio_recognition.d.cts +80 -0
  378. package/dist/voice/audio_recognition.d.ts +80 -0
  379. package/dist/voice/audio_recognition.d.ts.map +1 -0
  380. package/dist/voice/audio_recognition.js +351 -0
  381. package/dist/voice/audio_recognition.js.map +1 -0
  382. package/dist/voice/events.cjs +145 -0
  383. package/dist/voice/events.cjs.map +1 -0
  384. package/dist/voice/events.d.cts +124 -0
  385. package/dist/voice/events.d.ts +124 -0
  386. package/dist/voice/events.d.ts.map +1 -0
  387. package/dist/voice/events.js +110 -0
  388. package/dist/voice/events.js.map +1 -0
  389. package/dist/voice/generation.cjs +700 -0
  390. package/dist/voice/generation.cjs.map +1 -0
  391. package/dist/voice/generation.d.cts +115 -0
  392. package/dist/voice/generation.d.ts +115 -0
  393. package/dist/voice/generation.d.ts.map +1 -0
  394. package/dist/voice/generation.js +672 -0
  395. package/dist/voice/generation.js.map +1 -0
  396. package/dist/voice/index.cjs +40 -0
  397. package/dist/voice/index.cjs.map +1 -0
  398. package/dist/voice/index.d.cts +5 -0
  399. package/dist/voice/index.d.ts +5 -0
  400. package/dist/voice/index.d.ts.map +1 -0
  401. package/dist/voice/index.js +11 -0
  402. package/dist/voice/index.js.map +1 -0
  403. package/dist/voice/io.cjs +245 -0
  404. package/dist/voice/io.cjs.map +1 -0
  405. package/dist/voice/io.d.cts +101 -0
  406. package/dist/voice/io.d.ts +101 -0
  407. package/dist/voice/io.d.ts.map +1 -0
  408. package/dist/voice/io.js +217 -0
  409. package/dist/voice/io.js.map +1 -0
  410. package/dist/voice/room_io/_input.cjs +121 -0
  411. package/dist/voice/room_io/_input.cjs.map +1 -0
  412. package/dist/voice/room_io/_input.d.cts +24 -0
  413. package/dist/voice/room_io/_input.d.ts +24 -0
  414. package/dist/voice/room_io/_input.d.ts.map +1 -0
  415. package/dist/voice/room_io/_input.js +102 -0
  416. package/dist/voice/room_io/_input.js.map +1 -0
  417. package/dist/voice/room_io/_output.cjs +358 -0
  418. package/dist/voice/room_io/_output.cjs.map +1 -0
  419. package/dist/voice/room_io/_output.d.cts +75 -0
  420. package/dist/voice/room_io/_output.d.ts +75 -0
  421. package/dist/voice/room_io/_output.d.ts.map +1 -0
  422. package/dist/voice/room_io/_output.js +342 -0
  423. package/dist/voice/room_io/_output.js.map +1 -0
  424. package/dist/voice/room_io/index.cjs +25 -0
  425. package/dist/voice/room_io/index.cjs.map +1 -0
  426. package/dist/voice/room_io/index.d.cts +3 -0
  427. package/dist/voice/room_io/index.d.ts +3 -0
  428. package/dist/voice/room_io/index.d.ts.map +1 -0
  429. package/dist/voice/room_io/index.js +3 -0
  430. package/dist/voice/room_io/index.js.map +1 -0
  431. package/dist/voice/room_io/room_io.cjs +370 -0
  432. package/dist/voice/room_io/room_io.cjs.map +1 -0
  433. package/dist/voice/room_io/room_io.d.cts +73 -0
  434. package/dist/voice/room_io/room_io.d.ts +73 -0
  435. package/dist/voice/room_io/room_io.d.ts.map +1 -0
  436. package/dist/voice/room_io/room_io.js +361 -0
  437. package/dist/voice/room_io/room_io.js.map +1 -0
  438. package/dist/{pipeline/index.cjs → voice/run_context.cjs} +16 -11
  439. package/dist/voice/run_context.cjs.map +1 -0
  440. package/dist/voice/run_context.d.cts +12 -0
  441. package/dist/voice/run_context.d.ts +12 -0
  442. package/dist/voice/run_context.d.ts.map +1 -0
  443. package/dist/voice/run_context.js +14 -0
  444. package/dist/voice/run_context.js.map +1 -0
  445. package/dist/voice/speech_handle.cjs +105 -0
  446. package/dist/voice/speech_handle.cjs.map +1 -0
  447. package/dist/voice/speech_handle.d.cts +46 -0
  448. package/dist/voice/speech_handle.d.ts +46 -0
  449. package/dist/voice/speech_handle.d.ts.map +1 -0
  450. package/dist/voice/speech_handle.js +81 -0
  451. package/dist/voice/speech_handle.js.map +1 -0
  452. package/dist/voice/transcription/_utils.cjs +45 -0
  453. package/dist/voice/transcription/_utils.cjs.map +1 -0
  454. package/dist/voice/transcription/_utils.d.cts +3 -0
  455. package/dist/voice/transcription/_utils.d.ts +3 -0
  456. package/dist/voice/transcription/_utils.d.ts.map +1 -0
  457. package/dist/voice/transcription/_utils.js +21 -0
  458. package/dist/voice/transcription/_utils.js.map +1 -0
  459. package/dist/voice/transcription/index.cjs +23 -0
  460. package/dist/voice/transcription/index.cjs.map +1 -0
  461. package/dist/voice/transcription/index.d.cts +2 -0
  462. package/dist/voice/transcription/index.d.ts +2 -0
  463. package/dist/voice/transcription/index.d.ts.map +1 -0
  464. package/dist/voice/transcription/index.js +2 -0
  465. package/dist/voice/transcription/index.js.map +1 -0
  466. package/dist/voice/transcription/synchronizer.cjs +380 -0
  467. package/dist/voice/transcription/synchronizer.cjs.map +1 -0
  468. package/dist/voice/transcription/synchronizer.d.cts +86 -0
  469. package/dist/voice/transcription/synchronizer.d.ts +86 -0
  470. package/dist/voice/transcription/synchronizer.d.ts.map +1 -0
  471. package/dist/voice/transcription/synchronizer.js +355 -0
  472. package/dist/voice/transcription/synchronizer.js.map +1 -0
  473. package/dist/worker.cjs +22 -4
  474. package/dist/worker.cjs.map +1 -1
  475. package/dist/worker.d.cts +1 -1
  476. package/dist/worker.d.ts +1 -1
  477. package/dist/worker.d.ts.map +1 -1
  478. package/dist/worker.js +22 -4
  479. package/dist/worker.js.map +1 -1
  480. package/package.json +9 -2
  481. package/src/_exceptions.ts +137 -0
  482. package/src/audio.ts +12 -1
  483. package/src/cli.ts +37 -0
  484. package/src/constants.ts +2 -1
  485. package/src/http_server.ts +1 -0
  486. package/src/index.ts +13 -10
  487. package/src/inference_runner.ts +2 -3
  488. package/src/ipc/inference_proc_executor.ts +2 -2
  489. package/src/ipc/job_executor.ts +1 -1
  490. package/src/ipc/job_proc_executor.ts +1 -1
  491. package/src/ipc/job_proc_lazy_main.ts +1 -1
  492. package/src/job.ts +18 -0
  493. package/src/llm/__snapshots__/chat_context.test.ts.snap +527 -0
  494. package/src/llm/__snapshots__/tool_context.test.ts.snap +177 -0
  495. package/src/llm/__snapshots__/utils.test.ts.snap +65 -0
  496. package/src/llm/chat_context.test.ts +450 -0
  497. package/src/llm/chat_context.ts +501 -103
  498. package/src/llm/index.ts +53 -18
  499. package/src/llm/llm.ts +149 -50
  500. package/src/llm/provider_format/google.test.ts +772 -0
  501. package/src/llm/provider_format/google.ts +130 -0
  502. package/src/llm/provider_format/index.ts +23 -0
  503. package/src/llm/provider_format/openai.test.ts +581 -0
  504. package/src/llm/provider_format/openai.ts +118 -0
  505. package/src/llm/provider_format/utils.ts +183 -0
  506. package/src/llm/realtime.ts +151 -0
  507. package/src/llm/remote_chat_context.test.ts +290 -0
  508. package/src/llm/remote_chat_context.ts +114 -0
  509. package/src/llm/tool_context.test.ts +198 -0
  510. package/src/llm/tool_context.ts +259 -0
  511. package/src/llm/tool_context.type.test.ts +115 -0
  512. package/src/llm/utils.test.ts +670 -0
  513. package/src/llm/utils.ts +324 -0
  514. package/src/metrics/base.ts +110 -78
  515. package/src/metrics/index.ts +3 -9
  516. package/src/metrics/usage_collector.ts +19 -13
  517. package/src/metrics/utils.ts +24 -69
  518. package/src/multimodal/index.ts +0 -1
  519. package/src/plugin.ts +26 -8
  520. package/src/stream/deferred_stream.test.ts +755 -0
  521. package/src/stream/deferred_stream.ts +110 -0
  522. package/src/stream/identity_transform.test.ts +179 -0
  523. package/src/stream/identity_transform.ts +18 -0
  524. package/src/stream/index.ts +7 -0
  525. package/src/stream/merge_readable_streams.ts +40 -0
  526. package/src/stream/stream_channel.test.ts +129 -0
  527. package/src/stream/stream_channel.ts +32 -0
  528. package/src/stt/stream_adapter.ts +3 -5
  529. package/src/stt/stt.ts +135 -17
  530. package/src/tokenize/basic/basic.ts +13 -5
  531. package/src/tokenize/basic/sentence.ts +20 -6
  532. package/src/tokenize/token_stream.ts +7 -4
  533. package/src/transcription.ts +2 -3
  534. package/src/tts/index.ts +0 -1
  535. package/src/tts/stream_adapter.ts +42 -16
  536. package/src/tts/tts.ts +203 -21
  537. package/src/types.ts +42 -0
  538. package/src/utils.test.ts +658 -0
  539. package/src/utils.ts +375 -44
  540. package/src/vad.ts +90 -22
  541. package/src/voice/agent.test.ts +80 -0
  542. package/src/voice/agent.ts +332 -0
  543. package/src/voice/agent_activity.ts +1913 -0
  544. package/src/voice/agent_session.ts +460 -0
  545. package/src/voice/audio_recognition.ts +474 -0
  546. package/src/voice/events.ts +252 -0
  547. package/src/voice/generation.ts +881 -0
  548. package/src/voice/index.ts +7 -0
  549. package/src/voice/io.ts +304 -0
  550. package/src/voice/room_io/_input.ts +144 -0
  551. package/src/voice/room_io/_output.ts +436 -0
  552. package/src/voice/room_io/index.ts +5 -0
  553. package/src/voice/room_io/room_io.ts +495 -0
  554. package/src/voice/run_context.ts +20 -0
  555. package/src/voice/speech_handle.ts +104 -0
  556. package/src/voice/transcription/_utils.ts +25 -0
  557. package/src/voice/transcription/index.ts +4 -0
  558. package/src/voice/transcription/synchronizer.ts +478 -0
  559. package/src/worker.ts +22 -2
  560. package/dist/llm/function_context.cjs +0 -103
  561. package/dist/llm/function_context.cjs.map +0 -1
  562. package/dist/llm/function_context.d.cts +0 -47
  563. package/dist/llm/function_context.d.ts +0 -47
  564. package/dist/llm/function_context.d.ts.map +0 -1
  565. package/dist/llm/function_context.js +0 -78
  566. package/dist/llm/function_context.js.map +0 -1
  567. package/dist/llm/function_context.test.cjs +0 -218
  568. package/dist/llm/function_context.test.cjs.map +0 -1
  569. package/dist/llm/function_context.test.js +0 -217
  570. package/dist/llm/function_context.test.js.map +0 -1
  571. package/dist/multimodal/multimodal_agent.cjs +0 -486
  572. package/dist/multimodal/multimodal_agent.cjs.map +0 -1
  573. package/dist/multimodal/multimodal_agent.d.cts +0 -48
  574. package/dist/multimodal/multimodal_agent.d.ts +0 -48
  575. package/dist/multimodal/multimodal_agent.d.ts.map +0 -1
  576. package/dist/multimodal/multimodal_agent.js +0 -461
  577. package/dist/multimodal/multimodal_agent.js.map +0 -1
  578. package/dist/pipeline/agent_output.cjs +0 -197
  579. package/dist/pipeline/agent_output.cjs.map +0 -1
  580. package/dist/pipeline/agent_output.d.cts +0 -33
  581. package/dist/pipeline/agent_output.d.ts +0 -33
  582. package/dist/pipeline/agent_output.d.ts.map +0 -1
  583. package/dist/pipeline/agent_output.js +0 -172
  584. package/dist/pipeline/agent_output.js.map +0 -1
  585. package/dist/pipeline/agent_playout.cjs +0 -175
  586. package/dist/pipeline/agent_playout.cjs.map +0 -1
  587. package/dist/pipeline/agent_playout.d.cts +0 -40
  588. package/dist/pipeline/agent_playout.d.ts +0 -40
  589. package/dist/pipeline/agent_playout.d.ts.map +0 -1
  590. package/dist/pipeline/agent_playout.js +0 -139
  591. package/dist/pipeline/agent_playout.js.map +0 -1
  592. package/dist/pipeline/human_input.cjs +0 -171
  593. package/dist/pipeline/human_input.cjs.map +0 -1
  594. package/dist/pipeline/human_input.d.cts +0 -30
  595. package/dist/pipeline/human_input.d.ts +0 -30
  596. package/dist/pipeline/human_input.d.ts.map +0 -1
  597. package/dist/pipeline/human_input.js +0 -146
  598. package/dist/pipeline/human_input.js.map +0 -1
  599. package/dist/pipeline/index.cjs.map +0 -1
  600. package/dist/pipeline/index.d.cts +0 -2
  601. package/dist/pipeline/index.d.ts +0 -2
  602. package/dist/pipeline/index.d.ts.map +0 -1
  603. package/dist/pipeline/index.js +0 -11
  604. package/dist/pipeline/index.js.map +0 -1
  605. package/dist/pipeline/pipeline_agent.cjs +0 -859
  606. package/dist/pipeline/pipeline_agent.cjs.map +0 -1
  607. package/dist/pipeline/pipeline_agent.d.cts +0 -150
  608. package/dist/pipeline/pipeline_agent.d.ts +0 -150
  609. package/dist/pipeline/pipeline_agent.d.ts.map +0 -1
  610. package/dist/pipeline/pipeline_agent.js +0 -837
  611. package/dist/pipeline/pipeline_agent.js.map +0 -1
  612. package/dist/pipeline/speech_handle.cjs +0 -176
  613. package/dist/pipeline/speech_handle.cjs.map +0 -1
  614. package/dist/pipeline/speech_handle.d.cts +0 -37
  615. package/dist/pipeline/speech_handle.d.ts +0 -37
  616. package/dist/pipeline/speech_handle.d.ts.map +0 -1
  617. package/dist/pipeline/speech_handle.js +0 -152
  618. package/dist/pipeline/speech_handle.js.map +0 -1
  619. package/src/llm/function_context.test.ts +0 -248
  620. package/src/llm/function_context.ts +0 -142
  621. package/src/multimodal/multimodal_agent.ts +0 -592
  622. package/src/pipeline/agent_output.ts +0 -219
  623. package/src/pipeline/agent_playout.ts +0 -192
  624. package/src/pipeline/human_input.ts +0 -188
  625. package/src/pipeline/index.ts +0 -15
  626. package/src/pipeline/pipeline_agent.ts +0 -1197
  627. package/src/pipeline/speech_handle.ts +0 -201
@@ -1,592 +0,0 @@
1
- // SPDX-FileCopyrightText: 2024 LiveKit, Inc.
2
- //
3
- // SPDX-License-Identifier: Apache-2.0
4
- import type {
5
- LocalTrackPublication,
6
- NoiseCancellationOptions,
7
- RemoteAudioTrack,
8
- RemoteParticipant,
9
- RemoteTrack,
10
- RemoteTrackPublication,
11
- Room,
12
- } from '@livekit/rtc-node';
13
- import {
14
- AudioSource,
15
- AudioStream,
16
- LocalAudioTrack,
17
- RoomEvent,
18
- TrackPublishOptions,
19
- TrackSource,
20
- } from '@livekit/rtc-node';
21
- import { randomUUID } from 'node:crypto';
22
- import { EventEmitter } from 'node:events';
23
- import { AudioByteStream } from '../audio.js';
24
- import {
25
- ATTRIBUTE_SEGMENT_ID,
26
- ATTRIBUTE_TRANSCRIPTION_FINAL,
27
- ATTRIBUTE_TRANSCRIPTION_TRACK_ID,
28
- TOPIC_TRANSCRIPTION,
29
- } from '../constants.js';
30
- import * as llm from '../llm/index.js';
31
- import { log } from '../log.js';
32
- import type { MultimodalLLMMetrics } from '../metrics/base.js';
33
- import { TextAudioSynchronizer, defaultTextSyncOptions } from '../transcription.js';
34
- import { findMicroTrackId } from '../utils.js';
35
- import { AgentPlayout, type PlayoutHandle } from './agent_playout.js';
36
-
37
- /**
38
- * @internal
39
- * @beta
40
- */
41
- export abstract class RealtimeSession extends EventEmitter {
42
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
43
- abstract conversation: any; // openai.realtime.Conversation
44
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
45
- abstract inputAudioBuffer: any; // openai.realtime.InputAudioBuffer
46
- abstract fncCtx: llm.FunctionContext | undefined;
47
- abstract recoverFromTextResponse(itemId: string): void;
48
- }
49
-
50
- /**
51
- * @internal
52
- * @beta
53
- */
54
- export abstract class RealtimeModel {
55
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
56
- abstract session(options: any): RealtimeSession; // openai.realtime.ModelOptions
57
- abstract close(): Promise<void>;
58
- abstract sampleRate: number;
59
- abstract numChannels: number;
60
- abstract inFrameSize: number;
61
- abstract outFrameSize: number;
62
- }
63
-
64
- export type AgentState = 'initializing' | 'thinking' | 'listening' | 'speaking';
65
- export const AGENT_STATE_ATTRIBUTE = 'lk.agent.state';
66
-
67
- /** @beta */
68
- export class MultimodalAgent extends EventEmitter {
69
- model: RealtimeModel;
70
- room: Room | null = null;
71
- linkedParticipant: RemoteParticipant | null = null;
72
- subscribedTrack: RemoteAudioTrack | null = null;
73
- readMicroTask: Promise<void> | null = null;
74
-
75
- #textResponseRetries = 0;
76
- #maxTextResponseRetries: number;
77
- #transcriptionId?: string;
78
-
79
- constructor({
80
- model,
81
- chatCtx,
82
- fncCtx,
83
- maxTextResponseRetries = 5,
84
- noiseCancellation,
85
- }: {
86
- model: RealtimeModel;
87
- chatCtx?: llm.ChatContext;
88
- fncCtx?: llm.FunctionContext;
89
- maxTextResponseRetries?: number;
90
- noiseCancellation?: NoiseCancellationOptions;
91
- }) {
92
- super();
93
- this.model = model;
94
- this.#chatCtx = chatCtx;
95
- this.#fncCtx = fncCtx;
96
- this.#maxTextResponseRetries = maxTextResponseRetries;
97
- this.#noiseCancellation = noiseCancellation;
98
- }
99
-
100
- #participant: RemoteParticipant | string | null = null;
101
- #agentPublication: LocalTrackPublication | null = null;
102
- #localTrackSid: string | null = null;
103
- #localSource: AudioSource | null = null;
104
- #agentPlayout: AgentPlayout | null = null;
105
- #playingHandle: PlayoutHandle | undefined = undefined;
106
- #logger = log();
107
- #session: RealtimeSession | null = null;
108
- #fncCtx: llm.FunctionContext | undefined = undefined;
109
- #chatCtx: llm.ChatContext | undefined = undefined;
110
- #noiseCancellation: NoiseCancellationOptions | undefined = undefined;
111
-
112
- #_started: boolean = false;
113
- #_pendingFunctionCalls: Set<string> = new Set();
114
- #_speaking: boolean = false;
115
-
116
- get fncCtx(): llm.FunctionContext | undefined {
117
- return this.#fncCtx;
118
- }
119
-
120
- set fncCtx(ctx: llm.FunctionContext | undefined) {
121
- this.#fncCtx = ctx;
122
- if (this.#session) {
123
- this.#session.fncCtx = ctx;
124
- }
125
- }
126
-
127
- get #pendingFunctionCalls(): Set<string> {
128
- return this.#_pendingFunctionCalls;
129
- }
130
-
131
- set #pendingFunctionCalls(calls: Set<string>) {
132
- this.#_pendingFunctionCalls = calls;
133
- this.#updateState();
134
- }
135
-
136
- get #speaking(): boolean {
137
- return this.#_speaking;
138
- }
139
-
140
- set #speaking(isSpeaking: boolean) {
141
- this.#_speaking = isSpeaking;
142
- this.#updateState();
143
- }
144
-
145
- get #started(): boolean {
146
- return this.#_started;
147
- }
148
-
149
- set #started(started: boolean) {
150
- this.#_started = started;
151
- this.#updateState();
152
- }
153
-
154
- start(
155
- room: Room,
156
- participant: RemoteParticipant | string | null = null,
157
- ): Promise<RealtimeSession> {
158
- return new Promise(async (resolve, reject) => {
159
- if (this.#started) {
160
- reject(new Error('MultimodalAgent already started'));
161
- }
162
- this.#updateState();
163
-
164
- room.on(RoomEvent.ParticipantConnected, (participant: RemoteParticipant) => {
165
- // automatically link to the first participant that connects, if not already linked
166
- if (this.linkedParticipant) {
167
- return;
168
- }
169
- this.#linkParticipant(participant.identity!);
170
- });
171
- room.on(
172
- RoomEvent.TrackPublished,
173
- (trackPublication: RemoteTrackPublication, participant: RemoteParticipant) => {
174
- if (
175
- this.linkedParticipant &&
176
- participant.identity === this.linkedParticipant.identity &&
177
- trackPublication.source === TrackSource.SOURCE_MICROPHONE &&
178
- !trackPublication.subscribed
179
- ) {
180
- trackPublication.setSubscribed(true);
181
- }
182
- },
183
- );
184
- room.on(RoomEvent.TrackSubscribed, this.#handleTrackSubscription.bind(this));
185
-
186
- this.room = room;
187
- this.#participant = participant;
188
-
189
- this.#localSource = new AudioSource(this.model.sampleRate, this.model.numChannels);
190
- this.#agentPlayout = new AgentPlayout(
191
- this.#localSource,
192
- this.model.sampleRate,
193
- this.model.numChannels,
194
- this.model.inFrameSize,
195
- this.model.outFrameSize,
196
- );
197
- const onPlayoutStarted = () => {
198
- this.emit('agent_started_speaking');
199
- this.#speaking = true;
200
- };
201
-
202
- const onPlayoutStopped = (interrupted: boolean) => {
203
- this.emit('agent_stopped_speaking');
204
- this.#speaking = false;
205
- if (this.#playingHandle) {
206
- let text = this.#playingHandle.synchronizer.playedText;
207
- if (interrupted) {
208
- text += '…';
209
- }
210
- const msg = llm.ChatMessage.create({
211
- role: llm.ChatRole.ASSISTANT,
212
- text,
213
- });
214
-
215
- if (interrupted) {
216
- this.emit('agent_speech_interrupted', msg);
217
- } else {
218
- this.emit('agent_speech_committed', msg);
219
- }
220
- this.#logger.child({ transcription: text, interrupted }).debug('committed agent speech');
221
- }
222
- };
223
-
224
- this.#agentPlayout.on('playout_started', onPlayoutStarted);
225
- this.#agentPlayout.on('playout_stopped', onPlayoutStopped);
226
-
227
- const track = LocalAudioTrack.createAudioTrack('assistant_voice', this.#localSource);
228
- const options = new TrackPublishOptions();
229
- options.source = TrackSource.SOURCE_MICROPHONE;
230
- this.#agentPublication = (await room.localParticipant?.publishTrack(track, options)) || null;
231
- if (!this.#agentPublication) {
232
- this.#logger.error('Failed to publish track');
233
- reject(new Error('Failed to publish track'));
234
- return;
235
- }
236
-
237
- await this.#agentPublication.waitForSubscription();
238
-
239
- if (participant) {
240
- if (typeof participant === 'string') {
241
- this.#linkParticipant(participant);
242
- } else {
243
- this.#linkParticipant(participant.identity!);
244
- }
245
- } else {
246
- // No participant specified, try to find the first participant in the room
247
- for (const participant of room.remoteParticipants.values()) {
248
- this.#linkParticipant(participant.identity!);
249
- break;
250
- }
251
- }
252
-
253
- this.#session = this.model.session({ fncCtx: this.#fncCtx, chatCtx: this.#chatCtx });
254
- this.#started = true;
255
-
256
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
257
- this.#session.on('response_content_added', (message: any) => {
258
- // openai.realtime.RealtimeContent
259
- if (message.contentType === 'text') return;
260
-
261
- const synchronizer = new TextAudioSynchronizer(defaultTextSyncOptions);
262
- synchronizer.on('textUpdated', async (text) => {
263
- if (!this.#transcriptionId) {
264
- this.#transcriptionId = randomUUID();
265
- }
266
- await this.#publishTranscription(
267
- this.room!.localParticipant!.identity!,
268
- this.#getLocalTrackSid()!,
269
- text.text,
270
- text.final,
271
- text.id,
272
- this.#transcriptionId,
273
- );
274
- if (text.final) {
275
- this.#transcriptionId = undefined;
276
- }
277
- });
278
-
279
- const handle = this.#agentPlayout?.play(
280
- message.itemId,
281
- message.contentIndex,
282
- synchronizer,
283
- message.textStream,
284
- message.audioStream,
285
- );
286
- this.#playingHandle = handle;
287
- });
288
-
289
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
290
- this.#session.on('response_content_done', (message: any) => {
291
- // openai.realtime.RealtimeContent
292
- if (message.contentType === 'text') {
293
- if (this.#textResponseRetries >= this.#maxTextResponseRetries) {
294
- throw new Error(
295
- 'The OpenAI Realtime API returned a text response ' +
296
- `after ${this.#maxTextResponseRetries} retries. ` +
297
- 'Please try to reduce the number of text system or ' +
298
- 'assistant messages in the chat context.',
299
- );
300
- }
301
-
302
- this.#textResponseRetries++;
303
- this.#logger
304
- .child({
305
- itemId: message.itemId,
306
- text: message.text,
307
- retries: this.#textResponseRetries,
308
- })
309
- .warn(
310
- 'The OpenAI Realtime API returned a text response instead of audio. ' +
311
- 'Attempting to recover to audio mode...',
312
- );
313
- this.#session!.recoverFromTextResponse(message.itemId);
314
- } else {
315
- this.#textResponseRetries = 0;
316
- }
317
- });
318
-
319
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
320
- this.#session.on('input_speech_committed', async (ev: any) => {
321
- // openai.realtime.InputSpeechCommittedEvent
322
- const participantIdentity = this.linkedParticipant?.identity;
323
- const trackSid = this.subscribedTrack?.sid;
324
- if (participantIdentity && trackSid) {
325
- if (!this.#transcriptionId) {
326
- this.#transcriptionId = randomUUID();
327
- }
328
- await this.#publishTranscription(
329
- participantIdentity,
330
- trackSid,
331
- '…',
332
- false,
333
- ev.itemId,
334
- this.#transcriptionId,
335
- );
336
- } else {
337
- this.#logger.error('Participant or track not set');
338
- }
339
- });
340
-
341
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
342
- this.#session.on('input_speech_transcription_completed', async (ev: any) => {
343
- // openai.realtime.InputSpeechTranscriptionCompletedEvent
344
- const transcription = ev.transcript;
345
- const participantIdentity = this.linkedParticipant?.identity;
346
- const trackSid = this.subscribedTrack?.sid;
347
- if (participantIdentity && trackSid) {
348
- if (!this.#transcriptionId) {
349
- this.#transcriptionId = randomUUID();
350
- }
351
- await this.#publishTranscription(
352
- participantIdentity,
353
- trackSid,
354
- transcription,
355
- true,
356
- ev.itemId,
357
- this.#transcriptionId,
358
- );
359
- this.#transcriptionId = undefined;
360
- } else {
361
- this.#logger.error('Participant or track not set');
362
- }
363
- const userMsg = llm.ChatMessage.create({
364
- role: llm.ChatRole.USER,
365
- text: transcription,
366
- });
367
- this.emit('user_speech_committed', userMsg);
368
- this.#logger.child({ transcription }).debug('committed user speech');
369
- });
370
-
371
- this.#session.on('input_speech_started', async (ev: any) => {
372
- this.emit('user_started_speaking');
373
- if (this.#playingHandle && !this.#playingHandle.done) {
374
- this.#playingHandle.interrupt();
375
-
376
- this.#session!.conversation.item.truncate(
377
- this.#playingHandle.itemId,
378
- this.#playingHandle.contentIndex,
379
- Math.floor((this.#playingHandle.audioSamples / 24000) * 1000),
380
- );
381
-
382
- this.#playingHandle = undefined;
383
- }
384
-
385
- const participantIdentity = this.linkedParticipant?.identity;
386
- const trackSid = this.subscribedTrack?.sid;
387
- if (participantIdentity && trackSid) {
388
- if (!this.#transcriptionId) {
389
- this.#transcriptionId = randomUUID();
390
- }
391
- await this.#publishTranscription(
392
- participantIdentity,
393
- trackSid,
394
- '…',
395
- false,
396
- ev.itemId,
397
- this.#transcriptionId,
398
- );
399
- }
400
- });
401
-
402
- // eslint-disable-next-line @typescript-eslint/no-unused-vars
403
- this.#session.on('input_speech_stopped', (ev: any) => {
404
- this.emit('user_stopped_speaking');
405
- });
406
-
407
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
408
- this.#session.on('function_call_started', (ev: any) => {
409
- this.#pendingFunctionCalls.add(ev.callId);
410
- this.#updateState();
411
- });
412
-
413
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
414
- this.#session.on('function_call_completed', (ev: any) => {
415
- this.#pendingFunctionCalls.delete(ev.callId);
416
- this.#updateState();
417
- });
418
-
419
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
420
- this.#session.on('function_call_failed', (ev: any) => {
421
- this.#pendingFunctionCalls.delete(ev.callId);
422
- this.#updateState();
423
- });
424
-
425
- this.#session.on('metrics_collected', (metrics: MultimodalLLMMetrics) => {
426
- this.emit('metrics_collected', metrics);
427
- });
428
-
429
- resolve(this.#session);
430
- });
431
- }
432
-
433
- #linkParticipant(participantIdentity: string): void {
434
- if (!this.room) {
435
- this.#logger.error('Room is not set');
436
- return;
437
- }
438
-
439
- this.linkedParticipant = this.room.remoteParticipants.get(participantIdentity) || null;
440
- if (!this.linkedParticipant) {
441
- this.#logger.error(`Participant with identity ${participantIdentity} not found`);
442
- return;
443
- }
444
-
445
- if (this.linkedParticipant.trackPublications.size > 0) {
446
- this.#subscribeToMicrophone();
447
- }
448
-
449
- // also check if already subscribed
450
- for (const publication of this.linkedParticipant.trackPublications.values()) {
451
- if (publication.source === TrackSource.SOURCE_MICROPHONE && publication.track) {
452
- this.#handleTrackSubscription(publication.track, publication, this.linkedParticipant);
453
- break;
454
- }
455
- }
456
- }
457
-
458
- #subscribeToMicrophone(): void {
459
- if (!this.linkedParticipant) {
460
- this.#logger.error('Participant is not set');
461
- return;
462
- }
463
-
464
- let microphonePublication: RemoteTrackPublication | undefined = undefined;
465
- for (const publication of this.linkedParticipant.trackPublications.values()) {
466
- if (publication.source === TrackSource.SOURCE_MICROPHONE) {
467
- microphonePublication = publication;
468
- break;
469
- }
470
- }
471
- if (!microphonePublication) {
472
- return;
473
- }
474
-
475
- if (!microphonePublication.subscribed) {
476
- microphonePublication.setSubscribed(true);
477
- }
478
- }
479
-
480
- #handleTrackSubscription(
481
- track: RemoteTrack,
482
- publication: RemoteTrackPublication,
483
- participant: RemoteParticipant,
484
- ) {
485
- if (
486
- publication.source !== TrackSource.SOURCE_MICROPHONE ||
487
- participant.identity !== this.linkedParticipant?.identity
488
- ) {
489
- return;
490
- }
491
- const readAudioStreamTask = async (audioStream: AudioStream) => {
492
- const bstream = new AudioByteStream(
493
- this.model.sampleRate,
494
- this.model.numChannels,
495
- this.model.inFrameSize,
496
- );
497
-
498
- for await (const frame of audioStream) {
499
- const audioData = frame.data;
500
- for (const frame of bstream.write(audioData.buffer)) {
501
- this.#session!.inputAudioBuffer.append(frame);
502
- }
503
- }
504
- };
505
- this.subscribedTrack = track;
506
-
507
- this.readMicroTask = new Promise<void>((resolve, reject) => {
508
- const audioStreamOptions = {
509
- sampleRate: this.model.sampleRate,
510
- numChannels: this.model.numChannels,
511
- ...(this.#noiseCancellation ? { noiseCancellation: this.#noiseCancellation } : {}),
512
- };
513
- readAudioStreamTask(new AudioStream(track, audioStreamOptions)).then(resolve).catch(reject);
514
- });
515
- }
516
-
517
- #getLocalTrackSid(): string | null {
518
- if (!this.#localTrackSid && this.room && this.room.localParticipant) {
519
- this.#localTrackSid = findMicroTrackId(this.room, this.room.localParticipant!.identity!);
520
- }
521
- return this.#localTrackSid;
522
- }
523
-
524
- async #publishTranscription(
525
- participantIdentity: string,
526
- trackSid: string,
527
- text: string,
528
- isFinal: boolean,
529
- id: string,
530
- segmentId: string,
531
- ): Promise<void> {
532
- this.#logger.debug(
533
- `Publishing transcription ${participantIdentity} ${trackSid} ${text} ${isFinal} ${id}`,
534
- );
535
- if (!this.room?.localParticipant) {
536
- this.#logger.error('Room or local participant not set');
537
- return;
538
- }
539
-
540
- this.room.localParticipant.publishTranscription({
541
- participantIdentity,
542
- trackSid,
543
- segments: [
544
- {
545
- text,
546
- final: isFinal,
547
- id,
548
- startTime: BigInt(0),
549
- endTime: BigInt(0),
550
- language: '',
551
- },
552
- ],
553
- });
554
-
555
- const stream = await this.room.localParticipant.streamText({
556
- topic: TOPIC_TRANSCRIPTION,
557
- senderIdentity: participantIdentity,
558
- attributes: {
559
- [ATTRIBUTE_TRANSCRIPTION_TRACK_ID]: trackSid,
560
- [ATTRIBUTE_TRANSCRIPTION_FINAL]: isFinal.toString(),
561
- [ATTRIBUTE_SEGMENT_ID]: segmentId,
562
- },
563
- });
564
- await stream.write(text);
565
- await stream.close();
566
- }
567
-
568
- #updateState() {
569
- let newState: AgentState = 'initializing';
570
- if (this.#pendingFunctionCalls.size > 0) {
571
- newState = 'thinking';
572
- } else if (this.#speaking) {
573
- newState = 'speaking';
574
- } else if (this.#started) {
575
- newState = 'listening';
576
- }
577
-
578
- this.#setState(newState);
579
- }
580
-
581
- #setState(state: AgentState) {
582
- if (this.room?.isConnected && this.room.localParticipant) {
583
- const currentState = this.room.localParticipant.attributes![AGENT_STATE_ATTRIBUTE];
584
- if (currentState !== state) {
585
- this.room.localParticipant.setAttributes({
586
- [AGENT_STATE_ATTRIBUTE]: state,
587
- });
588
- this.#logger.debug(`${AGENT_STATE_ATTRIBUTE}: ${currentState} ->${state}`);
589
- }
590
- }
591
- }
592
- }