ruby_llm 1.16.0 → 2.0.0.rc1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (474) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +85 -32
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -6
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +25 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
  91. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +100 -0
  92. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
  93. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
  94. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  96. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
  97. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
  98. data/lib/ruby_llm/accounting/usage.rb +245 -0
  99. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  100. data/lib/ruby_llm/active_record/attachment_helpers.rb +180 -0
  101. data/lib/ruby_llm/active_record/batch.rb +97 -0
  102. data/lib/ruby_llm/active_record/chat_methods.rb +823 -305
  103. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  104. data/lib/ruby_llm/active_record/model.rb +135 -0
  105. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  106. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  107. data/lib/ruby_llm/active_record/usage.rb +61 -0
  108. data/lib/ruby_llm/agent.rb +1065 -151
  109. data/lib/ruby_llm/aliases.json +268 -100
  110. data/lib/ruby_llm/attachment.rb +187 -48
  111. data/lib/ruby_llm/batch.rb +432 -0
  112. data/lib/ruby_llm/cached_content.rb +112 -0
  113. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  114. data/lib/ruby_llm/chat.rb +1127 -198
  115. data/lib/ruby_llm/chunk.rb +10 -0
  116. data/lib/ruby_llm/citation.rb +105 -0
  117. data/lib/ruby_llm/configuration.rb +261 -24
  118. data/lib/ruby_llm/context.rb +128 -6
  119. data/lib/ruby_llm/cost.rb +217 -80
  120. data/lib/ruby_llm/downloaded_file.rb +33 -0
  121. data/lib/ruby_llm/embedding.rb +121 -17
  122. data/lib/ruby_llm/embedding_request.rb +53 -0
  123. data/lib/ruby_llm/error.rb +159 -23
  124. data/lib/ruby_llm/fallback.rb +133 -0
  125. data/lib/ruby_llm/files/mime_type.rb +97 -0
  126. data/lib/ruby_llm/image.rb +153 -32
  127. data/lib/ruby_llm/message.rb +233 -54
  128. data/lib/ruby_llm/model/modalities.rb +17 -4
  129. data/lib/ruby_llm/model/pricing.rb +24 -5
  130. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  131. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  132. data/lib/ruby_llm/model.rb +244 -2
  133. data/lib/ruby_llm/models/aliases.rb +41 -0
  134. data/lib/ruby_llm/models/registry.rb +165 -0
  135. data/lib/ruby_llm/models/schema.rb +99 -0
  136. data/lib/ruby_llm/models.json +71172 -33253
  137. data/lib/ruby_llm/models.rb +477 -215
  138. data/lib/ruby_llm/moderation.rb +139 -26
  139. data/lib/ruby_llm/ocr.rb +112 -0
  140. data/lib/ruby_llm/prompt.rb +79 -0
  141. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  142. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  143. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  144. data/lib/ruby_llm/protocol.rb +662 -0
  145. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  146. data/lib/ruby_llm/protocols/anthropic/chat.rb +540 -0
  147. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  148. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  149. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  150. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  151. data/lib/ruby_llm/protocols/anthropic/streaming.rb +166 -0
  152. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  153. data/lib/ruby_llm/protocols/anthropic.rb +100 -0
  154. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  155. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
  156. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  157. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  158. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  159. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  160. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  161. data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
  162. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  163. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  164. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  165. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  166. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  167. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  168. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
  169. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  170. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  171. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  172. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  173. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  174. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  175. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  176. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  177. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  178. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  179. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  180. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  181. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  182. data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
  183. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  184. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  185. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  186. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  187. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  188. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  189. data/lib/ruby_llm/protocols/converse/chat.rb +685 -0
  190. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  191. data/lib/ruby_llm/protocols/converse/streaming.rb +424 -0
  192. data/lib/ruby_llm/protocols/converse.rb +54 -0
  193. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  194. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  195. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  196. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  197. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  198. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  199. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  200. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  201. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  202. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  203. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  204. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  208. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  209. data/lib/ruby_llm/protocols/files.rb +119 -0
  210. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  211. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  212. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  213. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
  214. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  215. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  216. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  217. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  218. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  219. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  220. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  221. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  222. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  223. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  224. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  225. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  226. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  227. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  228. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  229. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  230. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  231. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  232. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  233. data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
  234. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  235. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  236. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  237. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  238. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  239. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  240. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  241. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  242. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  243. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  244. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  245. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  246. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  247. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  248. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  249. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  250. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  251. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  252. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  253. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  254. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  255. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  256. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  257. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  258. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  259. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  260. data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
  261. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  262. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  263. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  264. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  265. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  266. data/lib/ruby_llm/protocols/responses.rb +35 -0
  267. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  268. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  269. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  270. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  271. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  272. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  273. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  274. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  275. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  276. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  277. data/lib/ruby_llm/provider.rb +560 -128
  278. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  279. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  280. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  281. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  282. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  283. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  284. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  285. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  286. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  287. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  288. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  289. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  290. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  291. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  292. data/lib/ruby_llm/providers/azure.rb +77 -78
  293. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  294. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  295. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  296. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  297. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  298. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  299. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  300. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  301. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  302. data/lib/ruby_llm/providers/cohere.rb +31 -0
  303. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  304. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  305. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  306. data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
  307. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  308. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  309. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  310. data/lib/ruby_llm/providers/gemini.rb +15 -8
  311. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  312. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  313. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  314. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  315. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  316. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  317. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  318. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  319. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  320. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  321. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  322. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  323. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  324. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  325. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  326. data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
  327. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  328. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  329. data/lib/ruby_llm/providers/mistral.rb +16 -4
  330. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  331. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  332. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  333. data/lib/ruby_llm/providers/ollama.rb +9 -8
  334. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  335. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  336. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  337. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  338. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  339. data/lib/ruby_llm/providers/openai.rb +92 -11
  340. data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
  341. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  342. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  343. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  344. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  345. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  346. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  347. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  348. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  349. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  350. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  351. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  352. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  353. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  354. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  355. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  356. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  357. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  358. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  359. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  360. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  361. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  362. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  363. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  364. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  365. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  366. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  367. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  368. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  369. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  370. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  371. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  372. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  373. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  374. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  375. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  376. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  377. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  378. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  379. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  380. data/lib/ruby_llm/providers/xai.rb +15 -5
  381. data/lib/ruby_llm/railtie.rb +7 -16
  382. data/lib/ruby_llm/rerank.rb +105 -0
  383. data/lib/ruby_llm/research_job.rb +241 -0
  384. data/lib/ruby_llm/search_results.rb +68 -0
  385. data/lib/ruby_llm/server_tool_call.rb +73 -0
  386. data/lib/ruby_llm/speech.rb +159 -0
  387. data/lib/ruby_llm/speech_chunk.rb +33 -0
  388. data/lib/ruby_llm/support/deprecator.rb +22 -0
  389. data/lib/ruby_llm/support/inspectable.rb +49 -0
  390. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  391. data/lib/ruby_llm/support/utils.rb +147 -0
  392. data/lib/ruby_llm/thinking.rb +127 -20
  393. data/lib/ruby_llm/tokenization.rb +59 -0
  394. data/lib/ruby_llm/tokens.rb +103 -33
  395. data/lib/ruby_llm/tool.rb +266 -91
  396. data/lib/ruby_llm/tool_call.rb +36 -3
  397. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  398. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  399. data/lib/ruby_llm/transcription.rb +138 -13
  400. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  401. data/lib/ruby_llm/transport/connection.rb +193 -0
  402. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  403. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  404. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  405. data/lib/ruby_llm/uploaded_file.rb +144 -0
  406. data/lib/ruby_llm/version.rb +2 -1
  407. data/lib/ruby_llm/video.rb +136 -0
  408. data/lib/ruby_llm/video_job.rb +150 -0
  409. data/lib/ruby_llm/workflow.rb +91 -0
  410. data/lib/ruby_llm.rb +380 -6
  411. data/lib/tasks/ruby_llm.rake +21 -16
  412. data/skills/rubyllm/SKILL.md +81 -0
  413. data/skills/rubyllm/agents/openai.yaml +4 -0
  414. metadata +338 -97
  415. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  416. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  417. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  418. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  419. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  420. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  421. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  422. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  423. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  424. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  425. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  426. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  427. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  428. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  429. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  430. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  431. data/lib/ruby_llm/aliases.rb +0 -41
  432. data/lib/ruby_llm/connection.rb +0 -159
  433. data/lib/ruby_llm/content.rb +0 -91
  434. data/lib/ruby_llm/deprecator.rb +0 -24
  435. data/lib/ruby_llm/error_middleware.rb +0 -81
  436. data/lib/ruby_llm/instrumentation.rb +0 -36
  437. data/lib/ruby_llm/mime_type.rb +0 -96
  438. data/lib/ruby_llm/model/info.rb +0 -164
  439. data/lib/ruby_llm/model_registry.rb +0 -39
  440. data/lib/ruby_llm/models_schema.json +0 -171
  441. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  442. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  443. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  444. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  445. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  446. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  447. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  448. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  449. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  450. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  451. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  452. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  453. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  454. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  455. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  456. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  457. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  458. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  459. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  460. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  461. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  462. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  463. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  464. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  465. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  466. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  467. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  468. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  469. data/lib/ruby_llm/streaming.rb +0 -179
  470. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  471. data/lib/ruby_llm/utils.rb +0 -130
  472. data/lib/tasks/models.rake +0 -593
  473. data/lib/tasks/release.rake +0 -94
  474. data/lib/tasks/vcr.rake +0 -124
@@ -0,0 +1,64 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class ElevenLabs
6
+ # Model catalog for ElevenLabs. GET v1/models returns a bare array of
7
+ # the speech synthesis models and leaves out the Scribe transcription
8
+ # models, so those are appended from the published catalog.
9
+ module Models
10
+ module_function
11
+
12
+ TRANSCRIPTION_MODELS = {
13
+ 'scribe_v2' => 'Scribe v2',
14
+ 'scribe_v2_realtime' => 'Scribe v2 Realtime'
15
+ }.freeze
16
+
17
+ def models_url
18
+ 'v1/models'
19
+ end
20
+
21
+ def parse_list_models_response(response, slug)
22
+ listed = Array(response.body).map do |data|
23
+ build_model(data['model_id'], slug, name: data['name'], description: data['description'], data: data)
24
+ end
25
+ listed_ids = listed.map(&:id)
26
+
27
+ listed + TRANSCRIPTION_MODELS.except(*listed_ids).map do |model_id, name|
28
+ build_model(model_id, slug, name: name, transcription: true)
29
+ end
30
+ end
31
+
32
+ def build_model(model_id, slug, name: nil, description: nil, data: {}, transcription: false)
33
+ transcription ||= TRANSCRIPTION_MODELS.key?(model_id)
34
+
35
+ Model.new(
36
+ id: model_id,
37
+ name: name || model_id,
38
+ provider: slug,
39
+ family: transcription ? 'scribe' : 'eleven',
40
+ modalities: modalities_for(data, transcription),
41
+ capabilities: capabilities_from(data, transcription),
42
+ pricing: {},
43
+ metadata: { description: description }.compact
44
+ )
45
+ end
46
+
47
+ def modalities_for(data, transcription)
48
+ return { input: ['audio'], output: ['text'] } if transcription
49
+ return { input: ['audio'], output: ['audio'] } if data['can_do_voice_conversion']
50
+
51
+ { input: ['text'], output: ['audio'] }
52
+ end
53
+
54
+ def capabilities_from(data, transcription)
55
+ capabilities = []
56
+ capabilities << 'transcription' if transcription
57
+ capabilities << 'speech_generation' if data['can_do_text_to_speech']
58
+ capabilities << 'fine_tuning' if data['can_be_finetuned']
59
+ capabilities
60
+ end
61
+ end
62
+ end
63
+ end
64
+ end
@@ -0,0 +1,67 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class ElevenLabs
6
+ # Speech dialect for the ElevenLabs text-to-speech API. The voice id
7
+ # is a path segment and the container is an output_format query
8
+ # value, so the endpoint cannot be named without the voice and format
9
+ # the caller asked for. The response is raw audio bytes.
10
+ module Speech
11
+ DEFAULT_VOICE = 'JBFqnCBsd6RMkjVDRZzb'
12
+ DEFAULT_OUTPUT_FORMAT = 'mp3_44100_128'
13
+
14
+ OUTPUT_FORMATS = {
15
+ 'alaw' => 'alaw_8000',
16
+ 'mp3' => 'mp3_44100_128',
17
+ 'opus' => 'opus_48000_128',
18
+ 'pcm' => 'pcm_44100',
19
+ 'ulaw' => 'ulaw_8000',
20
+ 'wav' => 'wav_44100'
21
+ }.freeze
22
+
23
+ def speak(input, model:, voice:, format:, provider_options: {}, &block)
24
+ track_usage(:speech) do
25
+ payload = render_speech_payload(input, model:, voice:, format:, provider_options:)
26
+ if block
27
+ next stream_speech_response(speech_url(voice:, format:, streaming: true), payload,
28
+ model:, voice:, format:, &block)
29
+ end
30
+
31
+ response = @connection.post speech_url(voice:, format:), payload, usage: @usage_tracker
32
+ parse_speech_response(response, model:, voice:, format:)
33
+ end
34
+ end
35
+
36
+ def speech_url(voice: nil, format: nil, streaming: false)
37
+ path = "v1/text-to-speech/#{voice || DEFAULT_VOICE}"
38
+ path += '/stream' if streaming
39
+ "#{path}?output_format=#{output_format_for(format)}"
40
+ end
41
+
42
+ def render_speech_payload(input, model:, voice: nil, format: nil, provider_options: {}) # rubocop:disable Lint/UnusedMethodArgument
43
+ Support::Utils.deep_merge({ text: input, model_id: model }, provider_options)
44
+ end
45
+
46
+ def parse_speech_response(response, model:, voice:, format:)
47
+ RubyLLM::Speech.new(
48
+ data: response.body,
49
+ model: model,
50
+ voice: voice || DEFAULT_VOICE,
51
+ format: container_for(format)
52
+ )
53
+ end
54
+
55
+ def output_format_for(format)
56
+ return DEFAULT_OUTPUT_FORMAT unless format
57
+
58
+ OUTPUT_FORMATS.fetch(format.to_s, format.to_s)
59
+ end
60
+
61
+ def container_for(format)
62
+ output_format_for(format).split('_').first
63
+ end
64
+ end
65
+ end
66
+ end
67
+ end
@@ -0,0 +1,127 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class ElevenLabs
6
+ module StreamingTranscription # :nodoc: all
7
+ SAMPLE_RATES = [8000, 16_000, 22_050, 24_000, 44_100, 48_000].freeze
8
+
9
+ def stream_transcription(payload, model:, &block)
10
+ validate_streaming_transcription(payload)
11
+ audio = RubyLLM::Transcription::WavAudio.new(payload.fetch(:file).io.read)
12
+ raise ArgumentError, 'Streaming transcription requires non-empty audio' if audio.data.empty?
13
+
14
+ audio_format = streaming_audio_format(audio)
15
+ url = streaming_transcription_url(payload, audio_format:)
16
+ @usage_tracker.start
17
+ expected_commits = (audio.data.bytesize.to_f / transcription_segment_bytes(audio)).ceil
18
+ transcripts = collect_transcription(url, audio, expected_commits, &block)
19
+ result = build_streaming_transcription(transcripts, audio, model:)
20
+ block.call(TranscriptionChunk.new(type: TranscriptionChunk::DONE, text: result.text, raw: transcripts.last))
21
+ result
22
+ ensure
23
+ payload[:file]&.io&.close
24
+ end
25
+
26
+ def collect_transcription(url, audio, expected_commits, &block)
27
+ transcripts = []
28
+ commits = Queue.new
29
+ Transport::WebsocketConnection.open(url, headers: @provider.headers, config: @config) do |socket|
30
+ write = ->(connection) { send_transcription_audio(connection, audio, commits) }
31
+ socket.each_message(write:) do |message|
32
+ event = JSON.parse(message)
33
+ transcript = process_transcription_event(event, prefix: transcripts.any?, &block)
34
+ next unless transcript
35
+
36
+ transcripts << transcript
37
+ commits << true
38
+ socket.close if transcripts.size == expected_commits
39
+ end
40
+ end
41
+ unless transcripts.size == expected_commits && transcripts.any?
42
+ raise Error, 'ElevenLabs transcription ended before its committed transcript'
43
+ end
44
+
45
+ transcripts
46
+ end
47
+
48
+ def build_streaming_transcription(transcripts, audio, model:)
49
+ RubyLLM::Transcription.new(
50
+ text: transcripts.map { |item| item.fetch('text') }.join(' '), model:,
51
+ duration: audio.duration, language: transcripts.last['language_code'],
52
+ words: transcripts.flat_map { |item| item['words'] || [] }
53
+ )
54
+ end
55
+
56
+ def validate_streaming_transcription(payload)
57
+ unsupported = payload.keys & %i[diarize num_speakers temperature timestamps_granularity]
58
+ unless unsupported.empty?
59
+ raise ArgumentError, "ElevenLabs streaming transcription does not accept #{unsupported.join(', ')}"
60
+ end
61
+
62
+ if payload.fetch(:commit_strategy, 'manual') != 'manual' || payload[:include_timestamps] == false ||
63
+ payload[:filter_background_audio]
64
+ raise ArgumentError, 'ElevenLabs file streaming requires manual commits and word timestamps'
65
+ end
66
+ end
67
+
68
+ def streaming_audio_format(audio)
69
+ return "pcm_#{audio.sample_rate}" if pcm_audio?(audio)
70
+ return 'ulaw_8000' if [audio.channels, audio.encoding, audio.sample_rate,
71
+ audio.bits_per_sample] == [1, 7, 8000, 8]
72
+
73
+ raise ArgumentError, 'ElevenLabs streaming requires mono 16-bit PCM WAV or 8 kHz mu-law WAV audio'
74
+ end
75
+
76
+ def pcm_audio?(audio)
77
+ [audio.channels, audio.encoding, audio.bits_per_sample] == [1, 1, 16] &&
78
+ SAMPLE_RATES.include?(audio.sample_rate)
79
+ end
80
+
81
+ def streaming_transcription_url(payload, audio_format:)
82
+ language_detection = payload.fetch(:include_language_detection, true)
83
+ params = payload.except(:file).merge(audio_format:, commit_strategy: 'manual', include_timestamps: true,
84
+ include_language_detection: language_detection)
85
+ uri = URI.join("#{@provider.api_base.sub(%r{/+\z}, '')}/",
86
+ "v1/speech-to-text/realtime?#{URI.encode_www_form(params)}")
87
+ uri.scheme = uri.scheme == 'https' ? 'wss' : 'ws'
88
+ uri.to_s
89
+ end
90
+
91
+ def transcription_segment_bytes(audio)
92
+ audio.sample_rate * audio.channels * audio.bits_per_sample / 8 * 20
93
+ end
94
+
95
+ def send_transcription_audio(socket, audio, commits)
96
+ offset = 0
97
+ segment_bytes = transcription_segment_bytes(audio)
98
+ while offset < audio.data.bytesize
99
+ length = [16_384, segment_bytes - (offset % segment_bytes)].min
100
+ data = audio.data.byteslice(offset, length)
101
+ offset += data.bytesize
102
+ commit = (offset % segment_bytes).zero? || offset == audio.data.bytesize
103
+ socket.send_text(JSON.generate(message_type: 'input_audio_chunk',
104
+ 'audio_base_64' => Base64.strict_encode64(data),
105
+ sample_rate: audio.sample_rate, commit:))
106
+ commits.pop if commit
107
+ end
108
+ end
109
+
110
+ def process_transcription_event(event, prefix: false)
111
+ case event['message_type']
112
+ when 'partial_transcript'
113
+ yield TranscriptionChunk.new(type: TranscriptionChunk::PARTIAL, text: event['text'], raw: event)
114
+ when 'committed_transcript'
115
+ text = event['text']
116
+ yield TranscriptionChunk.new(type: TranscriptionChunk::DELTA, delta: prefix ? " #{text}" : text, raw: event)
117
+ when 'committed_transcript_with_timestamps'
118
+ return event
119
+ else
120
+ raise Error, event['error'] if event['error']
121
+ end
122
+ nil
123
+ end
124
+ end
125
+ end
126
+ end
127
+ end
@@ -0,0 +1,61 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class ElevenLabs
6
+ # Transcription dialect for the ElevenLabs speech-to-text API. The
7
+ # request is multipart with the model id in the form, and giving
8
+ # speaker names turns on diarization and caps the speaker count at
9
+ # the number of names.
10
+ module Transcription
11
+ def render_transcription_options(timestamps:, format:, streaming:)
12
+ return {} if timestamps.nil?
13
+
14
+ value = timestamps.to_s
15
+ allowed = streaming ? ['word'] : %w[none word character]
16
+ unless allowed.include?(value) && (format.nil? || format == value)
17
+ raise ArgumentError, "ElevenLabs timestamps must be #{allowed.join(', ')} and match format when provided"
18
+ end
19
+
20
+ streaming ? { include_timestamps: true } : { timestamps_granularity: value }
21
+ end
22
+
23
+ # rubocop:disable-next Lint/UnusedMethodArgument
24
+ def render_transcription_payload(file_part, model:, language:, format: nil, speaker_names: nil,
25
+ speaker_references: nil, provider_options: {}, prompt: nil,
26
+ temperature: nil)
27
+ payload = {
28
+ model_id: model,
29
+ file: file_part,
30
+ language_code: language,
31
+ temperature: temperature,
32
+ timestamps_granularity: format
33
+ }.compact
34
+
35
+ if speaker_names
36
+ payload[:diarize] = true
37
+ payload[:num_speakers] = speaker_names.size
38
+ end
39
+
40
+ payload.merge(provider_options)
41
+ end
42
+
43
+ def transcription_url
44
+ 'v1/speech-to-text'
45
+ end
46
+
47
+ def parse_transcription_response(response, model:)
48
+ data = response.body
49
+
50
+ RubyLLM::Transcription.new(
51
+ text: data['text'],
52
+ model: model,
53
+ language: data['language_code'],
54
+ duration: data['audio_duration_secs'],
55
+ words: data['words']
56
+ )
57
+ end
58
+ end
59
+ end
60
+ end
61
+ end
@@ -0,0 +1,15 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ # The ElevenLabs audio API: text to speech, speech to text, and the model
6
+ # catalog behind them. Image and video generation and workspace assets use
7
+ # separate protocols.
8
+ class ElevenLabs < Protocol
9
+ include ElevenLabs::Models
10
+ include ElevenLabs::Speech
11
+ include ElevenLabs::Transcription
12
+ include ElevenLabs::StreamingTranscription
13
+ end
14
+ end
15
+ end
@@ -0,0 +1,119 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'stringio'
4
+
5
+ module RubyLLM
6
+ module Protocols
7
+ # Provider-managed file storage APIs.
8
+ class Files < Protocol
9
+ # rubocop:disable-next Lint/UnusedMethodArgument
10
+ def upload(file, filename: nil, purpose: nil, expires_in: nil, uri: nil, content_type: nil, provider_options: {})
11
+ attachment = file_attachment(file, filename:)
12
+ options = { purpose:, expires_in: }.compact.merge(provider_options.transform_keys(&:to_sym))
13
+ response = @connection.post(files_url, render_upload_payload(attachment, **options),
14
+ idempotent: false) do |request|
15
+ request.headers.delete('Content-Type')
16
+ upload_headers(request)
17
+ end
18
+ parse_file_response(response.body)
19
+ end
20
+
21
+ def find(file_id)
22
+ response = @connection.get(file_info_url(file_id)) { |request| file_headers(request) }
23
+ parse_file_response(response.body)
24
+ end
25
+
26
+ def download(file_id)
27
+ response = @connection.get(download_file_url(file_id)) do |request|
28
+ request.headers['Accept'] = 'application/octet-stream'
29
+ file_headers(request)
30
+ end
31
+ response.body
32
+ end
33
+
34
+ def list_uris(_uri)
35
+ raise Error, "#{@provider.slug} doesn't support file listing"
36
+ end
37
+
38
+ private
39
+
40
+ def files_url
41
+ 'files'
42
+ end
43
+
44
+ def file_info_url(file_id)
45
+ "#{files_url}/#{file_id}"
46
+ end
47
+
48
+ def download_file_url(file_id)
49
+ "#{file_info_url(file_id)}/content"
50
+ end
51
+
52
+ # rubocop:disable-next Lint/UnusedMethodArgument
53
+ def render_upload_payload(attachment, purpose: nil, expires_in: nil, visibility: nil,
54
+ display_name: nil, uri: nil, content_type: nil)
55
+ { file: file_part(attachment) }
56
+ end
57
+
58
+ def multipart_payload(attachment, **fields)
59
+ { file: file_part(attachment) }.merge(fields.compact)
60
+ end
61
+
62
+ def upload_headers(_request); end
63
+
64
+ def file_headers(_request); end
65
+
66
+ def file_attachment(file, filename: nil)
67
+ return file if file.is_a?(Attachment) && filename.nil?
68
+
69
+ file = file.source if file.is_a?(Attachment)
70
+ Attachment.new(file, filename:, config:)
71
+ end
72
+
73
+ def file_part(attachment, content_type: nil)
74
+ Faraday::Multipart::FilePart.new(file_part_source(attachment), content_type || file_content_type(attachment),
75
+ attachment.filename)
76
+ end
77
+
78
+ def file_content_type(attachment)
79
+ attachment.extension == 'jsonl' ? 'application/jsonl' : attachment.mime_type
80
+ end
81
+
82
+ def file_part_source(attachment)
83
+ if attachment.path?
84
+ attachment.source.to_s
85
+ elsif attachment.io_like?
86
+ attachment.source.tap { |io| io.rewind if io.respond_to?(:rewind) }
87
+ else
88
+ StringIO.new(attachment.content)
89
+ end
90
+ end
91
+
92
+ def timestamp(value)
93
+ return if value.nil?
94
+ return Time.at(value) if value.is_a?(Numeric)
95
+ return Time.at(value.to_i) if value.to_s.match?(/\A\d+\z/)
96
+
97
+ Time.iso8601(value.to_s)
98
+ end
99
+
100
+ def uploaded_file(data, **attributes)
101
+ UploadedFile.new(**attributes, provider: @provider.slug, metadata: data)
102
+ end
103
+
104
+ def with_file_body(attachment, &)
105
+ if attachment.path?
106
+ File.open(attachment.source, 'rb', &)
107
+ else
108
+ body = attachment.io_like? ? attachment.source : StringIO.new(attachment.content)
109
+ body.rewind if body.respond_to?(:rewind)
110
+ yield body
111
+ end
112
+ end
113
+
114
+ def file_size(attachment)
115
+ attachment.path? ? File.size(attachment.source) : attachment.content.bytesize
116
+ end
117
+ end
118
+ end
119
+ end
@@ -0,0 +1,162 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Gemini Batch API with inlined generateContent requests.
7
+ module Batches
8
+ include RubyLLM::Batch::Helpers
9
+ include Gemini::EmbeddingBatches
10
+
11
+ # The wire enum is BATCH_STATE_*; the SDKs print JOB_STATE_*. Match the
12
+ # suffix so either spelling works.
13
+ TERMINAL = %w[SUCCEEDED FAILED CANCELLED EXPIRED].freeze
14
+ private_constant :TERMINAL
15
+
16
+ def create_batch(requests)
17
+ model = single_batch_model!(requests, 'gemini')
18
+ action = embedding_batch?(requests) ? 'asyncBatchEmbedContent' : 'batchGenerateContent'
19
+ response = @connection.post("models/#{model}:#{action}", {
20
+ batch: {
21
+ displayName: "ruby_llm_#{SecureRandom.hex(8)}",
22
+ inputConfig: {
23
+ requests: {
24
+ requests: requests.flat_map do |request|
25
+ if embedding_batch_payload?(request.fetch(:payload))
26
+ embedding_batch_requests(request, model)
27
+ else
28
+ [gemini_batch_request(request, model)]
29
+ end
30
+ end
31
+ }
32
+ }
33
+ }
34
+ }, idempotent: false)
35
+
36
+ parse_batch_response(response.body)
37
+ end
38
+
39
+ def find_batch(id)
40
+ parse_batch_response @connection.get(batch_name(id)).body
41
+ end
42
+
43
+ def cancel_batch(id)
44
+ @connection.post("#{batch_name(id)}:cancel", {})
45
+ find_batch(id)
46
+ end
47
+
48
+ # Inline answers are correlated by the key we sent (the submission index);
49
+ # Gemini also returns them in order, so we fall back to position.
50
+ def batch_results(id)
51
+ body = @connection.get(batch_name(id)).body
52
+ inlined = inline_batch_responses(body)
53
+ return parse_embedding_batch_results(inlined) if embedding_batch_response?(body)
54
+
55
+ inlined.each_with_index.map { |response, index| parse_inline_response(response, index) }
56
+ end
57
+
58
+ private
59
+
60
+ def inline_batch_responses(body)
61
+ body.dig('response', 'inlinedEmbedContentResponses', 'inlinedResponses') ||
62
+ body.dig('output', 'inlinedEmbedContentResponses', 'inlinedResponses') ||
63
+ body.dig('response', 'inlinedResponses', 'inlinedResponses') ||
64
+ body.dig('output', 'inlinedResponses', 'inlinedResponses') ||
65
+ body.dig('metadata', 'output', 'inlinedResponses', 'inlinedResponses') || []
66
+ end
67
+
68
+ def gemini_batch_request(request, model)
69
+ {
70
+ request: batch_schema_payload(batch_payload(request)).merge(model: "models/#{model}"),
71
+ metadata: {
72
+ custom_id: request[:custom_id]
73
+ }
74
+ }
75
+ end
76
+
77
+ # batchGenerateContent ignores responseJsonSchema and returns JSON of
78
+ # its own shape, while the legacy responseSchema field is honored, so
79
+ # batches carry the schema in Gemini's Schema dialect.
80
+ def batch_schema_payload(payload)
81
+ config = payload[:generationConfig]
82
+ return payload unless config.is_a?(Hash) && config.key?(:responseJsonSchema)
83
+
84
+ schema = response_schema(config[:responseJsonSchema])
85
+ payload.merge(generationConfig: config.except(:responseJsonSchema).merge(responseSchema: schema))
86
+ end
87
+
88
+ JSON_SCHEMA_ONLY_KEYS = %w[$schema $id additionalProperties strict].freeze
89
+ private_constant :JSON_SCHEMA_ONLY_KEYS
90
+
91
+ def response_schema(node)
92
+ case node
93
+ when Hash then response_schema_hash(node)
94
+ when Array then node.map { |value| response_schema(value) }
95
+ else node
96
+ end
97
+ end
98
+
99
+ def response_schema_hash(node)
100
+ schema = node.each_with_object({}) do |(key, value), converted|
101
+ next if JSON_SCHEMA_ONLY_KEYS.include?(key.to_s)
102
+
103
+ converted[key] = if key.to_s == 'properties' && value.is_a?(Hash)
104
+ value.transform_values { |property| response_schema(property) }
105
+ else
106
+ response_schema(value)
107
+ end
108
+ end
109
+ nullable_type(schema)
110
+ end
111
+
112
+ def nullable_type(schema)
113
+ key = schema.key?(:type) ? :type : 'type'
114
+ types = Array(schema[key])
115
+ return schema unless types.length > 1 && types.include?('null')
116
+
117
+ schema.merge(key => (types - ['null']).first, nullable: true)
118
+ end
119
+
120
+ def batch_name(id)
121
+ id.to_s.start_with?('batches/') ? id : "batches/#{id}"
122
+ end
123
+
124
+ # A batch starts as an Operation wrapping the batch in `metadata`; polling
125
+ # returns the batch directly. Read either shape.
126
+ def parse_batch_response(data)
127
+ batch = data['metadata'] || data
128
+ state = batch['state']
129
+ request_counts = batch['batchStats']
130
+
131
+ {
132
+ id: data['name'] || batch['name'],
133
+ raw_status: state,
134
+ completed: TERMINAL.any? { |terminal| state&.end_with?(terminal) },
135
+ request_counts:,
136
+ request_count: (request_counts&.fetch('requestCount', nil)&.to_i unless embedding_batch_response?(data))
137
+ }
138
+ end
139
+
140
+ def parse_batch_status(raw_status, completed:)
141
+ return :pending unless completed
142
+ return :succeeded if raw_status&.end_with?('SUCCEEDED')
143
+ return :cancelled if raw_status&.end_with?('CANCELLED')
144
+
145
+ :failed
146
+ end
147
+
148
+ def parse_inline_response(inline, index)
149
+ key = inline.dig('metadata', 'custom_id') || inline.dig('metadata', 'key')
150
+ index = batch_result_index(key) if key
151
+
152
+ if inline['response']
153
+ body = inline['response']
154
+ [index, parse_completion_body(body, raw: body)]
155
+ else
156
+ [index, nil, batch_failure(key || index, inline.dig('error', 'message'))]
157
+ end
158
+ end
159
+ end
160
+ end
161
+ end
162
+ end
@@ -0,0 +1,59 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Explicit content caching via the Gemini cachedContents API.
7
+ module Caches
8
+ module_function
9
+
10
+ def caches_url
11
+ 'cachedContents'
12
+ end
13
+
14
+ def cache_url(name)
15
+ cache_name(name)
16
+ end
17
+
18
+ def render_cache_payload(content, model:, ttl: nil, instructions: nil, attachments: [])
19
+ payload = {
20
+ model: cache_model_name(model),
21
+ contents: [{ role: 'user', parts: Media.format_content(content, attachments) }]
22
+ }
23
+ payload[:systemInstruction] = { parts: [{ text: instructions }] } if instructions
24
+ payload[:ttl] = format_cache_ttl(ttl) if ttl
25
+ payload
26
+ end
27
+
28
+ def render_cache_update_payload(ttl:)
29
+ { ttl: format_cache_ttl(ttl) }
30
+ end
31
+
32
+ def parse_cache_response(data)
33
+ CachedContent.new(
34
+ name: data['name'],
35
+ model: data['model']&.split('/')&.last,
36
+ provider_instance: @provider,
37
+ created_at: (Time.iso8601(data['createTime']) if data['createTime']),
38
+ expires_at: (Time.iso8601(data['expireTime']) if data['expireTime']),
39
+ tokens: data.dig('usageMetadata', 'totalTokenCount'),
40
+ metadata: data
41
+ )
42
+ end
43
+
44
+ def cache_name(name)
45
+ name = name.name if name.is_a?(CachedContent)
46
+ name.to_s.include?('/') ? name.to_s : "cachedContents/#{name}"
47
+ end
48
+
49
+ def cache_model_name(model_id)
50
+ "models/#{model_id}"
51
+ end
52
+
53
+ def format_cache_ttl(ttl)
54
+ ttl.is_a?(String) ? ttl : "#{ttl.to_i}s"
55
+ end
56
+ end
57
+ end
58
+ end
59
+ end