ruby_llm 1.16.0 → 2.0.0.rc2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (475) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +85 -32
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
  91. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
  92. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
  93. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
  94. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  96. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
  97. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
  98. data/lib/ruby_llm/accounting/usage.rb +245 -0
  99. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  100. data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
  101. data/lib/ruby_llm/active_record/batch.rb +97 -0
  102. data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
  103. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  104. data/lib/ruby_llm/active_record/model.rb +135 -0
  105. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  106. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  107. data/lib/ruby_llm/active_record/usage.rb +61 -0
  108. data/lib/ruby_llm/agent.rb +1065 -151
  109. data/lib/ruby_llm/aliases.json +268 -100
  110. data/lib/ruby_llm/attachment.rb +187 -48
  111. data/lib/ruby_llm/batch.rb +432 -0
  112. data/lib/ruby_llm/cached_content.rb +112 -0
  113. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  114. data/lib/ruby_llm/chat.rb +1127 -198
  115. data/lib/ruby_llm/chunk.rb +10 -0
  116. data/lib/ruby_llm/citation.rb +105 -0
  117. data/lib/ruby_llm/configuration.rb +261 -24
  118. data/lib/ruby_llm/context.rb +128 -6
  119. data/lib/ruby_llm/cost.rb +217 -80
  120. data/lib/ruby_llm/downloaded_file.rb +33 -0
  121. data/lib/ruby_llm/embedding.rb +121 -17
  122. data/lib/ruby_llm/embedding_request.rb +53 -0
  123. data/lib/ruby_llm/error.rb +159 -23
  124. data/lib/ruby_llm/fallback.rb +133 -0
  125. data/lib/ruby_llm/files/mime_type.rb +97 -0
  126. data/lib/ruby_llm/image.rb +153 -32
  127. data/lib/ruby_llm/message.rb +233 -54
  128. data/lib/ruby_llm/model/modalities.rb +17 -4
  129. data/lib/ruby_llm/model/pricing.rb +24 -5
  130. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  131. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  132. data/lib/ruby_llm/model.rb +244 -2
  133. data/lib/ruby_llm/models/aliases.rb +41 -0
  134. data/lib/ruby_llm/models/registry.rb +165 -0
  135. data/lib/ruby_llm/models/schema.rb +99 -0
  136. data/lib/ruby_llm/models.json +73261 -33733
  137. data/lib/ruby_llm/models.rb +477 -215
  138. data/lib/ruby_llm/moderation.rb +139 -26
  139. data/lib/ruby_llm/ocr.rb +112 -0
  140. data/lib/ruby_llm/prompt.rb +79 -0
  141. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  142. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  143. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  144. data/lib/ruby_llm/protocol.rb +662 -0
  145. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  146. data/lib/ruby_llm/protocols/anthropic/chat.rb +548 -0
  147. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  148. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  149. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  150. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  151. data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
  152. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  153. data/lib/ruby_llm/protocols/anthropic.rb +101 -0
  154. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  155. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
  156. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  157. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  158. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  159. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  160. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  161. data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
  162. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  163. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  164. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  165. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  166. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  167. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  168. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
  169. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  170. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  171. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  172. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  173. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  174. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  175. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  176. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  177. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  178. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  179. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  180. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  181. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  182. data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
  183. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  184. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  185. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  186. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  187. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  188. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  189. data/lib/ruby_llm/protocols/converse/chat.rb +695 -0
  190. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  191. data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
  192. data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
  193. data/lib/ruby_llm/protocols/converse.rb +54 -0
  194. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  195. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  196. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  197. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  198. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  199. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  200. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  201. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  202. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  203. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  204. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  208. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  209. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  210. data/lib/ruby_llm/protocols/files.rb +119 -0
  211. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  212. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  213. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  214. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
  215. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  216. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  217. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  218. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  219. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  220. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  221. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  222. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  223. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  224. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  225. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  226. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  227. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  228. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  229. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  230. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  231. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  232. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  233. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  234. data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
  235. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  236. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  237. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  238. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  239. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  240. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  241. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  242. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  243. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  244. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  245. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  246. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  247. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  248. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  249. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  250. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  251. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  252. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  253. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  254. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  255. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  256. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  257. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  258. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  259. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  260. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  261. data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
  262. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  263. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  264. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  265. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  266. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  267. data/lib/ruby_llm/protocols/responses.rb +35 -0
  268. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  269. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  270. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  271. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  272. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  273. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  274. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  275. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  276. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  277. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  278. data/lib/ruby_llm/provider.rb +560 -128
  279. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  280. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  281. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  282. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  283. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  284. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  285. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  286. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  287. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  288. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  289. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  290. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  291. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  292. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  293. data/lib/ruby_llm/providers/azure.rb +77 -78
  294. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  295. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  296. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  297. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  298. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  299. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  300. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  301. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  302. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  303. data/lib/ruby_llm/providers/cohere.rb +31 -0
  304. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  305. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  306. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  307. data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
  308. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  309. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  310. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  311. data/lib/ruby_llm/providers/gemini.rb +15 -8
  312. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  313. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  314. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  315. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  316. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  317. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  318. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  319. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  320. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  321. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  322. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  323. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  324. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  325. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  326. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  327. data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
  328. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  329. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  330. data/lib/ruby_llm/providers/mistral.rb +16 -4
  331. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  332. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  333. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  334. data/lib/ruby_llm/providers/ollama.rb +9 -8
  335. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  336. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  337. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  338. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  339. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  340. data/lib/ruby_llm/providers/openai.rb +92 -11
  341. data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
  342. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  343. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  344. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  345. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  346. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  347. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  348. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  349. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  350. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  351. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  352. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  353. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  354. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  355. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  356. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  357. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  358. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  359. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  360. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  361. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  362. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  363. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  364. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  365. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  366. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  367. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  368. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  369. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  370. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  371. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  372. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  373. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  374. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  375. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  376. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  377. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  378. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  379. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  380. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  381. data/lib/ruby_llm/providers/xai.rb +15 -5
  382. data/lib/ruby_llm/railtie.rb +7 -16
  383. data/lib/ruby_llm/rerank.rb +105 -0
  384. data/lib/ruby_llm/research_job.rb +241 -0
  385. data/lib/ruby_llm/search_results.rb +68 -0
  386. data/lib/ruby_llm/server_tool_call.rb +73 -0
  387. data/lib/ruby_llm/speech.rb +159 -0
  388. data/lib/ruby_llm/speech_chunk.rb +33 -0
  389. data/lib/ruby_llm/support/deprecator.rb +22 -0
  390. data/lib/ruby_llm/support/inspectable.rb +49 -0
  391. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  392. data/lib/ruby_llm/support/utils.rb +147 -0
  393. data/lib/ruby_llm/thinking.rb +127 -20
  394. data/lib/ruby_llm/tokenization.rb +59 -0
  395. data/lib/ruby_llm/tokens.rb +103 -33
  396. data/lib/ruby_llm/tool.rb +266 -91
  397. data/lib/ruby_llm/tool_call.rb +36 -3
  398. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  399. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  400. data/lib/ruby_llm/transcription.rb +138 -13
  401. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  402. data/lib/ruby_llm/transport/connection.rb +193 -0
  403. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  404. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  405. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  406. data/lib/ruby_llm/uploaded_file.rb +144 -0
  407. data/lib/ruby_llm/version.rb +2 -1
  408. data/lib/ruby_llm/video.rb +136 -0
  409. data/lib/ruby_llm/video_job.rb +150 -0
  410. data/lib/ruby_llm/workflow.rb +91 -0
  411. data/lib/ruby_llm.rb +380 -6
  412. data/lib/tasks/ruby_llm.rake +21 -16
  413. data/skills/rubyllm/SKILL.md +81 -0
  414. data/skills/rubyllm/agents/openai.yaml +4 -0
  415. metadata +339 -97
  416. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  417. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  418. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  419. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  420. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  421. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  422. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  423. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  424. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  425. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  426. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  427. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  428. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  429. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  430. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  431. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  432. data/lib/ruby_llm/aliases.rb +0 -41
  433. data/lib/ruby_llm/connection.rb +0 -159
  434. data/lib/ruby_llm/content.rb +0 -91
  435. data/lib/ruby_llm/deprecator.rb +0 -24
  436. data/lib/ruby_llm/error_middleware.rb +0 -81
  437. data/lib/ruby_llm/instrumentation.rb +0 -36
  438. data/lib/ruby_llm/mime_type.rb +0 -96
  439. data/lib/ruby_llm/model/info.rb +0 -164
  440. data/lib/ruby_llm/model_registry.rb +0 -39
  441. data/lib/ruby_llm/models_schema.json +0 -171
  442. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  443. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  444. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  445. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  446. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  447. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  448. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  449. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  450. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  451. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  452. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  453. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  454. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  455. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  456. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  457. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  458. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  459. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  460. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  461. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  462. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  463. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  464. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  465. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  466. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  467. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  468. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  469. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  470. data/lib/ruby_llm/streaming.rb +0 -179
  471. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  472. data/lib/ruby_llm/utils.rb +0 -130
  473. data/lib/tasks/models.rake +0 -593
  474. data/lib/tasks/release.rake +0 -94
  475. data/lib/tasks/vcr.rake +0 -124
@@ -0,0 +1,86 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Inline embedding requests and results for Gemini's asynchronous Batch API.
7
+ module EmbeddingBatches
8
+ private
9
+
10
+ def embedding_batch?(requests)
11
+ kinds = requests.map { |request| embedding_batch_payload?(request.fetch(:payload)) }.uniq
12
+ raise Error, 'Gemini batches take chat or embedding requests, not both' if kinds.size > 1
13
+
14
+ kinds.first
15
+ end
16
+
17
+ def embedding_batch_payload?(payload)
18
+ payload.key?(:content) || payload.key?(:requests)
19
+ end
20
+
21
+ def embedding_batch_response?(data)
22
+ batch = data['metadata'] || data
23
+ batch['@type']&.end_with?('.EmbedContentBatch') ||
24
+ data.dig('response', '@type')&.end_with?('.EmbedContentBatchOutput') ||
25
+ data.dig('response', 'inlinedEmbedContentResponses') ||
26
+ data.dig('output', 'inlinedEmbedContentResponses')
27
+ end
28
+
29
+ def embedding_batch_requests(request, model)
30
+ payload = request.fetch(:payload)
31
+ array_input = payload.key?(:requests)
32
+ inputs = array_input ? payload.fetch(:requests) : [payload]
33
+ raise ArgumentError, 'Gemini embedding batches require at least one text per request' if inputs.empty?
34
+
35
+ inputs.each_with_index.map do |input, index|
36
+ {
37
+ request: input.merge(model: "models/#{model}"),
38
+ metadata: {
39
+ custom_id: request.fetch(:custom_id), model:, array_input:,
40
+ embedding_index: index, embedding_count: inputs.size
41
+ }
42
+ }
43
+ end
44
+ end
45
+
46
+ def parse_embedding_batch_results(responses)
47
+ groups = responses.each_with_index.group_by do |inline, index|
48
+ inline.dig('metadata', 'custom_id') || index.to_s
49
+ end
50
+ groups.filter_map do |key, indexed|
51
+ parse_embedding_batch_group(key, indexed.map(&:first))
52
+ end
53
+ end
54
+
55
+ def parse_embedding_batch_group(key, responses)
56
+ index = batch_result_index(key)
57
+ error = responses.find { |inline| inline['error'] }
58
+ return [index, nil, batch_failure(key, error.dig('error', 'message'))] if error
59
+
60
+ metadata = responses.first.fetch('metadata')
61
+ return if responses.size < metadata.fetch('embedding_count')
62
+
63
+ vectors = embedding_batch_vectors(responses)
64
+ return [index, nil, batch_failure(key, 'Gemini returned no embedding')] unless vectors
65
+
66
+ embedding = Embedding.new(
67
+ vectors: metadata['array_input'] ? vectors : vectors.first,
68
+ model: metadata.fetch('model'), input_tokens: embedding_batch_tokens(responses)
69
+ )
70
+ [index, embedding]
71
+ end
72
+
73
+ def embedding_batch_vectors(responses)
74
+ ordered = responses.sort_by { |inline| inline.dig('metadata', 'embedding_index') }
75
+ vectors = ordered.map { |inline| inline.dig('response', 'embedding', 'values') }
76
+ vectors if vectors.all? { |vector| vector && !vector.empty? }
77
+ end
78
+
79
+ def embedding_batch_tokens(responses)
80
+ counts = responses.filter_map { |inline| inline.dig('response', 'usageMetadata', 'promptTokenCount') }
81
+ counts.sum unless counts.empty?
82
+ end
83
+ end
84
+ end
85
+ end
86
+ end
@@ -0,0 +1,70 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Embeddings methods for the Gemini API integration
7
+ module Embeddings
8
+ module_function
9
+
10
+ def embedding_url(model:)
11
+ "models/#{model}:batchEmbedContents"
12
+ end
13
+
14
+ def render_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil, with: [],
15
+ provider_options: {})
16
+ requests = if with.any?
17
+ raise ArgumentError, 'embed one text at a time when embedding attachments' if text.is_a?(Array)
18
+
19
+ [media_embedding_payload(text, with, model:, dimensions:, task_type:, title:)]
20
+ else
21
+ [text].flatten.map do |t|
22
+ single_embedding_payload(t, model:, dimensions:, task_type:, title:)
23
+ end
24
+ end
25
+
26
+ Support::Utils.deep_merge({ requests: requests }, provider_options)
27
+ end
28
+
29
+ def supports_embedding_media?
30
+ true
31
+ end
32
+
33
+ def render_embedding(text, dimensions: nil, **options)
34
+ payload = render_embedding_payload(text, dimensions:, **options)
35
+ return payload if text.is_a?(Array) || !payload.key?(:requests)
36
+
37
+ payload.fetch(:requests).first
38
+ end
39
+ public :render_embedding
40
+
41
+ def parse_embedding_response(response, model:, text:)
42
+ vectors = response.body['embeddings']&.map { |e| e['values'] }
43
+ vectors = vectors.first if vectors&.length == 1 && !text.is_a?(Array)
44
+
45
+ Embedding.new(vectors:, model:)
46
+ end
47
+
48
+ private
49
+
50
+ def single_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil)
51
+ embedding_payload([{ text: text.to_s }], model:, dimensions:, task_type:, title:)
52
+ end
53
+
54
+ def media_embedding_payload(text, attachments, model:, dimensions:, task_type: nil, title: nil)
55
+ embedding_payload(Media.format_content(text, attachments), model:, dimensions:, task_type:, title:)
56
+ end
57
+
58
+ def embedding_payload(parts, model:, dimensions:, task_type:, title:)
59
+ {
60
+ model: "models/#{model}",
61
+ content: { parts: parts },
62
+ outputDimensionality: dimensions,
63
+ taskType: task_type,
64
+ title: title
65
+ }.compact
66
+ end
67
+ end
68
+ end
69
+ end
70
+ end
@@ -0,0 +1,32 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ module FileTranscription # :nodoc: all
7
+ def transcribe(audio_file, model:, language:, format: nil, speaker_names: nil,
8
+ speaker_references: nil, provider_options: {}, prompt: nil, temperature: nil)
9
+ raise_transcription_streaming_unsupported if block_given?
10
+ validate_transcription_request(format:, speaker_references:, temperature:)
11
+ attachments = Attachment.wrap(audio_file, config: @config)
12
+ unless attachments.one? && attachments.first.audio?
13
+ raise ArgumentError, 'Dedicated transcription requires exactly one audio file'
14
+ end
15
+
16
+ track_usage(:transcription) do
17
+ payload = render_transcription_payload(attachments.first, model:, language:, speaker_names:,
18
+ provider_options:, prompt:)
19
+ response = @connection.post(transcription_url(model), payload, usage: @usage_tracker)
20
+ parse_transcription_response(response, model:)
21
+ end
22
+ end
23
+
24
+ def validate_transcription_request(format:, speaker_references:, temperature:)
25
+ return unless format || speaker_references || temperature
26
+
27
+ raise ArgumentError, 'Dedicated transcription does not accept format, speaker references, or temperature'
28
+ end
29
+ end
30
+ end
31
+ end
32
+ end
@@ -0,0 +1,115 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Gemini Files API.
7
+ class Files < Protocols::Files
8
+ PROCESSING_POLL_INTERVAL = 2
9
+ PROCESSING_TIMEOUT = 600
10
+
11
+ # rubocop:disable-next Lint/UnusedMethodArgument
12
+ def upload(file, filename: nil, purpose: nil, expires_in: nil, uri: nil, content_type: nil,
13
+ provider_options: {})
14
+ attachment = file_attachment(file, filename:)
15
+ display_name = provider_options[:display_name] || attachment.filename
16
+ upload_url = start_resumable_upload(attachment, display_name:)
17
+ response = upload_file_bytes(upload_url, attachment)
18
+ await_active(parse_file_response(response.body.fetch('file')))
19
+ end
20
+
21
+ def download(file_id)
22
+ file = find(file_id)
23
+ download_uri = file.metadata['downloadUri']
24
+ raise Error, 'gemini file has no download URI' unless download_uri
25
+
26
+ gemini_connection(request_json: false, response_json: false).get(download_uri) do |request|
27
+ request.headers.merge!(@provider.headers)
28
+ end.body
29
+ end
30
+
31
+ private
32
+
33
+ # Gemini rejects a file reference until processing finishes, which
34
+ # for video takes far longer than the upload itself.
35
+ def await_active(file)
36
+ deadline = Time.now + PROCESSING_TIMEOUT
37
+
38
+ while file.status == 'PROCESSING'
39
+ raise Error, "gemini is still processing #{file.id}" if Time.now >= deadline
40
+
41
+ sleep PROCESSING_POLL_INTERVAL
42
+ file = find(file.id)
43
+ end
44
+
45
+ raise Error, "gemini failed to process #{file.id}" if file.status == 'FAILED'
46
+
47
+ file
48
+ end
49
+
50
+ def file_info_url(file_id)
51
+ gemini_file_name(file_id)
52
+ end
53
+
54
+ def start_resumable_upload(attachment, display_name:)
55
+ response = gemini_connection(request_json: true, response_json: true).post(gemini_upload_url) do |request|
56
+ request.headers.merge!(@provider.headers)
57
+ request.headers['X-Goog-Upload-Protocol'] = 'resumable'
58
+ request.headers['X-Goog-Upload-Command'] = 'start'
59
+ request.headers['X-Goog-Upload-Header-Content-Length'] = attachment.content.bytesize.to_s
60
+ request.headers['X-Goog-Upload-Header-Content-Type'] = file_content_type(attachment)
61
+ request.headers['Content-Type'] = 'application/json'
62
+ request.body = { file: { display_name: display_name } }
63
+ end
64
+
65
+ response.headers['x-goog-upload-url'] || response.headers['X-Goog-Upload-URL'] ||
66
+ raise(Error, 'gemini did not return an upload URL')
67
+ end
68
+
69
+ def upload_file_bytes(upload_url, attachment)
70
+ gemini_connection(request_json: false, response_json: true).post(upload_url) do |request|
71
+ request.headers.merge!(@provider.headers)
72
+ request.headers['Content-Length'] = attachment.content.bytesize.to_s
73
+ request.headers['X-Goog-Upload-Offset'] = '0'
74
+ request.headers['X-Goog-Upload-Command'] = 'upload, finalize'
75
+ request.body = attachment.content
76
+ end
77
+ end
78
+
79
+ def parse_file_response(data)
80
+ uploaded_file(
81
+ data,
82
+ id: data['name'],
83
+ filename: data['displayName'],
84
+ byte_size: data['sizeBytes']&.to_i,
85
+ created_at: timestamp(data['createTime']),
86
+ expires_at: timestamp(data['expirationTime']),
87
+ mime_type: data['mimeType'],
88
+ status: data['state'],
89
+ uri: data['uri']
90
+ )
91
+ end
92
+
93
+ def gemini_file_name(file_id)
94
+ file_id.to_s.start_with?('files/') ? file_id : "files/#{file_id}"
95
+ end
96
+
97
+ def gemini_upload_url
98
+ base = @provider.api_base.to_s.sub(%r{/+\z}, '')
99
+ "#{base.sub(%r{/v1beta\z}, '/upload/v1beta').sub(%r{/v1\z}, '/upload/v1')}/files"
100
+ end
101
+
102
+ def gemini_connection(request_json:, response_json:)
103
+ connection = Transport::Connection.basic(@config) do |connection|
104
+ connection.request :json if request_json
105
+ connection.response :json if response_json
106
+ connection.adapter @config.faraday_adapter
107
+ connection.use :llm_errors, provider: @provider
108
+ end
109
+ connection.url_prefix = @provider.api_base
110
+ connection
111
+ end
112
+ end
113
+ end
114
+ end
115
+ end
@@ -0,0 +1,183 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Image generation methods for the Gemini API implementation
7
+ module Images
8
+ IMAGE_SIZES = %w[512 512P 512PX 1K 2K 4K].freeze
9
+ ASPECT_RATIO = /\A\d+:\d+\z/
10
+ PIXEL_DIMENSIONS = /\A(\d+)\s*[x×]\s*(\d+)\z/i
11
+
12
+ def images_url(with: nil, mask: nil) # rubocop:disable Lint/UnusedMethodArgument
13
+ id = model_id(@model)
14
+
15
+ "models/#{id}:#{image_endpoint_action(id)}"
16
+ end
17
+
18
+ def render_image_payload(prompt, model:, size:, count: nil, with: nil, mask: nil, provider_options: {}) # rubocop:disable Lint/UnusedMethodArgument
19
+ @model = model
20
+ payload = if gemini_image_model?(model)
21
+ render_gemini_image_payload(prompt, with:, count:, size:)
22
+ else
23
+ render_imagen_payload(prompt, count:, size:)
24
+ end
25
+
26
+ Support::Utils.deep_merge(payload, provider_options)
27
+ end
28
+
29
+ def parse_image_response(response, model:)
30
+ parse_image_responses(response, model:).first
31
+ end
32
+
33
+ def parse_image_responses(response, model:)
34
+ data = response.body
35
+ return parse_gemini_image_responses(data, model:) if gemini_image_model?(model)
36
+
37
+ parse_imagen_responses(data, model:)
38
+ end
39
+
40
+ private
41
+
42
+ def validate_paint_inputs!(with:, mask:)
43
+ if gemini_image_model?(@model)
44
+ raise UnsupportedAttachmentError, 'image mask' if mask
45
+
46
+ return
47
+ end
48
+
49
+ return if with.nil? && mask.nil?
50
+
51
+ raise UnsupportedAttachmentError, 'image reference'
52
+ end
53
+
54
+ def render_imagen_payload(prompt, count: nil, size: nil)
55
+ RubyLLM.logger.debug { "Ignoring size #{size}. Imagen sizing is not supported." } if size
56
+
57
+ {
58
+ instances: [
59
+ {
60
+ prompt: prompt
61
+ }
62
+ ],
63
+ parameters: {
64
+ sampleCount: count || 1
65
+ }
66
+ }
67
+ end
68
+
69
+ def render_gemini_image_payload(prompt, with:, count: nil, size: nil)
70
+ generation_config = { responseModalities: %w[TEXT IMAGE] }
71
+ generation_config[:candidateCount] = count if count && count > 1
72
+ image_config = build_image_config(size)
73
+ generation_config[:imageConfig] = image_config if image_config
74
+
75
+ {
76
+ contents: [
77
+ {
78
+ role: 'user',
79
+ parts: Media.format_content(prompt, image_attachments(with))
80
+ }
81
+ ],
82
+ generationConfig: generation_config
83
+ }
84
+ end
85
+
86
+ # Gemini sizes an image by aspect ratio and resolution tier rather
87
+ # than by pixel dimensions, so WxH becomes the ratio it reduces to.
88
+ def build_image_config(size)
89
+ value = size.to_s.strip
90
+ return nil if value.empty?
91
+ return { imageSize: value.upcase } if IMAGE_SIZES.include?(value.upcase)
92
+ return { aspectRatio: value } if value.match?(ASPECT_RATIO)
93
+
94
+ match = value.match(PIXEL_DIMENSIONS)
95
+ raise ArgumentError, unsupported_size_message(size) unless match
96
+
97
+ { aspectRatio: aspect_ratio(match[1].to_i, match[2].to_i) }
98
+ end
99
+
100
+ def aspect_ratio(width, height)
101
+ raise ArgumentError, unsupported_size_message("#{width}x#{height}") unless width.positive? && height.positive?
102
+
103
+ divisor = width.gcd(height)
104
+ "#{width / divisor}:#{height / divisor}"
105
+ end
106
+
107
+ def unsupported_size_message(size)
108
+ "Gemini cannot generate an image of size #{size.inspect}. Give pixel dimensions such as " \
109
+ '"1024x1024", an aspect ratio such as "16:9", or a resolution ' \
110
+ "such as #{IMAGE_SIZES.join(', ')}."
111
+ end
112
+
113
+ def parse_imagen_responses(data, model:)
114
+ predictions = Array(data['predictions']).select { |prediction| prediction['bytesBase64Encoded'] }
115
+ raise Error, 'Unexpected response format from Gemini image generation API' if predictions.empty?
116
+
117
+ predictions.map do |image_data|
118
+ Image.new(
119
+ data: image_data['bytesBase64Encoded'],
120
+ mime_type: image_data['mimeType'] || 'image/png',
121
+ model: model
122
+ )
123
+ end
124
+ end
125
+
126
+ def parse_gemini_image_responses(data, model:)
127
+ parts = gemini_image_parts(data)
128
+ raise Error, 'Unexpected response format from Gemini image generation API' if parts.empty?
129
+
130
+ parts.map.with_index do |image_data, index|
131
+ Image.new(
132
+ data: image_data['data'],
133
+ mime_type: image_data['mimeType'] || 'image/png',
134
+ model: data['modelVersion'] || model,
135
+ usage: index.zero? ? gemini_image_usage(data) : {}
136
+ )
137
+ end
138
+ end
139
+
140
+ def gemini_image_parts(data)
141
+ Array(data['candidates']).filter_map do |candidate|
142
+ parts = candidate.dig('content', 'parts') || []
143
+ parts.filter_map { |part| part['inlineData'] }.find do |inline_data|
144
+ image_mime_type?(inline_data['mimeType']) && inline_data['data']
145
+ end
146
+ end
147
+ end
148
+
149
+ def image_mime_type?(mime_type)
150
+ mime_type.nil? || mime_type.start_with?('image/')
151
+ end
152
+
153
+ def gemini_image_usage(data)
154
+ {
155
+ 'input_tokens' => input_tokens(data),
156
+ 'output_tokens' => calculate_output_tokens(data)
157
+ }.compact
158
+ end
159
+
160
+ def image_attachments(sources)
161
+ Attachment.wrap(sources, config: @config).each do |attachment|
162
+ raise UnsupportedAttachmentError, attachment.mime_type unless attachment.image?
163
+ end
164
+ end
165
+
166
+ def gemini_image_model?(model)
167
+ id = model_id(model).downcase
168
+ return true if id.start_with?('nano-banana', 'nanobanana')
169
+
170
+ id.start_with?('gemini-') && id.include?('-image')
171
+ end
172
+
173
+ def image_endpoint_action(model)
174
+ gemini_image_model?(model) ? 'generateContent' : 'predict'
175
+ end
176
+
177
+ def model_id(model)
178
+ model.respond_to?(:id) ? model.id : model.to_s
179
+ end
180
+ end
181
+ end
182
+ end
183
+ end
@@ -0,0 +1,140 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ class LiveTranscription < Protocol # :nodoc: all
7
+ def transcribe(audio_file, model:, language:, format: nil, speaker_names: nil,
8
+ speaker_references: nil, provider_options: {}, prompt: nil, temperature: nil, &block)
9
+ validate_transcription_request(format:, speaker_names:, speaker_references:, temperature:)
10
+ audio = transcription_audio(audio_file)
11
+ setup = render_transcription_setup(model:, language:, prompt:, provider_options:)
12
+
13
+ track_usage(:transcription) do
14
+ @usage_tracker.start
15
+ events = collect_transcription(audio, setup, &block)
16
+ result = parse_transcription_events(events, audio:, model:)
17
+ block&.call(TranscriptionChunk.new(type: TranscriptionChunk::DONE, text: result.text, raw: events.last))
18
+ result
19
+ end
20
+ end
21
+
22
+ def validate_transcription_request(format:, speaker_names:, speaker_references:, temperature:)
23
+ return unless format || speaker_names || speaker_references || temperature
24
+
25
+ raise ArgumentError, 'Google Live transcription does not accept format, diarization, or temperature'
26
+ end
27
+
28
+ def transcription_audio(file)
29
+ attachments = Attachment.wrap(file, config: @config)
30
+ raise ArgumentError, 'Transcription requires exactly one audio file' unless attachments.one?
31
+
32
+ audio = RubyLLM::Transcription::WavAudio.new(attachments.first.content)
33
+ unless [audio.encoding, audio.channels, audio.bits_per_sample] == [1, 1, 16] &&
34
+ audio.data.bytesize.positive? && audio.data.bytesize.even?
35
+ raise ArgumentError, 'Google Live transcription requires non-empty mono 16-bit PCM WAV audio'
36
+ end
37
+
38
+ audio
39
+ end
40
+
41
+ def render_transcription_setup(model:, language:, prompt:, provider_options:)
42
+ payload = { model: transcription_model_name(model), generationConfig: { responseModalities: ['TEXT'] },
43
+ inputAudioTranscription: { languageCodes: language && Array(language),
44
+ customVocabulary: prompt && Array(prompt) }.compact,
45
+ realtimeInputConfig: { automaticActivityDetection: { disabled: true } } }
46
+ payload = Support::Utils.deep_merge(payload, provider_options)
47
+ validate_transcription_setup(payload)
48
+ { setup: payload }
49
+ end
50
+
51
+ def validate_transcription_setup(payload)
52
+ config = payload.fetch(:inputAudioTranscription)
53
+ if config[:diarization] || config[:wordTimestamp]
54
+ raise ArgumentError, 'Google Live transcription does not support diarization or word timestamps'
55
+ end
56
+ return if payload.dig(:realtimeInputConfig, :automaticActivityDetection, :disabled) == true
57
+
58
+ raise ArgumentError, 'Google file transcription requires manual activity boundaries'
59
+ end
60
+
61
+ def transcription_model_name(model)
62
+ "models/#{model}"
63
+ end
64
+
65
+ def websocket_service
66
+ 'google.ai.generativelanguage.v1beta.GenerativeService.BidiGenerateContent'
67
+ end
68
+
69
+ def transcription_websocket_url
70
+ uri = URI.parse(@provider.api_base)
71
+ uri.scheme = uri.scheme == 'https' ? 'wss' : 'ws'
72
+ uri.path = "/ws/#{websocket_service}"
73
+ uri.to_s
74
+ end
75
+
76
+ def collect_transcription(audio, setup, &block)
77
+ events = []
78
+ ready = Queue.new
79
+ Transport::WebsocketConnection.open(
80
+ transcription_websocket_url, headers: @provider.headers.transform_keys(&:to_s), config: @config
81
+ ) do |socket|
82
+ socket.send_text(JSON.generate(setup))
83
+ write = lambda { |connection|
84
+ ready.pop
85
+ send_transcription_audio(connection, audio)
86
+ }
87
+ socket.each_message(write:) do |message|
88
+ event = JSON.parse(message)
89
+ events << event
90
+ ready << true if event.key?('setupComplete')
91
+ process_transcription_event(event, &block)
92
+ socket.close if event.dig('serverContent', 'generationComplete')
93
+ end
94
+ end
95
+ unless events.last&.dig('serverContent', 'generationComplete')
96
+ raise Error, 'Google Live transcription ended before generation completed'
97
+ end
98
+
99
+ events
100
+ end
101
+
102
+ def send_transcription_audio(socket, audio)
103
+ socket.send_text(JSON.generate(realtimeInput: { activityStart: {} }))
104
+ bytes = [audio.sample_rate / 10, 1].max * 2
105
+ offset = 0
106
+ while offset < audio.data.bytesize
107
+ chunk = audio.data.byteslice(offset, bytes)
108
+ input = { audio: { mimeType: "audio/pcm;rate=#{audio.sample_rate}", data: Base64.strict_encode64(chunk) } }
109
+ socket.send_text(JSON.generate(realtimeInput: input))
110
+ offset += chunk.bytesize
111
+ end
112
+ socket.send_text(JSON.generate(realtimeInput: { activityEnd: {} }))
113
+ end
114
+
115
+ def process_transcription_event(event)
116
+ raise Error, event.dig('error', 'message') || 'Google Live transcription failed' if event['error']
117
+ return unless block_given?
118
+
119
+ content = event['serverContent'] || {}
120
+ if content['interimInputTranscription']
121
+ yield TranscriptionChunk.new(type: TranscriptionChunk::PARTIAL,
122
+ text: content.dig('interimInputTranscription', 'text'), raw: event)
123
+ elsif content['inputTranscription']
124
+ yield TranscriptionChunk.new(type: TranscriptionChunk::DELTA,
125
+ delta: content.dig('inputTranscription', 'text'), raw: event)
126
+ end
127
+ end
128
+
129
+ def parse_transcription_events(events, audio:, model:)
130
+ text = events.filter_map { |event| event.dig('serverContent', 'inputTranscription', 'text') }.join
131
+ usage = events.reverse.find { |event| event['usageMetadata'] }&.fetch('usageMetadata') || {}
132
+ RubyLLM::Transcription.new(
133
+ text:, model:, duration: audio.duration,
134
+ input_tokens: usage['promptTokenCount'], output_tokens: usage['candidatesTokenCount']
135
+ )
136
+ end
137
+ end
138
+ end
139
+ end
140
+ end
@@ -4,21 +4,17 @@ require 'base64'
4
4
  require 'stringio'
5
5
 
6
6
  module RubyLLM
7
- module Providers
7
+ module Protocols
8
8
  class Gemini # rubocop:disable Style/Documentation
9
9
  # Media handling methods for the Gemini API integration
10
10
  module Media
11
11
  module_function
12
12
 
13
- def format_content(content)
14
- return content.value if content.is_a?(RubyLLM::Content::Raw)
15
- return [format_text(content.to_json)] if content.is_a?(Hash) || content.is_a?(Array)
16
- return [format_text(content)] unless content.is_a?(Content)
17
-
13
+ def format_content(content, attachments = [])
18
14
  parts = []
19
- parts << format_text(content.text) if content.text
15
+ parts << format_text(content) if content
20
16
 
21
- content.attachments.each do |attachment|
17
+ attachments.each do |attachment|
22
18
  parts << format_content_attachment(attachment)
23
19
  end
24
20
 
@@ -37,6 +33,8 @@ module RubyLLM
37
33
  end
38
34
 
39
35
  def format_attachment(attachment)
36
+ return format_file_data(attachment) if attachment.provider_file?
37
+
40
38
  {
41
39
  inline_data: {
42
40
  mime_type: attachment.mime_type,
@@ -45,6 +43,16 @@ module RubyLLM
45
43
  }
46
44
  end
47
45
 
46
+ def format_file_data(attachment)
47
+ uri = attachment.provider_file_uri || attachment.provider_file_id
48
+ {
49
+ file_data: {
50
+ mime_type: attachment.mime_type,
51
+ file_uri: uri
52
+ }
53
+ }
54
+ end
55
+
48
56
  def format_text_file(text_file)
49
57
  {
50
58
  text: text_file.for_llm
@@ -76,9 +84,7 @@ module RubyLLM
76
84
 
77
85
  text = text.join
78
86
  text = nil if text.empty?
79
- return text if attachments.empty?
80
-
81
- Content.new(text, attachments)
87
+ [text, attachments]
82
88
  end
83
89
 
84
90
  def build_inline_attachment(inline_data, index)