ruby_llm 1.16.0 → 2.0.0.rc4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (480) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +108 -44
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/legacy_content_sql.rb +34 -0
  91. data/lib/generators/ruby_llm/upgrade/online_copy_migration/data.rb +341 -0
  92. data/lib/generators/ruby_llm/upgrade/online_copy_migration/journal.rb +171 -0
  93. data/lib/generators/ruby_llm/upgrade/online_copy_migration/verification.rb +197 -0
  94. data/lib/generators/ruby_llm/upgrade/online_copy_migration.rb +174 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +468 -0
  96. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
  97. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +247 -0
  98. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +662 -0
  99. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +251 -0
  100. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  101. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +188 -0
  102. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +359 -0
  103. data/lib/ruby_llm/accounting/usage.rb +254 -0
  104. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  105. data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
  106. data/lib/ruby_llm/active_record/batch.rb +97 -0
  107. data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
  108. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  109. data/lib/ruby_llm/active_record/model.rb +135 -0
  110. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  111. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  112. data/lib/ruby_llm/active_record/usage.rb +61 -0
  113. data/lib/ruby_llm/agent.rb +1066 -151
  114. data/lib/ruby_llm/aliases.json +291 -101
  115. data/lib/ruby_llm/attachment.rb +192 -48
  116. data/lib/ruby_llm/batch.rb +432 -0
  117. data/lib/ruby_llm/cached_content.rb +112 -0
  118. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  119. data/lib/ruby_llm/chat.rb +1131 -198
  120. data/lib/ruby_llm/chunk.rb +10 -0
  121. data/lib/ruby_llm/citation.rb +105 -0
  122. data/lib/ruby_llm/configuration.rb +261 -24
  123. data/lib/ruby_llm/context.rb +128 -6
  124. data/lib/ruby_llm/cost.rb +217 -80
  125. data/lib/ruby_llm/downloaded_file.rb +33 -0
  126. data/lib/ruby_llm/embedding.rb +121 -17
  127. data/lib/ruby_llm/embedding_request.rb +53 -0
  128. data/lib/ruby_llm/error.rb +159 -23
  129. data/lib/ruby_llm/fallback.rb +133 -0
  130. data/lib/ruby_llm/files/mime_type.rb +97 -0
  131. data/lib/ruby_llm/image.rb +153 -32
  132. data/lib/ruby_llm/message.rb +242 -54
  133. data/lib/ruby_llm/model/modalities.rb +17 -4
  134. data/lib/ruby_llm/model/pricing.rb +24 -5
  135. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  136. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  137. data/lib/ruby_llm/model.rb +244 -2
  138. data/lib/ruby_llm/models/aliases.rb +41 -0
  139. data/lib/ruby_llm/models/registry.rb +165 -0
  140. data/lib/ruby_llm/models/schema.rb +99 -0
  141. data/lib/ruby_llm/models.json +76038 -34173
  142. data/lib/ruby_llm/models.rb +477 -215
  143. data/lib/ruby_llm/moderation.rb +139 -26
  144. data/lib/ruby_llm/ocr.rb +112 -0
  145. data/lib/ruby_llm/prompt.rb +79 -0
  146. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  147. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  148. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  149. data/lib/ruby_llm/protocol.rb +662 -0
  150. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  151. data/lib/ruby_llm/protocols/anthropic/chat.rb +543 -0
  152. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  153. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  154. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  155. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  156. data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
  157. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  158. data/lib/ruby_llm/protocols/anthropic.rb +101 -0
  159. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  160. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +110 -0
  161. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  162. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  163. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  164. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  165. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  166. data/lib/ruby_llm/protocols/chat_completions/chat.rb +484 -0
  167. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  168. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  169. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  170. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  171. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  172. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  173. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +63 -0
  174. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  175. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  176. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  177. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  178. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  179. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  180. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  181. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  182. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  183. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  184. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  185. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  186. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  187. data/lib/ruby_llm/protocols/cohere/rerank.rb +59 -0
  188. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  189. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  190. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  191. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  192. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  193. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  194. data/lib/ruby_llm/protocols/converse/chat.rb +694 -0
  195. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  196. data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
  197. data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
  198. data/lib/ruby_llm/protocols/converse.rb +54 -0
  199. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  200. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  201. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  202. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  203. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  204. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  208. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  209. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  210. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  211. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  212. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  213. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  214. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  215. data/lib/ruby_llm/protocols/files.rb +119 -0
  216. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  217. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  218. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  219. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +91 -0
  220. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  221. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  222. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  223. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  224. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  225. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  226. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  227. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  228. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  229. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  230. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  231. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  232. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  233. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  234. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  235. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  236. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  237. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  238. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  239. data/lib/ruby_llm/protocols/interactions/tools.rb +53 -0
  240. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  241. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  242. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  243. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  244. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  245. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  246. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  247. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  248. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  249. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  250. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  251. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  252. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  253. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  254. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  255. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  256. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  257. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  258. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  259. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  260. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  261. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  262. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  263. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  264. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  265. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  266. data/lib/ruby_llm/protocols/responses/chat.rb +466 -0
  267. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  268. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  269. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  270. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  271. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  272. data/lib/ruby_llm/protocols/responses.rb +35 -0
  273. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  274. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  275. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  276. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  277. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  278. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  279. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  280. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  281. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  282. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  283. data/lib/ruby_llm/provider.rb +560 -128
  284. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  285. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  286. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  287. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  288. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  289. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  290. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  291. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  292. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  293. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  294. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  295. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  296. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  297. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  298. data/lib/ruby_llm/providers/azure.rb +77 -78
  299. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  300. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  301. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  302. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  303. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  304. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  305. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  306. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  307. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  308. data/lib/ruby_llm/providers/cohere.rb +31 -0
  309. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  310. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  311. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  312. data/lib/ruby_llm/providers/deepseek/responses.rb +67 -0
  313. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  314. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  315. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  316. data/lib/ruby_llm/providers/gemini.rb +15 -8
  317. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  318. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  319. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  320. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  321. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  322. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  323. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  324. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  325. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  326. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  327. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  328. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  329. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  330. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  331. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  332. data/lib/ruby_llm/providers/mistral/ocr.rb +51 -0
  333. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  334. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  335. data/lib/ruby_llm/providers/mistral.rb +16 -4
  336. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  337. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  338. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  339. data/lib/ruby_llm/providers/ollama.rb +9 -8
  340. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  341. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  342. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  343. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  344. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  345. data/lib/ruby_llm/providers/openai.rb +92 -11
  346. data/lib/ruby_llm/providers/openrouter/chat.rb +126 -108
  347. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  348. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  349. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  350. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  351. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  352. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  353. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  354. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  355. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  356. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  357. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  358. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  359. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  360. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  361. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  362. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  363. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  364. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  365. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  366. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  367. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  368. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  369. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  370. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  371. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  372. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  373. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  374. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  375. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  376. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  377. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  378. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  379. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  380. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  381. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  382. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  383. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  384. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  385. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  386. data/lib/ruby_llm/providers/xai.rb +15 -5
  387. data/lib/ruby_llm/railtie.rb +7 -16
  388. data/lib/ruby_llm/rerank.rb +105 -0
  389. data/lib/ruby_llm/research_job.rb +241 -0
  390. data/lib/ruby_llm/search_results.rb +68 -0
  391. data/lib/ruby_llm/server_tool_call.rb +73 -0
  392. data/lib/ruby_llm/speech.rb +159 -0
  393. data/lib/ruby_llm/speech_chunk.rb +33 -0
  394. data/lib/ruby_llm/support/deprecator.rb +22 -0
  395. data/lib/ruby_llm/support/inspectable.rb +49 -0
  396. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  397. data/lib/ruby_llm/support/utils.rb +147 -0
  398. data/lib/ruby_llm/thinking.rb +127 -20
  399. data/lib/ruby_llm/tokenization.rb +59 -0
  400. data/lib/ruby_llm/tokens.rb +103 -33
  401. data/lib/ruby_llm/tool.rb +266 -91
  402. data/lib/ruby_llm/tool_call.rb +36 -3
  403. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  404. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  405. data/lib/ruby_llm/transcription.rb +138 -13
  406. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  407. data/lib/ruby_llm/transport/connection.rb +193 -0
  408. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  409. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  410. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  411. data/lib/ruby_llm/uploaded_file.rb +144 -0
  412. data/lib/ruby_llm/version.rb +2 -1
  413. data/lib/ruby_llm/video.rb +136 -0
  414. data/lib/ruby_llm/video_job.rb +150 -0
  415. data/lib/ruby_llm/workflow.rb +91 -0
  416. data/lib/ruby_llm.rb +380 -6
  417. data/lib/tasks/ruby_llm.rake +21 -16
  418. data/skills/rubyllm/SKILL.md +81 -0
  419. data/skills/rubyllm/agents/openai.yaml +4 -0
  420. metadata +344 -97
  421. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  422. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  423. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  424. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  425. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  426. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  427. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  428. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  429. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  430. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  431. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  432. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  433. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  434. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  435. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  436. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  437. data/lib/ruby_llm/aliases.rb +0 -41
  438. data/lib/ruby_llm/connection.rb +0 -159
  439. data/lib/ruby_llm/content.rb +0 -91
  440. data/lib/ruby_llm/deprecator.rb +0 -24
  441. data/lib/ruby_llm/error_middleware.rb +0 -81
  442. data/lib/ruby_llm/instrumentation.rb +0 -36
  443. data/lib/ruby_llm/mime_type.rb +0 -96
  444. data/lib/ruby_llm/model/info.rb +0 -164
  445. data/lib/ruby_llm/model_registry.rb +0 -39
  446. data/lib/ruby_llm/models_schema.json +0 -171
  447. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  448. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  449. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  450. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  451. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  452. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  453. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  454. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  455. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  456. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  457. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  458. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  459. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  460. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  461. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  462. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  463. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  464. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  465. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  466. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  467. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  468. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  469. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  470. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  471. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  472. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  473. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  474. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  475. data/lib/ruby_llm/streaming.rb +0 -179
  476. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  477. data/lib/ruby_llm/utils.rb +0 -130
  478. data/lib/tasks/models.rake +0 -593
  479. data/lib/tasks/release.rake +0 -94
  480. data/lib/tasks/vcr.rake +0 -124
@@ -0,0 +1,91 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Inline embedding requests and results for Gemini's asynchronous Batch API.
7
+ module EmbeddingBatches
8
+ private
9
+
10
+ def embedding_batch?(requests)
11
+ kinds = requests.map { |request| embedding_batch_payload?(request.fetch(:payload)) }.uniq
12
+ raise Error, 'Gemini batches take chat or embedding requests, not both' if kinds.size > 1
13
+
14
+ kinds.first
15
+ end
16
+
17
+ def embedding_batch_payload?(payload)
18
+ payload.key?(:content) || payload.key?(:requests)
19
+ end
20
+
21
+ def embedding_batch_response?(data)
22
+ batch = data['metadata'] || data
23
+ batch['@type']&.end_with?('.EmbedContentBatch') ||
24
+ data.dig('response', '@type')&.end_with?('.EmbedContentBatchOutput') ||
25
+ data.dig('response', 'inlinedEmbedContentResponses') ||
26
+ data.dig('output', 'inlinedEmbedContentResponses')
27
+ end
28
+
29
+ def embedding_batch_requests(request, model)
30
+ payload = request.fetch(:payload)
31
+ array_input = payload.key?(:requests)
32
+ inputs = array_input ? payload.fetch(:requests) : [payload]
33
+ raise ArgumentError, 'Gemini embedding batches require at least one text per request' if inputs.empty?
34
+
35
+ inputs.each_with_index.map do |input, index|
36
+ {
37
+ request: input.merge(model: "models/#{model}"),
38
+ metadata: {
39
+ custom_id: request.fetch(:custom_id), model:, array_input:,
40
+ embedding_index: index, embedding_count: inputs.size
41
+ }
42
+ }
43
+ end
44
+ end
45
+
46
+ def parse_embedding_batch_results(responses)
47
+ groups = responses.each_with_index.group_by do |inline, index|
48
+ inline.dig('metadata', 'custom_id') || index.to_s
49
+ end
50
+ groups.filter_map do |key, indexed|
51
+ parse_embedding_batch_group(key, indexed.map(&:first))
52
+ end
53
+ end
54
+
55
+ def parse_embedding_batch_group(key, responses)
56
+ index = batch_result_index(key)
57
+ error = responses.find { |inline| inline['error'] }
58
+ return [index, nil, batch_failure(key, error.dig('error', 'message'))] if error
59
+
60
+ metadata = responses.first.fetch('metadata')
61
+ return if responses.size < metadata.fetch('embedding_count')
62
+
63
+ positions = responses.map { |inline| inline.dig('metadata', 'embedding_index') }
64
+ unless positions.sort == (0...responses.size).to_a
65
+ return [index, nil, batch_failure(key, 'Invalid or duplicate embedding record positions')]
66
+ end
67
+
68
+ vectors = embedding_batch_vectors(responses)
69
+ return [index, nil, batch_failure(key, 'Gemini returned no embedding')] unless vectors
70
+
71
+ embedding = Embedding.new(
72
+ vectors: metadata['array_input'] ? vectors : vectors.first,
73
+ model: metadata.fetch('model'), input_tokens: embedding_batch_tokens(responses)
74
+ )
75
+ [index, embedding]
76
+ end
77
+
78
+ def embedding_batch_vectors(responses)
79
+ ordered = responses.sort_by { |inline| inline.dig('metadata', 'embedding_index') }
80
+ vectors = ordered.map { |inline| inline.dig('response', 'embedding', 'values') }
81
+ vectors if vectors.all? { |vector| vector && !vector.empty? }
82
+ end
83
+
84
+ def embedding_batch_tokens(responses)
85
+ counts = responses.filter_map { |inline| inline.dig('response', 'usageMetadata', 'promptTokenCount') }
86
+ counts.sum unless counts.empty?
87
+ end
88
+ end
89
+ end
90
+ end
91
+ end
@@ -0,0 +1,70 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Embeddings methods for the Gemini API integration
7
+ module Embeddings
8
+ module_function
9
+
10
+ def embedding_url(model:)
11
+ "models/#{model}:batchEmbedContents"
12
+ end
13
+
14
+ def render_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil, with: [],
15
+ provider_options: {})
16
+ requests = if with.any?
17
+ raise ArgumentError, 'embed one text at a time when embedding attachments' if text.is_a?(Array)
18
+
19
+ [media_embedding_payload(text, with, model:, dimensions:, task_type:, title:)]
20
+ else
21
+ [text].flatten.map do |t|
22
+ single_embedding_payload(t, model:, dimensions:, task_type:, title:)
23
+ end
24
+ end
25
+
26
+ Support::Utils.deep_merge({ requests: requests }, provider_options)
27
+ end
28
+
29
+ def supports_embedding_media?
30
+ true
31
+ end
32
+
33
+ def render_embedding(text, dimensions: nil, **options)
34
+ payload = render_embedding_payload(text, dimensions:, **options)
35
+ return payload if text.is_a?(Array) || !payload.key?(:requests)
36
+
37
+ payload.fetch(:requests).first
38
+ end
39
+ public :render_embedding
40
+
41
+ def parse_embedding_response(response, model:, text:)
42
+ vectors = response.body['embeddings']&.map { |e| e['values'] }
43
+ vectors = vectors.first if vectors&.length == 1 && !text.is_a?(Array)
44
+
45
+ Embedding.new(vectors:, model:)
46
+ end
47
+
48
+ private
49
+
50
+ def single_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil)
51
+ embedding_payload([{ text: text.to_s }], model:, dimensions:, task_type:, title:)
52
+ end
53
+
54
+ def media_embedding_payload(text, attachments, model:, dimensions:, task_type: nil, title: nil)
55
+ embedding_payload(Media.format_content(text, attachments), model:, dimensions:, task_type:, title:)
56
+ end
57
+
58
+ def embedding_payload(parts, model:, dimensions:, task_type:, title:)
59
+ {
60
+ model: "models/#{model}",
61
+ content: { parts: parts },
62
+ outputDimensionality: dimensions,
63
+ taskType: task_type,
64
+ title: title
65
+ }.compact
66
+ end
67
+ end
68
+ end
69
+ end
70
+ end
@@ -0,0 +1,32 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ module FileTranscription # :nodoc: all
7
+ def transcribe(audio_file, model:, language:, format: nil, speaker_names: nil,
8
+ speaker_references: nil, provider_options: {}, prompt: nil, temperature: nil)
9
+ raise_transcription_streaming_unsupported if block_given?
10
+ validate_transcription_request(format:, speaker_references:, temperature:)
11
+ attachments = Attachment.wrap(audio_file, config: @config)
12
+ unless attachments.one? && attachments.first.audio?
13
+ raise ArgumentError, 'Dedicated transcription requires exactly one audio file'
14
+ end
15
+
16
+ track_usage(:transcription) do
17
+ payload = render_transcription_payload(attachments.first, model:, language:, speaker_names:,
18
+ provider_options:, prompt:)
19
+ response = @connection.post(transcription_url(model), payload, usage: @usage_tracker)
20
+ parse_transcription_response(response, model:)
21
+ end
22
+ end
23
+
24
+ def validate_transcription_request(format:, speaker_references:, temperature:)
25
+ return unless format || speaker_references || temperature
26
+
27
+ raise ArgumentError, 'Dedicated transcription does not accept format, speaker references, or temperature'
28
+ end
29
+ end
30
+ end
31
+ end
32
+ end
@@ -0,0 +1,115 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Gemini Files API.
7
+ class Files < Protocols::Files
8
+ PROCESSING_POLL_INTERVAL = 2
9
+ PROCESSING_TIMEOUT = 600
10
+
11
+ # rubocop:disable-next Lint/UnusedMethodArgument
12
+ def upload(file, filename: nil, purpose: nil, expires_in: nil, uri: nil, content_type: nil,
13
+ provider_options: {})
14
+ attachment = file_attachment(file, filename:)
15
+ display_name = provider_options[:display_name] || attachment.filename
16
+ upload_url = start_resumable_upload(attachment, display_name:)
17
+ response = upload_file_bytes(upload_url, attachment)
18
+ await_active(parse_file_response(response.body.fetch('file')))
19
+ end
20
+
21
+ def download(file_id)
22
+ file = find(file_id)
23
+ download_uri = file.metadata['downloadUri']
24
+ raise Error, 'gemini file has no download URI' unless download_uri
25
+
26
+ gemini_connection(request_json: false, response_json: false).get(download_uri) do |request|
27
+ request.headers.merge!(@provider.headers)
28
+ end.body
29
+ end
30
+
31
+ private
32
+
33
+ # Gemini rejects a file reference until processing finishes, which
34
+ # for video takes far longer than the upload itself.
35
+ def await_active(file)
36
+ deadline = Time.now + PROCESSING_TIMEOUT
37
+
38
+ while file.status == 'PROCESSING'
39
+ raise Error, "gemini is still processing #{file.id}" if Time.now >= deadline
40
+
41
+ sleep PROCESSING_POLL_INTERVAL
42
+ file = find(file.id)
43
+ end
44
+
45
+ raise Error, "gemini failed to process #{file.id}" if file.status == 'FAILED'
46
+
47
+ file
48
+ end
49
+
50
+ def file_info_url(file_id)
51
+ gemini_file_name(file_id)
52
+ end
53
+
54
+ def start_resumable_upload(attachment, display_name:)
55
+ response = gemini_connection(request_json: true, response_json: true).post(gemini_upload_url) do |request|
56
+ request.headers.merge!(@provider.headers)
57
+ request.headers['X-Goog-Upload-Protocol'] = 'resumable'
58
+ request.headers['X-Goog-Upload-Command'] = 'start'
59
+ request.headers['X-Goog-Upload-Header-Content-Length'] = attachment.content.bytesize.to_s
60
+ request.headers['X-Goog-Upload-Header-Content-Type'] = file_content_type(attachment)
61
+ request.headers['Content-Type'] = 'application/json'
62
+ request.body = { file: { display_name: display_name } }
63
+ end
64
+
65
+ response.headers['x-goog-upload-url'] || response.headers['X-Goog-Upload-URL'] ||
66
+ raise(Error, 'gemini did not return an upload URL')
67
+ end
68
+
69
+ def upload_file_bytes(upload_url, attachment)
70
+ gemini_connection(request_json: false, response_json: true).post(upload_url) do |request|
71
+ request.headers.merge!(@provider.headers)
72
+ request.headers['Content-Length'] = attachment.content.bytesize.to_s
73
+ request.headers['X-Goog-Upload-Offset'] = '0'
74
+ request.headers['X-Goog-Upload-Command'] = 'upload, finalize'
75
+ request.body = attachment.content
76
+ end
77
+ end
78
+
79
+ def parse_file_response(data)
80
+ uploaded_file(
81
+ data,
82
+ id: data['name'],
83
+ filename: data['displayName'],
84
+ byte_size: data['sizeBytes']&.to_i,
85
+ created_at: timestamp(data['createTime']),
86
+ expires_at: timestamp(data['expirationTime']),
87
+ mime_type: data['mimeType'],
88
+ status: data['state'],
89
+ uri: data['uri']
90
+ )
91
+ end
92
+
93
+ def gemini_file_name(file_id)
94
+ file_id.to_s.start_with?('files/') ? file_id : "files/#{file_id}"
95
+ end
96
+
97
+ def gemini_upload_url
98
+ base = @provider.api_base.to_s.sub(%r{/+\z}, '')
99
+ "#{base.sub(%r{/v1beta\z}, '/upload/v1beta').sub(%r{/v1\z}, '/upload/v1')}/files"
100
+ end
101
+
102
+ def gemini_connection(request_json:, response_json:)
103
+ connection = Transport::Connection.basic(@config) do |connection|
104
+ connection.request :json if request_json
105
+ connection.response :json if response_json
106
+ connection.adapter @config.faraday_adapter
107
+ connection.use :llm_errors, provider: @provider
108
+ end
109
+ connection.url_prefix = @provider.api_base
110
+ connection
111
+ end
112
+ end
113
+ end
114
+ end
115
+ end
@@ -0,0 +1,183 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Image generation methods for the Gemini API implementation
7
+ module Images
8
+ IMAGE_SIZES = %w[512 512P 512PX 1K 2K 4K].freeze
9
+ ASPECT_RATIO = /\A\d+:\d+\z/
10
+ PIXEL_DIMENSIONS = /\A(\d+)\s*[x×]\s*(\d+)\z/i
11
+
12
+ def images_url(with: nil, mask: nil) # rubocop:disable Lint/UnusedMethodArgument
13
+ id = model_id(@model)
14
+
15
+ "models/#{id}:#{image_endpoint_action(id)}"
16
+ end
17
+
18
+ def render_image_payload(prompt, model:, size:, count: nil, with: nil, mask: nil, provider_options: {}) # rubocop:disable Lint/UnusedMethodArgument
19
+ @model = model
20
+ payload = if gemini_image_model?(model)
21
+ render_gemini_image_payload(prompt, with:, count:, size:)
22
+ else
23
+ render_imagen_payload(prompt, count:, size:)
24
+ end
25
+
26
+ Support::Utils.deep_merge(payload, provider_options)
27
+ end
28
+
29
+ def parse_image_response(response, model:)
30
+ parse_image_responses(response, model:).first
31
+ end
32
+
33
+ def parse_image_responses(response, model:)
34
+ data = response.body
35
+ return parse_gemini_image_responses(data, model:) if gemini_image_model?(model)
36
+
37
+ parse_imagen_responses(data, model:)
38
+ end
39
+
40
+ private
41
+
42
+ def validate_paint_inputs!(with:, mask:)
43
+ if gemini_image_model?(@model)
44
+ raise UnsupportedAttachmentError, 'image mask' if mask
45
+
46
+ return
47
+ end
48
+
49
+ return if with.nil? && mask.nil?
50
+
51
+ raise UnsupportedAttachmentError, 'image reference'
52
+ end
53
+
54
+ def render_imagen_payload(prompt, count: nil, size: nil)
55
+ RubyLLM.logger.debug { "Ignoring size #{size}. Imagen sizing is not supported." } if size
56
+
57
+ {
58
+ instances: [
59
+ {
60
+ prompt: prompt
61
+ }
62
+ ],
63
+ parameters: {
64
+ sampleCount: count || 1
65
+ }
66
+ }
67
+ end
68
+
69
+ def render_gemini_image_payload(prompt, with:, count: nil, size: nil)
70
+ generation_config = { responseModalities: %w[TEXT IMAGE] }
71
+ generation_config[:candidateCount] = count if count && count > 1
72
+ image_config = build_image_config(size)
73
+ generation_config[:imageConfig] = image_config if image_config
74
+
75
+ {
76
+ contents: [
77
+ {
78
+ role: 'user',
79
+ parts: Media.format_content(prompt, image_attachments(with))
80
+ }
81
+ ],
82
+ generationConfig: generation_config
83
+ }
84
+ end
85
+
86
+ # Gemini sizes an image by aspect ratio and resolution tier rather
87
+ # than by pixel dimensions, so WxH becomes the ratio it reduces to.
88
+ def build_image_config(size)
89
+ value = size.to_s.strip
90
+ return nil if value.empty?
91
+ return { imageSize: value.upcase } if IMAGE_SIZES.include?(value.upcase)
92
+ return { aspectRatio: value } if value.match?(ASPECT_RATIO)
93
+
94
+ match = value.match(PIXEL_DIMENSIONS)
95
+ raise ArgumentError, unsupported_size_message(size) unless match
96
+
97
+ { aspectRatio: aspect_ratio(match[1].to_i, match[2].to_i) }
98
+ end
99
+
100
+ def aspect_ratio(width, height)
101
+ raise ArgumentError, unsupported_size_message("#{width}x#{height}") unless width.positive? && height.positive?
102
+
103
+ divisor = width.gcd(height)
104
+ "#{width / divisor}:#{height / divisor}"
105
+ end
106
+
107
+ def unsupported_size_message(size)
108
+ "Gemini cannot generate an image of size #{size.inspect}. Give pixel dimensions such as " \
109
+ '"1024x1024", an aspect ratio such as "16:9", or a resolution ' \
110
+ "such as #{IMAGE_SIZES.join(', ')}."
111
+ end
112
+
113
+ def parse_imagen_responses(data, model:)
114
+ predictions = Array(data['predictions']).select { |prediction| prediction['bytesBase64Encoded'] }
115
+ raise Error, 'Unexpected response format from Gemini image generation API' if predictions.empty?
116
+
117
+ predictions.map do |image_data|
118
+ Image.new(
119
+ data: image_data['bytesBase64Encoded'],
120
+ mime_type: image_data['mimeType'] || 'image/png',
121
+ model: model
122
+ )
123
+ end
124
+ end
125
+
126
+ def parse_gemini_image_responses(data, model:)
127
+ parts = gemini_image_parts(data)
128
+ raise Error, 'Unexpected response format from Gemini image generation API' if parts.empty?
129
+
130
+ parts.map.with_index do |image_data, index|
131
+ Image.new(
132
+ data: image_data['data'],
133
+ mime_type: image_data['mimeType'] || 'image/png',
134
+ model: data['modelVersion'] || model,
135
+ usage: index.zero? ? gemini_image_usage(data) : {}
136
+ )
137
+ end
138
+ end
139
+
140
+ def gemini_image_parts(data)
141
+ Array(data['candidates']).filter_map do |candidate|
142
+ parts = candidate.dig('content', 'parts') || []
143
+ parts.filter_map { |part| part['inlineData'] }.find do |inline_data|
144
+ image_mime_type?(inline_data['mimeType']) && inline_data['data']
145
+ end
146
+ end
147
+ end
148
+
149
+ def image_mime_type?(mime_type)
150
+ mime_type.nil? || mime_type.start_with?('image/')
151
+ end
152
+
153
+ def gemini_image_usage(data)
154
+ {
155
+ 'input_tokens' => input_tokens(data),
156
+ 'output_tokens' => calculate_output_tokens(data)
157
+ }.compact
158
+ end
159
+
160
+ def image_attachments(sources)
161
+ Attachment.wrap(sources, config: @config).each do |attachment|
162
+ raise UnsupportedAttachmentError, attachment.mime_type unless attachment.image?
163
+ end
164
+ end
165
+
166
+ def gemini_image_model?(model)
167
+ id = model_id(model).downcase
168
+ return true if id.start_with?('nano-banana', 'nanobanana')
169
+
170
+ id.start_with?('gemini-') && id.include?('-image')
171
+ end
172
+
173
+ def image_endpoint_action(model)
174
+ gemini_image_model?(model) ? 'generateContent' : 'predict'
175
+ end
176
+
177
+ def model_id(model)
178
+ model.respond_to?(:id) ? model.id : model.to_s
179
+ end
180
+ end
181
+ end
182
+ end
183
+ end
@@ -0,0 +1,140 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ class LiveTranscription < Protocol # :nodoc: all
7
+ def transcribe(audio_file, model:, language:, format: nil, speaker_names: nil,
8
+ speaker_references: nil, provider_options: {}, prompt: nil, temperature: nil, &block)
9
+ validate_transcription_request(format:, speaker_names:, speaker_references:, temperature:)
10
+ audio = transcription_audio(audio_file)
11
+ setup = render_transcription_setup(model:, language:, prompt:, provider_options:)
12
+
13
+ track_usage(:transcription) do
14
+ @usage_tracker.start
15
+ events = collect_transcription(audio, setup, &block)
16
+ result = parse_transcription_events(events, audio:, model:)
17
+ block&.call(TranscriptionChunk.new(type: TranscriptionChunk::DONE, text: result.text, raw: events.last))
18
+ result
19
+ end
20
+ end
21
+
22
+ def validate_transcription_request(format:, speaker_names:, speaker_references:, temperature:)
23
+ return unless format || speaker_names || speaker_references || temperature
24
+
25
+ raise ArgumentError, 'Google Live transcription does not accept format, diarization, or temperature'
26
+ end
27
+
28
+ def transcription_audio(file)
29
+ attachments = Attachment.wrap(file, config: @config)
30
+ raise ArgumentError, 'Transcription requires exactly one audio file' unless attachments.one?
31
+
32
+ audio = RubyLLM::Transcription::WavAudio.new(attachments.first.content)
33
+ unless [audio.encoding, audio.channels, audio.bits_per_sample] == [1, 1, 16] &&
34
+ audio.data.bytesize.positive? && audio.data.bytesize.even?
35
+ raise ArgumentError, 'Google Live transcription requires non-empty mono 16-bit PCM WAV audio'
36
+ end
37
+
38
+ audio
39
+ end
40
+
41
+ def render_transcription_setup(model:, language:, prompt:, provider_options:)
42
+ payload = { model: transcription_model_name(model), generationConfig: { responseModalities: ['TEXT'] },
43
+ inputAudioTranscription: { languageCodes: language && Array(language),
44
+ customVocabulary: prompt && Array(prompt) }.compact,
45
+ realtimeInputConfig: { automaticActivityDetection: { disabled: true } } }
46
+ payload = Support::Utils.deep_merge(payload, provider_options)
47
+ validate_transcription_setup(payload)
48
+ { setup: payload }
49
+ end
50
+
51
+ def validate_transcription_setup(payload)
52
+ config = payload.fetch(:inputAudioTranscription)
53
+ if config[:diarization] || config[:wordTimestamp]
54
+ raise ArgumentError, 'Google Live transcription does not support diarization or word timestamps'
55
+ end
56
+ return if payload.dig(:realtimeInputConfig, :automaticActivityDetection, :disabled) == true
57
+
58
+ raise ArgumentError, 'Google file transcription requires manual activity boundaries'
59
+ end
60
+
61
+ def transcription_model_name(model)
62
+ "models/#{model}"
63
+ end
64
+
65
+ def websocket_service
66
+ 'google.ai.generativelanguage.v1beta.GenerativeService.BidiGenerateContent'
67
+ end
68
+
69
+ def transcription_websocket_url
70
+ uri = URI.parse(@provider.api_base)
71
+ uri.scheme = uri.scheme == 'https' ? 'wss' : 'ws'
72
+ uri.path = "/ws/#{websocket_service}"
73
+ uri.to_s
74
+ end
75
+
76
+ def collect_transcription(audio, setup, &block)
77
+ events = []
78
+ ready = Queue.new
79
+ Transport::WebsocketConnection.open(
80
+ transcription_websocket_url, headers: @provider.headers.transform_keys(&:to_s), config: @config
81
+ ) do |socket|
82
+ socket.send_text(JSON.generate(setup))
83
+ write = lambda { |connection|
84
+ ready.pop
85
+ send_transcription_audio(connection, audio)
86
+ }
87
+ socket.each_message(write:) do |message|
88
+ event = JSON.parse(message)
89
+ events << event
90
+ ready << true if event.key?('setupComplete')
91
+ process_transcription_event(event, &block)
92
+ socket.close if event.dig('serverContent', 'generationComplete')
93
+ end
94
+ end
95
+ unless events.last&.dig('serverContent', 'generationComplete')
96
+ raise Error, 'Google Live transcription ended before generation completed'
97
+ end
98
+
99
+ events
100
+ end
101
+
102
+ def send_transcription_audio(socket, audio)
103
+ socket.send_text(JSON.generate(realtimeInput: { activityStart: {} }))
104
+ bytes = [audio.sample_rate / 10, 1].max * 2
105
+ offset = 0
106
+ while offset < audio.data.bytesize
107
+ chunk = audio.data.byteslice(offset, bytes)
108
+ input = { audio: { mimeType: "audio/pcm;rate=#{audio.sample_rate}", data: Base64.strict_encode64(chunk) } }
109
+ socket.send_text(JSON.generate(realtimeInput: input))
110
+ offset += chunk.bytesize
111
+ end
112
+ socket.send_text(JSON.generate(realtimeInput: { activityEnd: {} }))
113
+ end
114
+
115
+ def process_transcription_event(event)
116
+ raise Error, event.dig('error', 'message') || 'Google Live transcription failed' if event['error']
117
+ return unless block_given?
118
+
119
+ content = event['serverContent'] || {}
120
+ if content['interimInputTranscription']
121
+ yield TranscriptionChunk.new(type: TranscriptionChunk::PARTIAL,
122
+ text: content.dig('interimInputTranscription', 'text'), raw: event)
123
+ elsif content['inputTranscription']
124
+ yield TranscriptionChunk.new(type: TranscriptionChunk::DELTA,
125
+ delta: content.dig('inputTranscription', 'text'), raw: event)
126
+ end
127
+ end
128
+
129
+ def parse_transcription_events(events, audio:, model:)
130
+ text = events.filter_map { |event| event.dig('serverContent', 'inputTranscription', 'text') }.join
131
+ usage = events.reverse.find { |event| event['usageMetadata'] }&.fetch('usageMetadata') || {}
132
+ RubyLLM::Transcription.new(
133
+ text:, model:, duration: audio.duration,
134
+ input_tokens: usage['promptTokenCount'], output_tokens: usage['candidatesTokenCount']
135
+ )
136
+ end
137
+ end
138
+ end
139
+ end
140
+ end