ruby_llm 1.16.0 → 2.0.0.rc2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (475) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +85 -32
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
  91. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
  92. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
  93. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
  94. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  96. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
  97. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
  98. data/lib/ruby_llm/accounting/usage.rb +245 -0
  99. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  100. data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
  101. data/lib/ruby_llm/active_record/batch.rb +97 -0
  102. data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
  103. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  104. data/lib/ruby_llm/active_record/model.rb +135 -0
  105. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  106. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  107. data/lib/ruby_llm/active_record/usage.rb +61 -0
  108. data/lib/ruby_llm/agent.rb +1065 -151
  109. data/lib/ruby_llm/aliases.json +268 -100
  110. data/lib/ruby_llm/attachment.rb +187 -48
  111. data/lib/ruby_llm/batch.rb +432 -0
  112. data/lib/ruby_llm/cached_content.rb +112 -0
  113. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  114. data/lib/ruby_llm/chat.rb +1127 -198
  115. data/lib/ruby_llm/chunk.rb +10 -0
  116. data/lib/ruby_llm/citation.rb +105 -0
  117. data/lib/ruby_llm/configuration.rb +261 -24
  118. data/lib/ruby_llm/context.rb +128 -6
  119. data/lib/ruby_llm/cost.rb +217 -80
  120. data/lib/ruby_llm/downloaded_file.rb +33 -0
  121. data/lib/ruby_llm/embedding.rb +121 -17
  122. data/lib/ruby_llm/embedding_request.rb +53 -0
  123. data/lib/ruby_llm/error.rb +159 -23
  124. data/lib/ruby_llm/fallback.rb +133 -0
  125. data/lib/ruby_llm/files/mime_type.rb +97 -0
  126. data/lib/ruby_llm/image.rb +153 -32
  127. data/lib/ruby_llm/message.rb +233 -54
  128. data/lib/ruby_llm/model/modalities.rb +17 -4
  129. data/lib/ruby_llm/model/pricing.rb +24 -5
  130. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  131. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  132. data/lib/ruby_llm/model.rb +244 -2
  133. data/lib/ruby_llm/models/aliases.rb +41 -0
  134. data/lib/ruby_llm/models/registry.rb +165 -0
  135. data/lib/ruby_llm/models/schema.rb +99 -0
  136. data/lib/ruby_llm/models.json +73261 -33733
  137. data/lib/ruby_llm/models.rb +477 -215
  138. data/lib/ruby_llm/moderation.rb +139 -26
  139. data/lib/ruby_llm/ocr.rb +112 -0
  140. data/lib/ruby_llm/prompt.rb +79 -0
  141. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  142. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  143. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  144. data/lib/ruby_llm/protocol.rb +662 -0
  145. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  146. data/lib/ruby_llm/protocols/anthropic/chat.rb +548 -0
  147. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  148. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  149. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  150. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  151. data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
  152. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  153. data/lib/ruby_llm/protocols/anthropic.rb +101 -0
  154. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  155. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
  156. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  157. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  158. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  159. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  160. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  161. data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
  162. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  163. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  164. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  165. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  166. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  167. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  168. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
  169. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  170. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  171. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  172. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  173. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  174. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  175. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  176. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  177. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  178. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  179. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  180. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  181. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  182. data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
  183. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  184. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  185. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  186. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  187. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  188. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  189. data/lib/ruby_llm/protocols/converse/chat.rb +695 -0
  190. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  191. data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
  192. data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
  193. data/lib/ruby_llm/protocols/converse.rb +54 -0
  194. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  195. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  196. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  197. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  198. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  199. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  200. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  201. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  202. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  203. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  204. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  208. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  209. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  210. data/lib/ruby_llm/protocols/files.rb +119 -0
  211. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  212. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  213. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  214. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
  215. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  216. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  217. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  218. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  219. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  220. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  221. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  222. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  223. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  224. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  225. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  226. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  227. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  228. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  229. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  230. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  231. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  232. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  233. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  234. data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
  235. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  236. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  237. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  238. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  239. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  240. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  241. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  242. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  243. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  244. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  245. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  246. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  247. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  248. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  249. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  250. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  251. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  252. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  253. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  254. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  255. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  256. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  257. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  258. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  259. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  260. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  261. data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
  262. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  263. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  264. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  265. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  266. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  267. data/lib/ruby_llm/protocols/responses.rb +35 -0
  268. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  269. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  270. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  271. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  272. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  273. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  274. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  275. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  276. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  277. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  278. data/lib/ruby_llm/provider.rb +560 -128
  279. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  280. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  281. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  282. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  283. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  284. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  285. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  286. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  287. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  288. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  289. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  290. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  291. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  292. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  293. data/lib/ruby_llm/providers/azure.rb +77 -78
  294. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  295. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  296. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  297. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  298. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  299. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  300. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  301. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  302. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  303. data/lib/ruby_llm/providers/cohere.rb +31 -0
  304. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  305. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  306. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  307. data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
  308. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  309. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  310. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  311. data/lib/ruby_llm/providers/gemini.rb +15 -8
  312. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  313. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  314. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  315. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  316. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  317. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  318. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  319. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  320. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  321. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  322. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  323. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  324. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  325. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  326. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  327. data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
  328. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  329. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  330. data/lib/ruby_llm/providers/mistral.rb +16 -4
  331. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  332. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  333. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  334. data/lib/ruby_llm/providers/ollama.rb +9 -8
  335. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  336. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  337. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  338. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  339. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  340. data/lib/ruby_llm/providers/openai.rb +92 -11
  341. data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
  342. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  343. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  344. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  345. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  346. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  347. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  348. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  349. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  350. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  351. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  352. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  353. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  354. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  355. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  356. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  357. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  358. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  359. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  360. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  361. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  362. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  363. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  364. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  365. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  366. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  367. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  368. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  369. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  370. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  371. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  372. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  373. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  374. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  375. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  376. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  377. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  378. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  379. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  380. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  381. data/lib/ruby_llm/providers/xai.rb +15 -5
  382. data/lib/ruby_llm/railtie.rb +7 -16
  383. data/lib/ruby_llm/rerank.rb +105 -0
  384. data/lib/ruby_llm/research_job.rb +241 -0
  385. data/lib/ruby_llm/search_results.rb +68 -0
  386. data/lib/ruby_llm/server_tool_call.rb +73 -0
  387. data/lib/ruby_llm/speech.rb +159 -0
  388. data/lib/ruby_llm/speech_chunk.rb +33 -0
  389. data/lib/ruby_llm/support/deprecator.rb +22 -0
  390. data/lib/ruby_llm/support/inspectable.rb +49 -0
  391. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  392. data/lib/ruby_llm/support/utils.rb +147 -0
  393. data/lib/ruby_llm/thinking.rb +127 -20
  394. data/lib/ruby_llm/tokenization.rb +59 -0
  395. data/lib/ruby_llm/tokens.rb +103 -33
  396. data/lib/ruby_llm/tool.rb +266 -91
  397. data/lib/ruby_llm/tool_call.rb +36 -3
  398. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  399. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  400. data/lib/ruby_llm/transcription.rb +138 -13
  401. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  402. data/lib/ruby_llm/transport/connection.rb +193 -0
  403. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  404. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  405. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  406. data/lib/ruby_llm/uploaded_file.rb +144 -0
  407. data/lib/ruby_llm/version.rb +2 -1
  408. data/lib/ruby_llm/video.rb +136 -0
  409. data/lib/ruby_llm/video_job.rb +150 -0
  410. data/lib/ruby_llm/workflow.rb +91 -0
  411. data/lib/ruby_llm.rb +380 -6
  412. data/lib/tasks/ruby_llm.rake +21 -16
  413. data/skills/rubyllm/SKILL.md +81 -0
  414. data/skills/rubyllm/agents/openai.yaml +4 -0
  415. metadata +339 -97
  416. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  417. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  418. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  419. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  420. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  421. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  422. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  423. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  424. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  425. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  426. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  427. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  428. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  429. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  430. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  431. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  432. data/lib/ruby_llm/aliases.rb +0 -41
  433. data/lib/ruby_llm/connection.rb +0 -159
  434. data/lib/ruby_llm/content.rb +0 -91
  435. data/lib/ruby_llm/deprecator.rb +0 -24
  436. data/lib/ruby_llm/error_middleware.rb +0 -81
  437. data/lib/ruby_llm/instrumentation.rb +0 -36
  438. data/lib/ruby_llm/mime_type.rb +0 -96
  439. data/lib/ruby_llm/model/info.rb +0 -164
  440. data/lib/ruby_llm/model_registry.rb +0 -39
  441. data/lib/ruby_llm/models_schema.json +0 -171
  442. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  443. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  444. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  445. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  446. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  447. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  448. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  449. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  450. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  451. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  452. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  453. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  454. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  455. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  456. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  457. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  458. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  459. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  460. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  461. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  462. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  463. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  464. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  465. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  466. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  467. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  468. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  469. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  470. data/lib/ruby_llm/streaming.rb +0 -179
  471. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  472. data/lib/ruby_llm/utils.rb +0 -130
  473. data/lib/tasks/models.rake +0 -593
  474. data/lib/tasks/release.rake +0 -94
  475. data/lib/tasks/vcr.rake +0 -124
@@ -0,0 +1,34 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class VertexAI < Provider
6
+ # The Anthropic protocol over Vertex AI rawPredict endpoints.
7
+ class Anthropic < Protocols::Anthropic
8
+ ANTHROPIC_VERSION = 'vertex-2023-10-16'
9
+
10
+ def completion_url
11
+ "#{@provider.model_path(@model.id, publisher: 'anthropic')}:rawPredict"
12
+ end
13
+
14
+ def stream_url
15
+ "#{@provider.model_path(@model.id, publisher: 'anthropic')}:streamRawPredict"
16
+ end
17
+
18
+ def render_payload(messages, **)
19
+ payload = super
20
+ payload.delete(:model)
21
+ payload.merge(anthropic_version: ANTHROPIC_VERSION)
22
+ end
23
+
24
+ def count_tokens(*, **)
25
+ raise Error, "#{@provider.name} doesn't support token counting for Claude models"
26
+ end
27
+
28
+ def supports_provider_file_references?
29
+ false
30
+ end
31
+ end
32
+ end
33
+ end
34
+ end
@@ -0,0 +1,19 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class VertexAI
6
+ # Feature capability gaps not represented in upstream model catalogs.
7
+ module Capabilities
8
+ def self.augment(capabilities, model_id:, modalities:)
9
+ return capabilities if model_id.include?('embedding')
10
+
11
+ additions = []
12
+ additions << 'tool_choice' if model_id == 'gemini-2.5-flash'
13
+ additions << 'transcription' if modalities[:input].include?('audio') && modalities[:output].include?('text')
14
+ capabilities | additions
15
+ end
16
+ end
17
+ end
18
+ end
19
+ end
@@ -0,0 +1,54 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class VertexAI < Provider
6
+ class ChatCompletions
7
+ # Vertex AI MaaS batch prediction rows using OpenAI JSONL shape.
8
+ module Batches
9
+ include Protocols::VertexAI::BatchPrediction
10
+
11
+ private
12
+
13
+ def vertex_batch_request(request)
14
+ {
15
+ custom_id: request[:custom_id],
16
+ method: 'POST',
17
+ url: '/v1/chat/completions',
18
+ body: batch_payload(request)
19
+ }
20
+ end
21
+
22
+ def validate_batch_requests!(requests)
23
+ return if requests.all? { |request| chat_completion_payload?(request.fetch(:payload)) }
24
+
25
+ raise Error, 'vertexai MaaS batch requests require chat completion payloads'
26
+ end
27
+
28
+ def chat_completion_payload?(payload)
29
+ payload.key?(:messages) || payload.key?('messages')
30
+ end
31
+
32
+ def vertex_batch_model_path(model)
33
+ publisher, name = model.split('/', 2)
34
+ raise Error, 'vertexai MaaS batch requests require publisher/model ids' unless publisher && name
35
+
36
+ @provider.model_path(name, publisher:)
37
+ end
38
+
39
+ def parse_vertex_batch_result(line, fallback_index)
40
+ index = vertex_batch_result_index(line, fallback_index)
41
+ response = line['response']
42
+ body = response.is_a?(Hash) ? response['body'] || response : response
43
+
44
+ if body
45
+ [index, parse_completion_body(body, raw: body)]
46
+ else
47
+ [index, nil, batch_failure(index, line.dig('status', 'message') || batch_error_message(line))]
48
+ end
49
+ end
50
+ end
51
+ end
52
+ end
53
+ end
54
+ end
@@ -0,0 +1,15 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class VertexAI < Provider
6
+ # MaaS models (publisher-prefixed ids like meta/llama-3.3-70b-instruct-maas)
7
+ # speak Chat Completions through Vertex AI's OpenAI-compatible endpoint.
8
+ class ChatCompletions < Protocols::ChatCompletions
9
+ def completion_url
10
+ "#{@provider.location_path}/endpoints/openapi/chat/completions"
11
+ end
12
+ end
13
+ end
14
+ end
15
+ end
@@ -0,0 +1,42 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class VertexAI
6
+ # Gemini Embedding 2 over Vertex AI's single-content embedding endpoint.
7
+ class EmbedContent < Protocols::Gemini
8
+ def embedding_url(model:)
9
+ "#{@provider.model_path(model)}:embedContent"
10
+ end
11
+
12
+ def render_embedding(text, **options)
13
+ return super unless text.is_a?(Array)
14
+
15
+ { requests: text.map { |value| render_embedding_payload(value, **options) } }
16
+ end
17
+
18
+ def render_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil, with: [],
19
+ provider_options: {})
20
+ if text.is_a?(Array) && text.size != 1
21
+ raise ArgumentError, 'Vertex AI embedContent accepts one text at a time'
22
+ end
23
+ raise ArgumentError, "#{model} takes task instructions and titles in the text" if task_type || title
24
+
25
+ payload = {
26
+ content: { parts: Protocols::Gemini::Media.format_content(text.is_a?(Array) ? text.first : text, with) },
27
+ outputDimensionality: dimensions
28
+ }.compact
29
+ Support::Utils.deep_merge(payload, provider_options)
30
+ end
31
+
32
+ def parse_embedding_response(response, model:, text:)
33
+ vectors = response.body.dig('embedding', 'values')
34
+ raise Error.new('Vertex AI returned no embedding', response:) if vectors.nil? || vectors.empty?
35
+
36
+ vectors = [vectors] if text.is_a?(Array)
37
+ Embedding.new(vectors:, model:, input_tokens: response.body.dig('usageMetadata', 'promptTokenCount'))
38
+ end
39
+ end
40
+ end
41
+ end
42
+ end
@@ -8,23 +8,38 @@ module RubyLLM
8
8
  module_function
9
9
 
10
10
  def embedding_url(model:)
11
- "projects/#{@config.vertexai_project_id}/locations/#{@config.vertexai_location}/publishers/google/models/#{model}:predict" # rubocop:disable Layout/LineLength
11
+ "#{@provider.model_path(model)}:predict"
12
12
  end
13
13
 
14
- def render_embedding_payload(text, model:, dimensions:) # rubocop:disable Lint/UnusedMethodArgument
15
- {
16
- instances: [text].flatten.map { |t| { content: t.to_s } }
17
- }.tap do |payload|
18
- payload[:parameters] = { outputDimensionality: dimensions } if dimensions
14
+ def supports_embedding_media?
15
+ false
16
+ end
17
+
18
+ # rubocop:disable-next Lint/UnusedMethodArgument
19
+ def render_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil, provider_options: {})
20
+ instances = [text].flatten.map do |t|
21
+ { content: t.to_s, task_type: task_type, title: title }.compact
19
22
  end
23
+ payload = { instances: instances }
24
+ payload[:parameters] = { outputDimensionality: dimensions } if dimensions
25
+
26
+ Support::Utils.deep_merge(payload, provider_options)
20
27
  end
21
28
 
22
29
  def parse_embedding_response(response, model:, text:)
23
30
  predictions = response.body['predictions']
24
31
  vectors = predictions&.map { |p| p.dig('embeddings', 'values') }
32
+ input_tokens = embedding_input_tokens(predictions)
25
33
  vectors = vectors.first if vectors&.length == 1 && !text.is_a?(Array)
26
34
 
27
- Embedding.new(vectors:, model:, input_tokens: 0)
35
+ Embedding.new(vectors:, model:, input_tokens:)
36
+ end
37
+
38
+ def embedding_input_tokens(predictions)
39
+ counts = Array(predictions).filter_map do |prediction|
40
+ prediction.dig('embeddings', 'statistics', 'token_count')
41
+ end
42
+ counts.sum unless counts.empty?
28
43
  end
29
44
  end
30
45
  end
@@ -0,0 +1,42 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class VertexAI < Provider
6
+ class Gemini
7
+ # Vertex AI Gemini batch prediction rows.
8
+ module Batches
9
+ include Protocols::VertexAI::BatchPrediction
10
+
11
+ private
12
+
13
+ def vertex_batch_request(request)
14
+ payload = RubyLLM::Support::Utils.deep_stringify_keys(batch_payload(request))
15
+ labels = payload.fetch('labels', {}).merge('ruby_llm_batch_id' => request[:custom_id])
16
+ { request: payload.merge('labels' => labels) }
17
+ end
18
+
19
+ def parse_vertex_batch_result(line, fallback_index)
20
+ index = vertex_batch_result_index(line, fallback_index)
21
+
22
+ if line['response']
23
+ body = line['response']
24
+ [index, parse_completion_body(body, raw: body)]
25
+ else
26
+ [index, nil, batch_failure(index, vertex_batch_status_message(line))]
27
+ end
28
+ end
29
+
30
+ # Gemini prediction rows carry status as a plain string, empty on
31
+ # success and the error text on failure.
32
+ def vertex_batch_status_message(line)
33
+ status = line['status']
34
+ message = status.is_a?(Hash) ? status['message'] : status
35
+
36
+ message.to_s.empty? ? batch_error_message(line) : message
37
+ end
38
+ end
39
+ end
40
+ end
41
+ end
42
+ end
@@ -0,0 +1,69 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class VertexAI < Provider
6
+ # The Gemini protocol over Vertex AI endpoints.
7
+ class Gemini < Protocols::Gemini
8
+ include VertexAI::Embeddings
9
+ include VertexAI::Models
10
+ include VertexAI::Videos
11
+
12
+ SERVER_TOOL_ALIASES = Protocols::Gemini::SERVER_TOOL_ALIASES.merge(
13
+ file_search: lambda { |options|
14
+ { tool: { retrieval: { vertexAiSearch: Support::Utils.deep_symbolize_keys(options) } } }
15
+ }
16
+ ).freeze
17
+
18
+ def server_tool_aliases
19
+ SERVER_TOOL_ALIASES
20
+ end
21
+
22
+ def completion_url
23
+ "#{@provider.model_path(@model.id)}:generateContent"
24
+ end
25
+
26
+ def stream_url
27
+ "#{@provider.model_path(@model.id)}:streamGenerateContent?alt=sse"
28
+ end
29
+
30
+ def count_tokens_url
31
+ "#{@provider.model_path(@model.id)}:countTokens"
32
+ end
33
+
34
+ def render_count_tokens_payload(messages, **options)
35
+ count_tokens_request(messages, **options)
36
+ end
37
+
38
+ def caches_url
39
+ "#{@provider.location_path}/cachedContents"
40
+ end
41
+
42
+ def cache_name(name)
43
+ name = name.name if name.is_a?(CachedContent)
44
+ name.to_s.include?('/') ? name.to_s : "#{caches_url}/#{name}"
45
+ end
46
+
47
+ def cache_model_name(model_id)
48
+ @provider.model_path(model_id)
49
+ end
50
+
51
+ def images_url(with: nil, mask: nil) # rubocop:disable Lint/UnusedMethodArgument
52
+ id = model_id(@model)
53
+
54
+ "#{@provider.model_path(id)}:#{image_endpoint_action(id)}"
55
+ end
56
+
57
+ def speech_url(model:)
58
+ "#{@provider.model_path(model)}:generateContent"
59
+ end
60
+
61
+ private
62
+
63
+ def transcription_url(model)
64
+ "#{@provider.model_path(model)}:generateContent"
65
+ end
66
+ end
67
+ end
68
+ end
69
+ end
@@ -0,0 +1,24 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class VertexAI
6
+ class LiveTranscription < Protocols::Gemini::LiveTranscription # :nodoc: all
7
+ def validate_transcription_request(...)
8
+ super
9
+ return if @config.vertexai_location == 'global'
10
+
11
+ raise ArgumentError, 'Vertex AI Live transcription requires vertexai_location = "global"'
12
+ end
13
+
14
+ def transcription_model_name(model)
15
+ @provider.model_path(model)
16
+ end
17
+
18
+ def websocket_service
19
+ 'google.cloud.aiplatform.v1beta1.LlmBidiService/BidiGenerateContent'
20
+ end
21
+ end
22
+ end
23
+ end
24
+ end
@@ -0,0 +1,28 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class VertexAI < Provider
6
+ # Mistral's models speak their own dialect of Chat Completions over Vertex AI
7
+ # rawPredict endpoints. We reuse Mistral's dialect wholesale and only swap the URLs.
8
+ class Mistral < Protocols::ChatCompletions
9
+ include Providers::Mistral::Chat
10
+ include Providers::Mistral::Embeddings
11
+ include Providers::Mistral::Media
12
+ include Providers::Mistral::Models
13
+
14
+ # The Mistral model families Vertex AI serves directly. Shared by the
15
+ # registry (which models to list) and protocol_for (where to route them).
16
+ MODELS = /\A(mistral|ministral|codestral|magistral|mathstral|pixtral|devstral|voxtral)/
17
+
18
+ def completion_url
19
+ "#{@provider.model_path(@model.id, publisher: 'mistralai')}:rawPredict"
20
+ end
21
+
22
+ def stream_url
23
+ "#{@provider.model_path(@model.id, publisher: 'mistralai')}:streamRawPredict"
24
+ end
25
+ end
26
+ end
27
+ end
28
+ end
@@ -5,75 +5,171 @@ module RubyLLM
5
5
  class VertexAI
6
6
  # Models methods for the Vertex AI integration
7
7
  module Models
8
- # Gemini and other Google models that aren't returned by the API
8
+ def self.models_dev_alias(model_id, models_dev_by_key, _provider_model = nil)
9
+ source = models_dev_by_key["gemini:#{model_id}"]
10
+ Model.new(source.to_h.merge(provider: 'vertexai')) if source
11
+ end
12
+
13
+ # Google models the publisher catalog omits in some regions while still
14
+ # serving them there. Every id must be callable in at least one region;
15
+ # ids that answer nowhere do not belong here.
9
16
  KNOWN_GOOGLE_MODELS = %w[
10
17
  gemini-2.5-flash-lite
11
18
  gemini-2.5-pro
12
19
  gemini-2.5-flash
13
20
  gemini-2.0-flash-lite-001
14
21
  gemini-2.0-flash-001
15
- gemini-2.0-flash
16
- gemini-2.0-flash-exp
17
22
  gemini-1.5-pro-002
18
23
  gemini-1.5-pro
19
- gemini-1.5-flash-002
20
- gemini-1.5-flash
21
- gemini-1.5-flash-8b
22
24
  gemini-pro
23
25
  gemini-pro-vision
24
- gemini-exp-1206
25
- gemini-exp-1121
26
26
  gemini-embedding-001
27
27
  text-embedding-005
28
28
  text-embedding-004
29
29
  text-multilingual-embedding-002
30
30
  ].freeze
31
31
 
32
+ # Every publisher with models Vertex AI serves as a service. The rest
33
+ # of the Model Garden is deploy-it-yourself and not callable directly.
34
+ PUBLISHERS = %w[google anthropic mistralai meta deepseek-ai qwen openai moonshotai zai-org].freeze
35
+
36
+ # Vertex AI serves a different slice of the catalog in each location and
37
+ # neither of these is a superset of the other, so a listing unions them.
38
+ CATALOG_LOCATIONS = %w[global us-central1].freeze
39
+
40
+ # The configured location has to answer: without it we would report a
41
+ # catalog the caller cannot reach. A supplementary location is a bonus,
42
+ # so a failure there is a warning and the rest of the union stands.
32
43
  def list_models
33
- all_models = []
34
- page_token = nil
44
+ fetched = []
45
+ counts = {}
46
+
47
+ catalog_connections.each do |location, connection|
48
+ models = location_models(connection)
49
+ counts[location] = models.size
50
+ fetched.concat(models)
51
+ rescue StandardError => e
52
+ raise if location == configured_location
53
+
54
+ RubyLLM.logger.warn "Skipping the Vertex AI catalog at #{location}: #{e.class}: #{e.message}"
55
+ end
56
+
57
+ models = fetched.uniq(&:id)
58
+ log_catalog(counts, models)
59
+
60
+ models + build_known_models(models.map(&:id))
61
+ end
62
+
63
+ private
64
+
65
+ def configured_location
66
+ @config.vertexai_location.to_s
67
+ end
68
+
69
+ # The location lives in the host, so locations sharing one api base,
70
+ # as they do behind a custom vertexai_api_base, are one catalog.
71
+ def catalog_connections
72
+ [configured_location, *CATALOG_LOCATIONS]
73
+ .uniq { |location| @provider.api_base_for(location) }
74
+ .to_h { |location| [location, connection_for(location)] }
75
+ end
76
+
77
+ def connection_for(location)
78
+ return @connection if location == configured_location
79
+
80
+ Transport::Connection.new(@provider, @config, api_base: @provider.api_base_for(location))
81
+ end
82
+
83
+ def log_catalog(counts, models)
84
+ per_location = counts.map { |location, count| "#{location} (#{count})" }.join(', ')
85
+ RubyLLM.logger.info "Fetched the Vertex AI catalog from #{per_location}: #{models.size} models"
86
+ end
87
+
88
+ # A publisher with nothing to offer in a region answers 200 with an
89
+ # empty list, so any error here is infrastructure, not an empty
90
+ # catalog. Reporting a partial catalog as a success would drop the
91
+ # missing publishers from the registry.
92
+ def location_models(connection)
93
+ failures = []
94
+ models = PUBLISHERS.flat_map do |publisher|
95
+ publisher_models(publisher, connection)
96
+ rescue StandardError => e
97
+ failures << "#{publisher} (#{e.class}: #{e.message})"
98
+ []
99
+ end
100
+
101
+ raise Error, "Could not fetch the Vertex AI catalog for #{failures.join(', ')}" if failures.any?
102
+
103
+ models
104
+ end
105
+
106
+ def publisher_models(publisher, connection)
107
+ catalog(publisher, connection).filter_map { |model_data| build_publisher_model(publisher, model_data) }
108
+ end
35
109
 
36
- all_models.concat(build_known_models)
110
+ # MaaS models are called as publisher/name through the OpenAI-compatible
111
+ # endpoint; directly served models by their bare catalog name.
112
+ def build_publisher_model(publisher, model_data)
113
+ name = model_data['name'].split('/').last
114
+ return if deployable?(model_data)
115
+
116
+ if name.end_with?('-maas')
117
+ build_model_from_api_data(model_data, "#{publisher}/#{name}")
118
+ elsif served_directly?(publisher, name)
119
+ build_model_from_api_data(model_data, name)
120
+ end
121
+ end
122
+
123
+ # Deploy-it-yourself Model Garden cards expose deploy actions; managed
124
+ # services Vertex AI serves on our behalf never do.
125
+ def deployable?(model_data)
126
+ actions = model_data['supportedActions'] || {}
127
+ actions.key?('deploy') || actions.key?('multiDeployVertex') || actions.key?('deployGke')
128
+ end
129
+
130
+ # Among the managed models, which publishers we route by bare name, and
131
+ # for Google (whose catalog is a grab-bag of vision, media, and AutoML
132
+ # products) which of those names are chat or embedding models.
133
+ def served_directly?(publisher, name)
134
+ case publisher
135
+ when 'google' then name.match?(/\Agemini|embedding/)
136
+ when 'anthropic' then true
137
+ when 'mistralai' then VertexAI::Mistral::MODELS.match?(name)
138
+ else false
139
+ end
140
+ end
141
+
142
+ def catalog(publisher, connection)
143
+ models = []
144
+ page_token = nil
37
145
 
38
146
  loop do
39
- response = @connection.get('publishers/google/models') do |req|
147
+ response = connection.get("publishers/#{publisher}/models") do |req|
40
148
  req.headers['x-goog-user-project'] = @config.vertexai_project_id
41
149
  req.params = { pageSize: 100 }
42
150
  req.params[:pageToken] = page_token if page_token
43
151
  end
44
152
 
45
- publisher_models = response.body['publisherModels'] || []
46
- publisher_models.each do |model_data|
47
- next if model_data['launchStage'] == 'DEPRECATED'
48
-
49
- model_id = extract_model_id_from_path(model_data['name'])
50
- all_models << build_model_from_api_data(model_data, model_id)
51
- end
52
-
153
+ models.concat(response.body['publisherModels'] || [])
53
154
  page_token = response.body['nextPageToken']
54
155
  break unless page_token
55
156
  end
56
157
 
57
- all_models
58
- rescue StandardError => e
59
- RubyLLM.logger.debug { "Error fetching Vertex AI models: #{e.message}" }
60
- build_known_models
158
+ models.reject { |model_data| model_data['launchStage'] == 'DEPRECATED' }
61
159
  end
62
160
 
63
- private
64
-
65
- def build_known_models
66
- KNOWN_GOOGLE_MODELS.map do |model_id|
67
- Model::Info.new(
161
+ def build_known_models(fetched_ids)
162
+ (KNOWN_GOOGLE_MODELS - fetched_ids).map do |model_id|
163
+ Model.new(
68
164
  id: model_id,
69
165
  name: model_id,
70
- provider: slug,
166
+ provider: @provider.slug,
71
167
  family: determine_model_family(model_id),
72
168
  created_at: nil,
73
169
  context_window: nil,
74
170
  max_output_tokens: nil,
75
171
  modalities: nil,
76
- capabilities: %w[streaming function_calling],
172
+ capabilities: extract_capabilities(model_id),
77
173
  pricing: nil,
78
174
  metadata: {
79
175
  source: 'known_models'
@@ -83,16 +179,16 @@ module RubyLLM
83
179
  end
84
180
 
85
181
  def build_model_from_api_data(model_data, model_id)
86
- Model::Info.new(
182
+ Model.new(
87
183
  id: model_id,
88
184
  name: model_id,
89
- provider: slug,
185
+ provider: @provider.slug,
90
186
  family: determine_model_family(model_id),
91
187
  created_at: nil,
92
188
  context_window: nil,
93
189
  max_output_tokens: nil,
94
190
  modalities: nil,
95
- capabilities: extract_capabilities(model_data),
191
+ capabilities: extract_capabilities(model_data['name']),
96
192
  pricing: nil,
97
193
  metadata: {
98
194
  version_id: model_data['versionId'],
@@ -104,12 +200,21 @@ module RubyLLM
104
200
  )
105
201
  end
106
202
 
107
- def extract_model_id_from_path(path)
108
- path.split('/').last
109
- end
110
-
111
203
  def determine_model_family(model_id)
112
204
  case model_id
205
+ when /^claude.*haiku/ then 'claude-haiku'
206
+ when /^claude.*sonnet/ then 'claude-sonnet'
207
+ when /^claude.*opus/ then 'claude-opus'
208
+ when /^claude/ then 'claude'
209
+ when %r{^meta/} then 'llama'
210
+ when %r{^deepseek-ai/} then 'deepseek'
211
+ when %r{^qwen/} then 'qwen'
212
+ when %r{^moonshotai/} then 'kimi'
213
+ when %r{^zai-org/} then 'glm'
214
+ when %r{^openai/} then 'gpt-oss'
215
+ when %r{^google/} then 'gemma'
216
+ when /^codestral/ then 'codestral'
217
+ when /^mi(ni)?stral/ then 'mistral'
113
218
  when /^gemini-2\.\d+/ then 'gemini-2'
114
219
  when /^gemini-1\.\d+/ then 'gemini-1.5'
115
220
  when /^text-embedding/ then 'text-embedding'
@@ -118,11 +223,8 @@ module RubyLLM
118
223
  end
119
224
  end
120
225
 
121
- def extract_capabilities(model_data)
122
- capabilities = ['streaming']
123
- model_name = model_data['name']
124
- capabilities << 'function_calling' if model_name.include?('gemini')
125
- capabilities.uniq
226
+ def extract_capabilities(name)
227
+ name.match?(/ocr|embedding/) ? %w[streaming] : %w[streaming function_calling]
126
228
  end
127
229
  end
128
230
  end