ruby_llm 1.16.0 → 2.0.0.rc2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (475) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +85 -32
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
  91. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
  92. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
  93. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
  94. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  96. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
  97. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
  98. data/lib/ruby_llm/accounting/usage.rb +245 -0
  99. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  100. data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
  101. data/lib/ruby_llm/active_record/batch.rb +97 -0
  102. data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
  103. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  104. data/lib/ruby_llm/active_record/model.rb +135 -0
  105. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  106. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  107. data/lib/ruby_llm/active_record/usage.rb +61 -0
  108. data/lib/ruby_llm/agent.rb +1065 -151
  109. data/lib/ruby_llm/aliases.json +268 -100
  110. data/lib/ruby_llm/attachment.rb +187 -48
  111. data/lib/ruby_llm/batch.rb +432 -0
  112. data/lib/ruby_llm/cached_content.rb +112 -0
  113. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  114. data/lib/ruby_llm/chat.rb +1127 -198
  115. data/lib/ruby_llm/chunk.rb +10 -0
  116. data/lib/ruby_llm/citation.rb +105 -0
  117. data/lib/ruby_llm/configuration.rb +261 -24
  118. data/lib/ruby_llm/context.rb +128 -6
  119. data/lib/ruby_llm/cost.rb +217 -80
  120. data/lib/ruby_llm/downloaded_file.rb +33 -0
  121. data/lib/ruby_llm/embedding.rb +121 -17
  122. data/lib/ruby_llm/embedding_request.rb +53 -0
  123. data/lib/ruby_llm/error.rb +159 -23
  124. data/lib/ruby_llm/fallback.rb +133 -0
  125. data/lib/ruby_llm/files/mime_type.rb +97 -0
  126. data/lib/ruby_llm/image.rb +153 -32
  127. data/lib/ruby_llm/message.rb +233 -54
  128. data/lib/ruby_llm/model/modalities.rb +17 -4
  129. data/lib/ruby_llm/model/pricing.rb +24 -5
  130. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  131. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  132. data/lib/ruby_llm/model.rb +244 -2
  133. data/lib/ruby_llm/models/aliases.rb +41 -0
  134. data/lib/ruby_llm/models/registry.rb +165 -0
  135. data/lib/ruby_llm/models/schema.rb +99 -0
  136. data/lib/ruby_llm/models.json +73261 -33733
  137. data/lib/ruby_llm/models.rb +477 -215
  138. data/lib/ruby_llm/moderation.rb +139 -26
  139. data/lib/ruby_llm/ocr.rb +112 -0
  140. data/lib/ruby_llm/prompt.rb +79 -0
  141. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  142. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  143. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  144. data/lib/ruby_llm/protocol.rb +662 -0
  145. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  146. data/lib/ruby_llm/protocols/anthropic/chat.rb +548 -0
  147. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  148. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  149. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  150. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  151. data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
  152. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  153. data/lib/ruby_llm/protocols/anthropic.rb +101 -0
  154. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  155. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
  156. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  157. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  158. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  159. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  160. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  161. data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
  162. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  163. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  164. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  165. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  166. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  167. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  168. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
  169. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  170. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  171. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  172. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  173. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  174. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  175. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  176. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  177. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  178. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  179. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  180. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  181. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  182. data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
  183. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  184. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  185. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  186. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  187. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  188. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  189. data/lib/ruby_llm/protocols/converse/chat.rb +695 -0
  190. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  191. data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
  192. data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
  193. data/lib/ruby_llm/protocols/converse.rb +54 -0
  194. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  195. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  196. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  197. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  198. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  199. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  200. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  201. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  202. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  203. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  204. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  208. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  209. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  210. data/lib/ruby_llm/protocols/files.rb +119 -0
  211. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  212. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  213. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  214. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
  215. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  216. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  217. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  218. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  219. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  220. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  221. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  222. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  223. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  224. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  225. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  226. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  227. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  228. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  229. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  230. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  231. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  232. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  233. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  234. data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
  235. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  236. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  237. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  238. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  239. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  240. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  241. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  242. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  243. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  244. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  245. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  246. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  247. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  248. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  249. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  250. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  251. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  252. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  253. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  254. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  255. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  256. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  257. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  258. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  259. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  260. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  261. data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
  262. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  263. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  264. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  265. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  266. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  267. data/lib/ruby_llm/protocols/responses.rb +35 -0
  268. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  269. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  270. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  271. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  272. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  273. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  274. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  275. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  276. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  277. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  278. data/lib/ruby_llm/provider.rb +560 -128
  279. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  280. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  281. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  282. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  283. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  284. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  285. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  286. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  287. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  288. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  289. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  290. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  291. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  292. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  293. data/lib/ruby_llm/providers/azure.rb +77 -78
  294. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  295. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  296. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  297. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  298. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  299. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  300. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  301. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  302. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  303. data/lib/ruby_llm/providers/cohere.rb +31 -0
  304. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  305. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  306. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  307. data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
  308. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  309. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  310. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  311. data/lib/ruby_llm/providers/gemini.rb +15 -8
  312. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  313. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  314. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  315. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  316. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  317. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  318. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  319. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  320. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  321. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  322. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  323. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  324. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  325. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  326. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  327. data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
  328. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  329. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  330. data/lib/ruby_llm/providers/mistral.rb +16 -4
  331. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  332. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  333. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  334. data/lib/ruby_llm/providers/ollama.rb +9 -8
  335. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  336. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  337. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  338. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  339. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  340. data/lib/ruby_llm/providers/openai.rb +92 -11
  341. data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
  342. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  343. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  344. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  345. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  346. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  347. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  348. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  349. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  350. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  351. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  352. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  353. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  354. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  355. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  356. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  357. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  358. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  359. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  360. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  361. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  362. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  363. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  364. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  365. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  366. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  367. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  368. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  369. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  370. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  371. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  372. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  373. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  374. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  375. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  376. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  377. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  378. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  379. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  380. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  381. data/lib/ruby_llm/providers/xai.rb +15 -5
  382. data/lib/ruby_llm/railtie.rb +7 -16
  383. data/lib/ruby_llm/rerank.rb +105 -0
  384. data/lib/ruby_llm/research_job.rb +241 -0
  385. data/lib/ruby_llm/search_results.rb +68 -0
  386. data/lib/ruby_llm/server_tool_call.rb +73 -0
  387. data/lib/ruby_llm/speech.rb +159 -0
  388. data/lib/ruby_llm/speech_chunk.rb +33 -0
  389. data/lib/ruby_llm/support/deprecator.rb +22 -0
  390. data/lib/ruby_llm/support/inspectable.rb +49 -0
  391. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  392. data/lib/ruby_llm/support/utils.rb +147 -0
  393. data/lib/ruby_llm/thinking.rb +127 -20
  394. data/lib/ruby_llm/tokenization.rb +59 -0
  395. data/lib/ruby_llm/tokens.rb +103 -33
  396. data/lib/ruby_llm/tool.rb +266 -91
  397. data/lib/ruby_llm/tool_call.rb +36 -3
  398. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  399. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  400. data/lib/ruby_llm/transcription.rb +138 -13
  401. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  402. data/lib/ruby_llm/transport/connection.rb +193 -0
  403. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  404. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  405. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  406. data/lib/ruby_llm/uploaded_file.rb +144 -0
  407. data/lib/ruby_llm/version.rb +2 -1
  408. data/lib/ruby_llm/video.rb +136 -0
  409. data/lib/ruby_llm/video_job.rb +150 -0
  410. data/lib/ruby_llm/workflow.rb +91 -0
  411. data/lib/ruby_llm.rb +380 -6
  412. data/lib/tasks/ruby_llm.rake +21 -16
  413. data/skills/rubyllm/SKILL.md +81 -0
  414. data/skills/rubyllm/agents/openai.yaml +4 -0
  415. metadata +339 -97
  416. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  417. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  418. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  419. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  420. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  421. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  422. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  423. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  424. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  425. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  426. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  427. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  428. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  429. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  430. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  431. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  432. data/lib/ruby_llm/aliases.rb +0 -41
  433. data/lib/ruby_llm/connection.rb +0 -159
  434. data/lib/ruby_llm/content.rb +0 -91
  435. data/lib/ruby_llm/deprecator.rb +0 -24
  436. data/lib/ruby_llm/error_middleware.rb +0 -81
  437. data/lib/ruby_llm/instrumentation.rb +0 -36
  438. data/lib/ruby_llm/mime_type.rb +0 -96
  439. data/lib/ruby_llm/model/info.rb +0 -164
  440. data/lib/ruby_llm/model_registry.rb +0 -39
  441. data/lib/ruby_llm/models_schema.json +0 -171
  442. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  443. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  444. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  445. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  446. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  447. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  448. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  449. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  450. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  451. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  452. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  453. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  454. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  455. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  456. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  457. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  458. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  459. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  460. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  461. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  462. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  463. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  464. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  465. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  466. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  467. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  468. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  469. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  470. data/lib/ruby_llm/streaming.rb +0 -179
  471. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  472. data/lib/ruby_llm/utils.rb +0 -130
  473. data/lib/tasks/models.rake +0 -593
  474. data/lib/tasks/release.rake +0 -94
  475. data/lib/tasks/vcr.rake +0 -124
@@ -0,0 +1,26 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Responses
6
+ # The Responses input-token counting endpoint.
7
+ module TokenCounting
8
+ COUNT_TOKENS_KEYS = %i[model input instructions tools tool_choice parallel_tool_calls reasoning text].freeze
9
+
10
+ module_function
11
+
12
+ def count_tokens_url
13
+ "#{completion_url}/input_tokens"
14
+ end
15
+
16
+ def render_count_tokens_payload(messages, model:, **options)
17
+ render_payload(messages, model: model, temperature: nil, **options).slice(*COUNT_TOKENS_KEYS)
18
+ end
19
+
20
+ def parse_count_tokens_response(response)
21
+ response.body.fetch('input_tokens')
22
+ end
23
+ end
24
+ end
25
+ end
26
+ end
@@ -0,0 +1,39 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Responses
6
+ # Tools methods of the OpenAI Responses API. Function definitions are
7
+ # flat rather than nested under a `function` key.
8
+ module Tools
9
+ module_function
10
+
11
+ def tool_for(tool)
12
+ definition = {
13
+ type: 'function',
14
+ name: tool.name,
15
+ description: tool.description,
16
+ parameters: ChatCompletions::Tools.parameters_schema_for(tool),
17
+ strict: false
18
+ }
19
+
20
+ return definition if tool.provider_options.empty?
21
+
22
+ RubyLLM::Support::Utils.deep_merge(definition, tool.provider_options)
23
+ end
24
+
25
+ def build_tool_choice(tool_choice)
26
+ case tool_choice
27
+ when :auto, :none, :required
28
+ tool_choice
29
+ else
30
+ {
31
+ type: 'function',
32
+ name: tool_choice
33
+ }
34
+ end
35
+ end
36
+ end
37
+ end
38
+ end
39
+ end
@@ -0,0 +1,35 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ # The OpenAI Responses API. Overrides the chat surface of Chat Completions;
6
+ # embeddings, images, moderation, and transcription are inherited. Runs
7
+ # stateless (store: false) and replays encrypted reasoning so multi-turn
8
+ # tool calls work without server-side state.
9
+ class Responses < ChatCompletions
10
+ include Responses::Approvals
11
+ include Responses::Chat
12
+ include Responses::Media
13
+ include Responses::Streaming
14
+ include Responses::Tools
15
+
16
+ SERVER_TOOL_ALIASES = {
17
+ web_search: { tool: { type: 'web_search' } },
18
+ file_search: { tool: { type: 'file_search' } },
19
+ code_execution: { tool: { type: 'code_interpreter', container: { type: 'auto' } } },
20
+ code_interpreter: { tool: { type: 'code_interpreter', container: { type: 'auto' } } },
21
+ image_generation: { tool: { type: 'image_generation' } },
22
+ mcp: lambda do |options|
23
+ options = Support::Utils.deep_symbolize_keys(options)
24
+ options[:server_url] = options.delete(:url) if options.key?(:url)
25
+ options[:server_label] = options.delete(:name) if options.key?(:name)
26
+ { tool: { type: 'mcp' }.merge(options) }
27
+ end
28
+ }.freeze
29
+
30
+ def server_tool_aliases
31
+ SERVER_TOOL_ALIASES
32
+ end
33
+ end
34
+ end
35
+ end
@@ -0,0 +1,155 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'stringio'
4
+
5
+ module RubyLLM
6
+ module Protocols
7
+ module VertexAI
8
+ # Shared Vertex AI batchPredictionJobs plumbing. The input row and output
9
+ # result shapes belong to each Vertex protocol.
10
+ module BatchPrediction
11
+ include RubyLLM::Batch::Helpers
12
+
13
+ TERMINAL = %w[
14
+ JOB_STATE_SUCCEEDED
15
+ JOB_STATE_FAILED
16
+ JOB_STATE_CANCELLED
17
+ JOB_STATE_EXPIRED
18
+ JOB_STATE_PARTIALLY_SUCCEEDED
19
+ ].freeze
20
+ private_constant :TERMINAL
21
+
22
+ def create_batch(requests)
23
+ model = single_batch_model!(requests, 'vertexai')
24
+ validate_batch_requests!(requests)
25
+ input_uri, output_uri = vertex_batch_storage_uris
26
+ @provider.upload_file(
27
+ StringIO.new(vertex_batch_jsonl(requests)),
28
+ filename: 'input.jsonl',
29
+ uri: input_uri,
30
+ content_type: 'application/jsonl'
31
+ )
32
+
33
+ response = @connection.post("#{@provider.location_path}/batchPredictionJobs",
34
+ vertex_batch_job(model, input_uri, output_uri), idempotent: false)
35
+
36
+ parse_batch_response(response.body)
37
+ end
38
+
39
+ def find_batch(id)
40
+ parse_batch_response @connection.get(vertex_batch_name(id)).body
41
+ end
42
+
43
+ def cancel_batch(id)
44
+ @connection.post("#{vertex_batch_name(id)}:cancel", {})
45
+ find_batch(id)
46
+ end
47
+
48
+ def batch_results(id)
49
+ job = @connection.get(vertex_batch_name(id)).body
50
+ output_uri = vertex_output_uri(job)
51
+ unless output_uri
52
+ status = parse_batch_status(job['state'], completed: TERMINAL.include?(job['state']))
53
+ return [] if %i[failed cancelled].include?(status)
54
+
55
+ raise Error, 'vertexai batch has no GCS output URI yet'
56
+ end
57
+
58
+ rows = @provider.list_file_uris(output_uri).grep(/\.jsonl\z/).flat_map do |uri|
59
+ @provider.download_file(uri).to_s.each_line.filter_map do |line|
60
+ next if line.strip.empty?
61
+
62
+ JSON.parse(line)
63
+ end
64
+ end
65
+ parse_vertex_batch_results(rows, job:)
66
+ end
67
+
68
+ private
69
+
70
+ def vertex_batch_job(model, input_uri, output_uri)
71
+ {
72
+ displayName: "ruby_llm_#{SecureRandom.hex(8)}",
73
+ model: vertex_batch_model_path(model),
74
+ inputConfig: {
75
+ instancesFormat: 'jsonl',
76
+ gcsSource: { uris: [input_uri] }
77
+ },
78
+ outputConfig: {
79
+ predictionsFormat: 'jsonl',
80
+ gcsDestination: { outputUriPrefix: output_uri }
81
+ }
82
+ }
83
+ end
84
+
85
+ def vertex_batch_jsonl(requests)
86
+ requests.map { |request| JSON.generate(vertex_batch_request(request)) }.join("\n")
87
+ end
88
+
89
+ def vertex_batch_request(_request)
90
+ raise NotImplementedError
91
+ end
92
+
93
+ def validate_batch_requests!(_requests); end
94
+
95
+ def vertex_batch_model_path(model)
96
+ @provider.model_path(model)
97
+ end
98
+
99
+ def vertex_batch_storage_uris
100
+ base = @config.vertexai_batch_gcs_uri.to_s.sub(%r{/+\z}, '')
101
+ if base.empty?
102
+ raise ConfigurationError, 'Set vertexai_batch_gcs_uri to a gs:// bucket prefix for Vertex AI batches'
103
+ end
104
+
105
+ prefix = "#{base}/ruby_llm_batches/#{SecureRandom.hex(8)}"
106
+ ["#{prefix}/input.jsonl", "#{prefix}/output"]
107
+ end
108
+
109
+ def vertex_batch_name(id)
110
+ id.to_s.start_with?('projects/') ? id : "#{@provider.location_path}/batchPredictionJobs/#{id}"
111
+ end
112
+
113
+ def parse_batch_response(data)
114
+ state = data['state']
115
+ request_counts = data['completionStats']
116
+
117
+ {
118
+ id: data['name'],
119
+ raw_status: state,
120
+ completed: TERMINAL.include?(state),
121
+ request_counts:,
122
+ request_count: data.dig('labels', 'ruby_llm_request_count')&.to_i || request_counts&.values&.sum(&:to_i),
123
+ model: data['model']
124
+ }
125
+ end
126
+
127
+ def parse_batch_status(raw_status, completed:)
128
+ return :pending unless completed
129
+ return :succeeded if %w[JOB_STATE_SUCCEEDED JOB_STATE_PARTIALLY_SUCCEEDED].include?(raw_status)
130
+ return :cancelled if raw_status == 'JOB_STATE_CANCELLED'
131
+
132
+ :failed
133
+ end
134
+
135
+ def vertex_batch_result_index(line, fallback_index)
136
+ key = line['custom_id'] || line.dig('request', 'labels', 'ruby_llm_batch_id')
137
+ key ? batch_result_index(key) : fallback_index
138
+ end
139
+
140
+ def parse_vertex_batch_results(rows, **)
141
+ rows.each_with_index.filter_map { |line, index| parse_vertex_batch_result(line, index) }
142
+ end
143
+
144
+ def parse_vertex_batch_result(_line, _fallback_index)
145
+ raise NotImplementedError
146
+ end
147
+
148
+ def vertex_output_uri(job)
149
+ job.dig('outputInfo', 'gcsOutputDirectory') ||
150
+ job.dig('outputConfig', 'gcsDestination', 'outputUriPrefix')
151
+ end
152
+ end
153
+ end
154
+ end
155
+ end
@@ -0,0 +1,85 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ module VertexAI
6
+ class EmbeddingPrediction
7
+ # Input identity and model-specific batch embedding request formats.
8
+ module Requests
9
+ private
10
+
11
+ def validate_embedding_request!(request)
12
+ text = request.fetch(:text)
13
+ values = text.is_a?(Array) ? text : [text]
14
+ unless values.any? && values.all? { |value| value.is_a?(String) && !value.empty? }
15
+ raise ArgumentError, 'Vertex AI embedding batches require nonempty text strings'
16
+ end
17
+
18
+ payload = request.fetch(:payload)
19
+ validate_embedding_payload_keys!(payload)
20
+ inputs = embedding_batch_inputs(payload)
21
+ validate_gemini_embedding_options!(payload, inputs) if GEMINI_MODELS.include?(request.fetch(:model))
22
+ return if inputs.size == values.size
23
+
24
+ raise ArgumentError, 'Vertex AI embedding batch payload does not match the number of texts'
25
+ end
26
+
27
+ def validate_embedding_payload_keys!(payload)
28
+ supported = if payload.key?(:instances)
29
+ %i[instances parameters]
30
+ elsif payload.key?(:requests)
31
+ [:requests]
32
+ else
33
+ %i[content outputDimensionality taskType title model]
34
+ end
35
+ unknown = payload.keys - supported
36
+ return if unknown.empty?
37
+
38
+ raise ArgumentError, "Unsupported Vertex AI embedding batch options: #{unknown.join(', ')}"
39
+ end
40
+
41
+ def validate_gemini_embedding_options!(payload, inputs)
42
+ parameters = payload.fetch(:parameters, {}).keys - [:outputDimensionality]
43
+ supported = %i[content task_type taskType title outputDimensionality model]
44
+ unsupported = inputs.flat_map(&:keys).uniq - supported
45
+ return if parameters.empty? && unsupported.empty?
46
+
47
+ raise ArgumentError, 'Vertex AI Gemini embedding batches do not support these options: ' \
48
+ "#{(parameters + unsupported).join(', ')}"
49
+ end
50
+
51
+ def embedding_batch_inputs(payload)
52
+ return payload.fetch(:instances) if payload.key?(:instances)
53
+ return payload.fetch(:requests) if payload.key?(:requests)
54
+ return [payload] if payload.key?(:content)
55
+
56
+ raise ArgumentError, 'Vertex AI embedding batches require embedding request payloads'
57
+ end
58
+
59
+ def render_embedding_batch_rows(request)
60
+ inputs = embedding_batch_inputs(request.fetch(:payload))
61
+ array = request.fetch(:text).is_a?(Array)
62
+ inputs.each_with_index.map do |input, index|
63
+ key = "rllm-#{request.fetch(:custom_id)}-#{index}-#{inputs.size}-#{array ? 'a' : 's'}"
64
+ if LEGACY_MODELS.include?(request.fetch(:model))
65
+ input.merge(key:)
66
+ else
67
+ { key:, request: render_gemini_embedding_request(input, request.fetch(:payload)) }
68
+ end
69
+ end
70
+ end
71
+
72
+ def render_gemini_embedding_request(input, payload)
73
+ content = input[:content]
74
+ content = { parts: [{ text: content }] } if content.is_a?(String)
75
+ config = {
76
+ output_dimensionality: input[:outputDimensionality] || payload.dig(:parameters, :outputDimensionality),
77
+ task_type: input[:task_type] || input[:taskType], title: input[:title]
78
+ }.compact
79
+ { content:, embed_content_config: config }
80
+ end
81
+ end
82
+ end
83
+ end
84
+ end
85
+ end
@@ -0,0 +1,74 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ module VertexAI
6
+ class EmbeddingPrediction
7
+ # Regroups provider output rows into the submitted embedding requests.
8
+ module Results
9
+ private
10
+
11
+ def embedding_row_metadata(row)
12
+ key = row['key'] || row.dig('instance', 'key')
13
+ match = /\Arllm-(\d+)-(\d+)-(\d+)-([as])\z/.match(key.to_s)
14
+ raise Error, "Unknown Vertex AI embedding record key: #{key.inspect}" unless match
15
+
16
+ [Integer(match[1]), Integer(match[2]), Integer(match[3]), match[4] == 'a']
17
+ end
18
+
19
+ def parse_embedding_batch_group(index, rows, model:)
20
+ metadata = rows.map { |row| embedding_row_metadata(row) }
21
+ error = embedding_group_error(rows, metadata)
22
+ return [index, nil, batch_failure(index, error)] if error
23
+ return if rows.size < metadata.first[2]
24
+
25
+ ordered = rows.sort_by { |row| embedding_row_metadata(row)[1] }
26
+ parse_embedding_batch_result(index, ordered, model:, array: metadata.first[3])
27
+ end
28
+
29
+ def parse_embedding_batch_result(index, rows, model:, array:)
30
+ vectors = embedding_batch_vectors(rows)
31
+ return [index, nil, batch_failure(index, 'Vertex AI returned no valid embedding')] unless vectors
32
+
33
+ counts = rows.map { |row| embedding_batch_tokens(row) }
34
+ result = Embedding.new(vectors: array ? vectors : vectors.first,
35
+ model:, input_tokens: (counts.sum if counts.all?))
36
+ [index, result]
37
+ end
38
+
39
+ def embedding_group_error(rows, metadata)
40
+ unless valid_embedding_metadata?(metadata)
41
+ return 'Invalid or duplicate Vertex AI embedding record positions'
42
+ end
43
+
44
+ rows.filter_map { |row| batch_error_value(row['error']) || batch_error_value(row['status']) }
45
+ .find { |error| !error.empty? }
46
+ end
47
+
48
+ def valid_embedding_metadata?(metadata)
49
+ count = metadata.first[2]
50
+ positions = metadata.map { |entry| entry[1] }
51
+ return false unless metadata.first[3] || count == 1
52
+
53
+ metadata.map { |entry| entry[2..] }.uniq.one? && count.positive? &&
54
+ positions.uniq.size == positions.size && positions.max < count
55
+ end
56
+
57
+ def embedding_batch_vectors(rows)
58
+ vectors = rows.map do |row|
59
+ row.dig('response', 'embedding', 'values') || row.dig('predictions', 0, 'embeddings', 'values')
60
+ end
61
+ vectors if vectors.all? { |vector| vector.is_a?(Array) && vector.any? && vector.all?(Numeric) }
62
+ end
63
+
64
+ def embedding_batch_tokens(row)
65
+ value = row.dig('response', 'tokenCount') ||
66
+ row.dig('response', 'usageMetadata', 'promptTokenCount') ||
67
+ row.dig('predictions', 0, 'embeddings', 'statistics', 'token_count')
68
+ Integer(value) unless value.nil?
69
+ end
70
+ end
71
+ end
72
+ end
73
+ end
74
+ end
@@ -0,0 +1,56 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ module VertexAI
6
+ # Text embeddings over Vertex AI's Cloud Storage batch prediction API.
7
+ class EmbeddingPrediction < Protocol
8
+ include BatchPrediction
9
+ include Requests
10
+ include Results
11
+
12
+ LEGACY_MODELS = %w[text-embedding-004 text-embedding-005 text-multilingual-embedding-002].freeze
13
+ GEMINI_MODELS = %w[gemini-embedding-001 gemini-embedding-2].freeze
14
+ MODELS = (LEGACY_MODELS + GEMINI_MODELS).freeze
15
+
16
+ private
17
+
18
+ def validate_batch_requests!(requests)
19
+ @requests = requests
20
+ model = requests.first.fetch(:model)
21
+ unless MODELS.include?(model)
22
+ raise Error,
23
+ "Vertex AI embedding batches are not supported for #{model.inspect}"
24
+ end
25
+
26
+ requests.each { |request| validate_embedding_request!(request) }
27
+ return unless LEGACY_MODELS.include?(model)
28
+ return if requests.map { |request| request.fetch(:payload).fetch(:parameters, {}) }.uniq.one?
29
+
30
+ raise ArgumentError,
31
+ 'Vertex AI legacy embedding batches require the same dimensions and parameters in every request'
32
+ end
33
+
34
+ def vertex_batch_job(model, input_uri, output_uri)
35
+ job = super.merge(labels: { ruby_llm_request_count: @requests.size.to_s })
36
+ return job unless LEGACY_MODELS.include?(model)
37
+
38
+ job.merge(instanceConfig: { instanceType: 'object', keyField: 'key' },
39
+ modelParameters: @requests.first.fetch(:payload).fetch(:parameters, {}))
40
+ end
41
+
42
+ def vertex_batch_jsonl(requests)
43
+ rows = requests.flat_map { |request| render_embedding_batch_rows(request) }
44
+ rows.map { |row| JSON.generate(row) }.join("\n")
45
+ end
46
+
47
+ def parse_vertex_batch_results(rows, job:)
48
+ model = job.fetch('model').split('/').last
49
+ rows.group_by { |row| embedding_row_metadata(row).first }.filter_map do |index, group|
50
+ parse_embedding_batch_group(index, group, model:)
51
+ end
52
+ end
53
+ end
54
+ end
55
+ end
56
+ end
@@ -0,0 +1,101 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ module VertexAI
6
+ # Google Cloud Storage-backed files for Vertex AI batch input and output.
7
+ class Files < Protocols::Files
8
+ # GCS object names allow spaces and other characters URI() rejects.
9
+ GCS_URI = %r{\Ags://([^/]+)/?(.*)\z}m
10
+
11
+ # rubocop:disable-next Lint/UnusedMethodArgument
12
+ def upload(file, filename: nil, purpose: nil, expires_in: nil, uri: nil, content_type: nil,
13
+ provider_options: {})
14
+ attachment = file_attachment(file, filename:)
15
+ target_uri = uri || storage_uri_for(attachment)
16
+ bucket_name, key = parse_gcs_uri(target_uri)
17
+
18
+ with_file_body(attachment) do |body|
19
+ bucket(bucket_name).create_file(body, key, content_type: content_type || file_content_type(attachment))
20
+ end
21
+
22
+ uploaded_file(
23
+ { 'uri' => target_uri },
24
+ id: target_uri,
25
+ uri: target_uri,
26
+ filename: attachment.filename,
27
+ byte_size: file_size(attachment),
28
+ mime_type: content_type || file_content_type(attachment)
29
+ )
30
+ end
31
+
32
+ def find(file_id)
33
+ bucket_name, key = parse_gcs_uri(file_id)
34
+ object = bucket(bucket_name).file(key)
35
+ raise Error, "GCS object not found: #{file_id}" unless object
36
+
37
+ uploaded_file(
38
+ { 'uri' => file_id },
39
+ id: file_id,
40
+ uri: file_id,
41
+ filename: File.basename(key),
42
+ byte_size: object.size,
43
+ created_at: object.created_at,
44
+ mime_type: object.content_type
45
+ )
46
+ end
47
+
48
+ def download(file_id)
49
+ bucket_name, key = parse_gcs_uri(file_id)
50
+ object = bucket(bucket_name).file(key)
51
+ raise Error, "GCS object not found: #{file_id}" unless object
52
+
53
+ object.download.string
54
+ end
55
+
56
+ def list_uris(prefix_uri)
57
+ bucket_name, prefix = parse_gcs_uri(prefix_uri)
58
+ uris = []
59
+ bucket(bucket_name).files(prefix: prefix).all do |object|
60
+ uris << "gs://#{bucket_name}/#{object.name}"
61
+ end
62
+ uris
63
+ end
64
+
65
+ private
66
+
67
+ def storage_uri_for(attachment)
68
+ base = @config.vertexai_batch_gcs_uri.to_s.sub(%r{/+\z}, '')
69
+ raise ConfigurationError, 'Set vertexai_batch_gcs_uri to a gs:// bucket prefix' if base.empty?
70
+
71
+ "#{base}/ruby_llm_uploads/#{SecureRandom.hex(8)}/#{attachment.filename}"
72
+ end
73
+
74
+ def storage
75
+ require 'google/cloud/storage'
76
+
77
+ options = { project_id: @config.vertexai_project_id }
78
+ if @config.vertexai_service_account_key
79
+ options[:credentials] =
80
+ JSON.parse(@config.vertexai_service_account_key)
81
+ end
82
+ ::Google::Cloud::Storage.new(**options)
83
+ rescue LoadError
84
+ raise Error, 'The google-cloud-storage gem is required for Vertex AI file uploads. ' \
85
+ 'Please add it to your Gemfile: gem "google-cloud-storage"'
86
+ end
87
+
88
+ def bucket(name)
89
+ storage.bucket(name) || raise(Error, "GCS bucket not found: #{name}")
90
+ end
91
+
92
+ def parse_gcs_uri(uri)
93
+ match = GCS_URI.match(uri.to_s)
94
+ raise ArgumentError, "Expected a gs:// URI, got: #{uri}" unless match
95
+
96
+ [match[1], match[2]]
97
+ end
98
+ end
99
+ end
100
+ end
101
+ end
@@ -0,0 +1,69 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ module VertexAI
6
+ # Vertex AI Search's Discovery Engine ranking API.
7
+ class Ranking < Protocol
8
+ def initialize(provider, model = nil)
9
+ super
10
+ @connection = provider.ranking_connection
11
+ end
12
+
13
+ def rerank_url
14
+ "#{@provider.ranking_config}:rank"
15
+ end
16
+
17
+ def render_rerank_payload(query, documents, model:, top_n: nil, provider_options: {})
18
+ validate_ranking_input(query, documents, top_n)
19
+ records = documents.each_with_index.map { |document, index| { id: index.to_s, content: document } }
20
+ { model: model, query: query, records: records, topN: top_n }.compact.merge(provider_options)
21
+ end
22
+
23
+ def parse_rerank_response(response, model:, documents: [])
24
+ records = response.body['records']
25
+ raise Error.new('Vertex AI Search returned no ranking records', response:) unless records.is_a?(Array)
26
+
27
+ seen = []
28
+ results = records.map do |record|
29
+ index = ranking_index(record, documents)
30
+ raise Error.new('Vertex AI Search returned a duplicate document id', response:) if seen.include?(index)
31
+
32
+ seen << index
33
+ Rerank::Result.new(index: index, document: documents[index], score: record.fetch('score'))
34
+ end
35
+ Rerank.new(results: results, model: model, raw: response.body)
36
+ end
37
+
38
+ private
39
+
40
+ def validate_ranking_input(query, documents, top_n)
41
+ unless query.is_a?(String) && !query.empty?
42
+ raise ArgumentError, 'Vertex AI Search reranking requires a nonempty query'
43
+ end
44
+ unless valid_ranking_documents?(documents)
45
+ raise ArgumentError, 'Vertex AI Search reranking accepts between 1 and 1000 nonempty text documents'
46
+ end
47
+ return if top_n.nil? || (top_n.is_a?(Integer) && top_n.positive?)
48
+
49
+ raise ArgumentError, 'top_n must be a positive integer'
50
+ end
51
+
52
+ def valid_ranking_documents?(documents)
53
+ documents.is_a?(Array) && (1..1000).cover?(documents.length) &&
54
+ documents.all? { |document| document.is_a?(String) && !document.empty? }
55
+ end
56
+
57
+ def ranking_index(record, documents)
58
+ id = record['id'] if record.is_a?(Hash)
59
+ unless id.is_a?(String) && id.match?(/\A(?:0|[1-9]\d*)\z/) && id.to_i < documents.length &&
60
+ record['score'].is_a?(Numeric)
61
+ raise Error, 'Vertex AI Search returned an invalid document id or score'
62
+ end
63
+
64
+ id.to_i
65
+ end
66
+ end
67
+ end
68
+ end
69
+ end