ruby_llm 1.16.0 → 2.0.0.rc2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (475) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +85 -32
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
  91. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
  92. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
  93. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
  94. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  96. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
  97. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
  98. data/lib/ruby_llm/accounting/usage.rb +245 -0
  99. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  100. data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
  101. data/lib/ruby_llm/active_record/batch.rb +97 -0
  102. data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
  103. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  104. data/lib/ruby_llm/active_record/model.rb +135 -0
  105. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  106. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  107. data/lib/ruby_llm/active_record/usage.rb +61 -0
  108. data/lib/ruby_llm/agent.rb +1065 -151
  109. data/lib/ruby_llm/aliases.json +268 -100
  110. data/lib/ruby_llm/attachment.rb +187 -48
  111. data/lib/ruby_llm/batch.rb +432 -0
  112. data/lib/ruby_llm/cached_content.rb +112 -0
  113. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  114. data/lib/ruby_llm/chat.rb +1127 -198
  115. data/lib/ruby_llm/chunk.rb +10 -0
  116. data/lib/ruby_llm/citation.rb +105 -0
  117. data/lib/ruby_llm/configuration.rb +261 -24
  118. data/lib/ruby_llm/context.rb +128 -6
  119. data/lib/ruby_llm/cost.rb +217 -80
  120. data/lib/ruby_llm/downloaded_file.rb +33 -0
  121. data/lib/ruby_llm/embedding.rb +121 -17
  122. data/lib/ruby_llm/embedding_request.rb +53 -0
  123. data/lib/ruby_llm/error.rb +159 -23
  124. data/lib/ruby_llm/fallback.rb +133 -0
  125. data/lib/ruby_llm/files/mime_type.rb +97 -0
  126. data/lib/ruby_llm/image.rb +153 -32
  127. data/lib/ruby_llm/message.rb +233 -54
  128. data/lib/ruby_llm/model/modalities.rb +17 -4
  129. data/lib/ruby_llm/model/pricing.rb +24 -5
  130. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  131. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  132. data/lib/ruby_llm/model.rb +244 -2
  133. data/lib/ruby_llm/models/aliases.rb +41 -0
  134. data/lib/ruby_llm/models/registry.rb +165 -0
  135. data/lib/ruby_llm/models/schema.rb +99 -0
  136. data/lib/ruby_llm/models.json +73261 -33733
  137. data/lib/ruby_llm/models.rb +477 -215
  138. data/lib/ruby_llm/moderation.rb +139 -26
  139. data/lib/ruby_llm/ocr.rb +112 -0
  140. data/lib/ruby_llm/prompt.rb +79 -0
  141. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  142. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  143. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  144. data/lib/ruby_llm/protocol.rb +662 -0
  145. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  146. data/lib/ruby_llm/protocols/anthropic/chat.rb +548 -0
  147. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  148. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  149. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  150. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  151. data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
  152. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  153. data/lib/ruby_llm/protocols/anthropic.rb +101 -0
  154. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  155. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
  156. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  157. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  158. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  159. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  160. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  161. data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
  162. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  163. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  164. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  165. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  166. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  167. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  168. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
  169. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  170. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  171. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  172. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  173. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  174. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  175. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  176. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  177. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  178. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  179. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  180. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  181. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  182. data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
  183. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  184. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  185. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  186. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  187. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  188. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  189. data/lib/ruby_llm/protocols/converse/chat.rb +695 -0
  190. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  191. data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
  192. data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
  193. data/lib/ruby_llm/protocols/converse.rb +54 -0
  194. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  195. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  196. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  197. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  198. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  199. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  200. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  201. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  202. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  203. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  204. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  208. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  209. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  210. data/lib/ruby_llm/protocols/files.rb +119 -0
  211. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  212. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  213. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  214. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
  215. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  216. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  217. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  218. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  219. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  220. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  221. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  222. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  223. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  224. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  225. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  226. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  227. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  228. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  229. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  230. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  231. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  232. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  233. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  234. data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
  235. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  236. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  237. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  238. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  239. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  240. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  241. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  242. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  243. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  244. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  245. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  246. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  247. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  248. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  249. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  250. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  251. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  252. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  253. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  254. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  255. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  256. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  257. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  258. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  259. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  260. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  261. data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
  262. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  263. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  264. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  265. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  266. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  267. data/lib/ruby_llm/protocols/responses.rb +35 -0
  268. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  269. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  270. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  271. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  272. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  273. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  274. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  275. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  276. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  277. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  278. data/lib/ruby_llm/provider.rb +560 -128
  279. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  280. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  281. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  282. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  283. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  284. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  285. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  286. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  287. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  288. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  289. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  290. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  291. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  292. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  293. data/lib/ruby_llm/providers/azure.rb +77 -78
  294. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  295. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  296. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  297. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  298. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  299. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  300. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  301. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  302. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  303. data/lib/ruby_llm/providers/cohere.rb +31 -0
  304. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  305. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  306. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  307. data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
  308. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  309. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  310. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  311. data/lib/ruby_llm/providers/gemini.rb +15 -8
  312. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  313. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  314. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  315. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  316. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  317. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  318. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  319. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  320. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  321. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  322. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  323. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  324. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  325. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  326. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  327. data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
  328. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  329. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  330. data/lib/ruby_llm/providers/mistral.rb +16 -4
  331. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  332. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  333. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  334. data/lib/ruby_llm/providers/ollama.rb +9 -8
  335. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  336. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  337. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  338. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  339. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  340. data/lib/ruby_llm/providers/openai.rb +92 -11
  341. data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
  342. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  343. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  344. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  345. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  346. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  347. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  348. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  349. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  350. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  351. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  352. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  353. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  354. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  355. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  356. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  357. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  358. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  359. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  360. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  361. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  362. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  363. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  364. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  365. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  366. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  367. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  368. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  369. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  370. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  371. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  372. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  373. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  374. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  375. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  376. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  377. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  378. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  379. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  380. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  381. data/lib/ruby_llm/providers/xai.rb +15 -5
  382. data/lib/ruby_llm/railtie.rb +7 -16
  383. data/lib/ruby_llm/rerank.rb +105 -0
  384. data/lib/ruby_llm/research_job.rb +241 -0
  385. data/lib/ruby_llm/search_results.rb +68 -0
  386. data/lib/ruby_llm/server_tool_call.rb +73 -0
  387. data/lib/ruby_llm/speech.rb +159 -0
  388. data/lib/ruby_llm/speech_chunk.rb +33 -0
  389. data/lib/ruby_llm/support/deprecator.rb +22 -0
  390. data/lib/ruby_llm/support/inspectable.rb +49 -0
  391. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  392. data/lib/ruby_llm/support/utils.rb +147 -0
  393. data/lib/ruby_llm/thinking.rb +127 -20
  394. data/lib/ruby_llm/tokenization.rb +59 -0
  395. data/lib/ruby_llm/tokens.rb +103 -33
  396. data/lib/ruby_llm/tool.rb +266 -91
  397. data/lib/ruby_llm/tool_call.rb +36 -3
  398. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  399. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  400. data/lib/ruby_llm/transcription.rb +138 -13
  401. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  402. data/lib/ruby_llm/transport/connection.rb +193 -0
  403. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  404. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  405. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  406. data/lib/ruby_llm/uploaded_file.rb +144 -0
  407. data/lib/ruby_llm/version.rb +2 -1
  408. data/lib/ruby_llm/video.rb +136 -0
  409. data/lib/ruby_llm/video_job.rb +150 -0
  410. data/lib/ruby_llm/workflow.rb +91 -0
  411. data/lib/ruby_llm.rb +380 -6
  412. data/lib/tasks/ruby_llm.rake +21 -16
  413. data/skills/rubyllm/SKILL.md +81 -0
  414. data/skills/rubyllm/agents/openai.yaml +4 -0
  415. metadata +339 -97
  416. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  417. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  418. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  419. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  420. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  421. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  422. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  423. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  424. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  425. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  426. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  427. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  428. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  429. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  430. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  431. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  432. data/lib/ruby_llm/aliases.rb +0 -41
  433. data/lib/ruby_llm/connection.rb +0 -159
  434. data/lib/ruby_llm/content.rb +0 -91
  435. data/lib/ruby_llm/deprecator.rb +0 -24
  436. data/lib/ruby_llm/error_middleware.rb +0 -81
  437. data/lib/ruby_llm/instrumentation.rb +0 -36
  438. data/lib/ruby_llm/mime_type.rb +0 -96
  439. data/lib/ruby_llm/model/info.rb +0 -164
  440. data/lib/ruby_llm/model_registry.rb +0 -39
  441. data/lib/ruby_llm/models_schema.json +0 -171
  442. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  443. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  444. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  445. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  446. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  447. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  448. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  449. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  450. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  451. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  452. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  453. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  454. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  455. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  456. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  457. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  458. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  459. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  460. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  461. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  462. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  463. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  464. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  465. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  466. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  467. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  468. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  469. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  470. data/lib/ruby_llm/streaming.rb +0 -179
  471. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  472. data/lib/ruby_llm/utils.rb +0 -130
  473. data/lib/tasks/models.rake +0 -593
  474. data/lib/tasks/release.rake +0 -94
  475. data/lib/tasks/vcr.rake +0 -124
@@ -0,0 +1,111 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ module GPUStack
6
+ # vLLM Responses with MCP servers configured on the deployment.
7
+ class Responses < Protocols::Responses
8
+ MCP_LABELS = %w[web_search_preview code_interpreter container].freeze
9
+ MCP_ALIASES = {
10
+ web_search: ['web_search_preview', ['search']],
11
+ web_fetch: ['web_search_preview', ['open']],
12
+ code_execution: ['code_interpreter', nil]
13
+ }.freeze
14
+ SERVER_TOOL_ALIASES = MCP_ALIASES.transform_values do |label, tools|
15
+ ->(options) { render_mcp_alias(options, label:, tools:) }
16
+ end.merge(mcp: Protocols::Responses::SERVER_TOOL_ALIASES.fetch(:mcp)).freeze
17
+
18
+ def self.render_mcp_alias(options, label:, tools:) # :nodoc:
19
+ options = Support::Utils.deep_symbolize_keys(options)
20
+ unless (options.keys - [:require_approval]).empty?
21
+ raise ArgumentError, 'GPUStack server tool aliases accept only require_approval; use mcp for custom filters'
22
+ end
23
+
24
+ tool = { type: 'mcp', server_label: label, **options }
25
+ tool[:allowed_tools] = tools.dup if tools
26
+ { tool: }
27
+ end
28
+
29
+ def server_tool_aliases
30
+ SERVER_TOOL_ALIASES
31
+ end
32
+
33
+ def merge_server_tool_entries(payload, entries)
34
+ tools = entries.map { |entry| Support::Utils.deep_symbolize_keys(entry) }
35
+ tools.each do |tool|
36
+ next unless tool[:type] == 'mcp'
37
+
38
+ validate_mcp_tool(tool)
39
+ end
40
+ super(payload, merge_mcp_filters(tools))
41
+ end
42
+
43
+ def format_assistant_items(message)
44
+ super.flat_map do |item|
45
+ data = Support::Utils.deep_symbolize_keys(item)
46
+ next item unless data[:type] && !Protocols::Responses::Chat::CLIENT_OUTPUT_ITEM_TYPES.include?(data[:type])
47
+
48
+ format_server_tool_history(data)
49
+ end
50
+ end
51
+
52
+ def parse_reasoning_summary(output)
53
+ summary = super
54
+ return summary unless summary.to_s.empty?
55
+
56
+ text = output.select { |item| item['type'] == 'reasoning' }.flat_map { |item| item['content'] || [] }
57
+ .filter_map { |part| part['text'] }.join
58
+ text unless text.empty?
59
+ end
60
+
61
+ private
62
+
63
+ def merge_mcp_filters(tools)
64
+ tools.each_with_object([]) do |tool, merged|
65
+ existing = merged.find { |entry| entry[:type] == 'mcp' && entry[:server_label] == tool[:server_label] }
66
+ if tool[:type] != 'mcp' || existing.nil?
67
+ merged << tool
68
+ elsif existing != tool
69
+ merge_mcp_filter(existing, tool)
70
+ end
71
+ end
72
+ end
73
+
74
+ def merge_mcp_filter(existing, tool)
75
+ filters = [existing[:allowed_tools], tool[:allowed_tools]]
76
+ compatible = existing.except(:allowed_tools) == tool.except(:allowed_tools)
77
+ unless compatible && filters.all? { |filter| filter.is_a?(Array) && !filter.include?('*') }
78
+ raise ArgumentError, 'Combine GPUStack MCP settings for each server in one entry with explicit tool names'
79
+ end
80
+
81
+ existing[:allowed_tools] |= tool[:allowed_tools]
82
+ end
83
+
84
+ def format_server_tool_history(item)
85
+ if item[:type] == 'mcp_call' && item[:name] && item[:id] && !item[:output].nil?
86
+ output = item[:output].is_a?(String) ? item[:output] : JSON.generate(item[:output])
87
+ [{ type: 'function_call', call_id: item[:id], name: item[:name], arguments: item[:arguments] },
88
+ { type: 'function_call_output', call_id: item[:id], output: output }]
89
+ else
90
+ [{ role: 'assistant', content: [{ type: 'output_text', text: JSON.generate(item) }] }]
91
+ end
92
+ end
93
+
94
+ def validate_mcp_tool(tool)
95
+ unless tool[:require_approval].to_s == 'never'
96
+ raise ArgumentError, "GPUStack MCP requires explicit require_approval: 'never'; vLLM has no approval events"
97
+ end
98
+ if tool[:server_url] || tool[:connector_id] || tool[:authorization]
99
+ raise ArgumentError, 'GPUStack MCP uses servers configured on vLLM, not per-request URLs or connectors'
100
+ end
101
+ unless MCP_LABELS.include?(tool[:server_label])
102
+ raise ArgumentError, "GPUStack MCP name must match a configured vLLM label: #{MCP_LABELS.join(', ')}"
103
+ end
104
+ return unless tool[:allowed_tools].is_a?(Hash) && tool[:allowed_tools][:read_only]
105
+
106
+ raise ArgumentError, 'vLLM filters MCP tools by name, not by read_only; use allowed_tools: [name]'
107
+ end
108
+ end
109
+ end
110
+ end
111
+ end
@@ -0,0 +1,22 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ module GPUStack
6
+ # vLLM's text tokenizer through GPUStack's model proxy.
7
+ module Tokenization
8
+ def tokenization_url
9
+ "#{@provider.backend_api_base}/tokenize"
10
+ end
11
+
12
+ def render_tokenization_payload(text, model:)
13
+ { model: model, prompt: text }
14
+ end
15
+
16
+ def parse_tokenization_response(response, model:)
17
+ RubyLLM::Tokenization.new(ids: response.body.fetch('tokens'), model: model, raw: response.body)
18
+ end
19
+ end
20
+ end
21
+ end
22
+ end
@@ -0,0 +1,96 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ module GPUStack
6
+ # vLLM-Omni video jobs through GPUStack's model proxy.
7
+ module Videos
8
+ def video_url
9
+ "#{@provider.backend_api_base}/v1/videos"
10
+ end
11
+
12
+ def render_video_payload(prompt, model:, with: [], provider_options: {})
13
+ raise ArgumentError, 'vLLM-Omni video generation requires a prompt' if prompt.nil?
14
+
15
+ options = video_options(provider_options)
16
+ payload = { model: model, prompt: prompt }.merge(video_references(with)).merge(options).compact
17
+ payload.transform_values do |value|
18
+ value.is_a?(Hash) || value.is_a?(Array) ? JSON.generate(value) : value
19
+ end
20
+ end
21
+
22
+ def post_video(url, payload)
23
+ @connection.post(url, payload, idempotent: false) do |request|
24
+ request.headers['Content-Type'] = 'multipart/form-data'
25
+ end
26
+ end
27
+
28
+ def parse_video_job(response, model:)
29
+ body = response.body
30
+ id = body['id']
31
+ raise Error.new('GPUStack did not return a video job id', response:) unless id
32
+
33
+ VideoJob.new(id: id, protocol: self, model: body['model'] || model, **video_job_state(body))
34
+ end
35
+
36
+ def video_job_url(job)
37
+ "#{video_url}/#{job.id}"
38
+ end
39
+
40
+ def parse_video_job_status(response, **)
41
+ video_job_state(response.body)
42
+ end
43
+
44
+ def download_video(job)
45
+ response = @connection.get("#{video_job_url(job)}/content")
46
+ Video.new(data: response.body, mime_type: response.headers['content-type'] || job.raw['media_type'],
47
+ model: job.model, raw: job.raw)
48
+ end
49
+
50
+ private
51
+
52
+ def video_options(options)
53
+ options = Support::Utils.deep_symbolize_keys(options)
54
+ if options[:num_outputs_per_prompt] && options[:num_outputs_per_prompt] != 1
55
+ raise ArgumentError, 'animate returns one video; num_outputs_per_prompt must be 1'
56
+ end
57
+
58
+ options
59
+ end
60
+
61
+ def video_references(attachments)
62
+ attachments.group_by(&:type).to_h do |type, group|
63
+ values = group.map { |attachment| { "#{type}_url" => video_reference(attachment) } }
64
+ [:"#{type}_reference", values.one? ? values.first : values]
65
+ end
66
+ end
67
+
68
+ def video_job_state(body)
69
+ status = case body['status']
70
+ when 'queued', 'in_progress' then :pending
71
+ when 'completed' then :completed
72
+ when 'failed' then :failed
73
+ else raise Error, "Unknown GPUStack video status: #{body['status'].inspect}"
74
+ end
75
+ { status: status, raw: body, error: body.dig('error', 'message') }
76
+ end
77
+
78
+ def validate_animate_inputs!(with:)
79
+ @provider.backend_api_base
80
+ with.each do |attachment|
81
+ if attachment.provider_file?
82
+ raise ArgumentError, 'vLLM-Omni video references require media bytes or URLs, not uploaded file ids'
83
+ end
84
+ next if attachment.image? || attachment.video? || attachment.audio?
85
+
86
+ raise UnsupportedAttachmentError, attachment.mime_type
87
+ end
88
+ end
89
+
90
+ def video_reference(attachment)
91
+ attachment.url? ? attachment.source.to_s : attachment.for_llm
92
+ end
93
+ end
94
+ end
95
+ end
96
+ end
@@ -0,0 +1,145 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Interactions
6
+ module Chat # :nodoc:
7
+ FINISH_REASONS = { 'completed' => :stop, 'requires_action' => :tool_calls, 'incomplete' => :max_tokens }.freeze
8
+
9
+ module_function
10
+
11
+ def finish_reasons = FINISH_REASONS
12
+
13
+ def completion_url
14
+ 'interactions'
15
+ end
16
+
17
+ def render_payload(messages, tools:, temperature:, model:, stream: false, max_output_tokens: nil,
18
+ schema: nil, thinking: nil, tool_prefs: nil, **)
19
+ config = { temperature: temperature, max_output_tokens: max_output_tokens }.compact
20
+ config.merge!(render_interaction_thinking(thinking)) if thinking
21
+ choice = tool_prefs&.dig(:choice)
22
+ config[:tool_choice] = render_interaction_choice(choice) if choice
23
+ payload = {
24
+ model: model.id, input: format_interaction_input(messages), stream: stream, store: false,
25
+ system_instruction: messages.select { |message| message.role == :system }.map(&:content).join("\n\n"),
26
+ generation_config: config, tools: render_interaction_tools(tools)
27
+ }
28
+ payload[:response_format] = { type: 'text', mime_type: 'application/json', schema: schema[:schema] } if schema
29
+ payload
30
+ end
31
+
32
+ def render_interaction_thinking(thinking)
33
+ if thinking.disabled? || thinking.budget&.zero?
34
+ raise ArgumentError, 'Gemini Interactions does not expose a thinking-off control'
35
+ end
36
+ raise ArgumentError, 'Gemini Interactions accepts thinking effort, not a token budget' if thinking.budget
37
+ if thinking.effort && !%i[minimal low medium high].include?(thinking.effort)
38
+ raise ArgumentError, 'Gemini Interactions thinking effort must be minimal, low, medium, or high'
39
+ end
40
+
41
+ { thinking_level: thinking.effort&.to_s,
42
+ thinking_summaries: render_interaction_summaries(thinking.display) }.compact
43
+ end
44
+
45
+ def render_interaction_summaries(display)
46
+ unless [nil, :summarized, :omitted].include?(display)
47
+ raise ArgumentError, 'Gemini Interactions thinking display must be summarized or omitted'
48
+ end
49
+
50
+ display == :omitted ? 'none' : 'auto'
51
+ end
52
+
53
+ def format_interaction_input(messages)
54
+ calls = messages.flat_map { |message| message.tool_calls.to_h.values }.to_h { |call| [call.id, call] }
55
+ messages.reject { |message| message.role == :system }.flat_map do |message|
56
+ render_interaction_turn(message, calls)
57
+ end
58
+ end
59
+
60
+ def render_interaction_turn(message, calls)
61
+ if interaction_state?(message.raw_content)
62
+ render_interaction_history(message.raw_content.dig('response', 'steps') || [])
63
+ elsif message.tool_result?
64
+ [render_interaction_result(message, calls)]
65
+ else
66
+ render_interaction_message(message)
67
+ end
68
+ end
69
+
70
+ def render_interaction_result(message, calls)
71
+ { type: 'function_result', call_id: message.tool_call_id, name: calls[message.tool_call_id]&.name,
72
+ result: render_interaction_content(message.content, message.attachments) }.compact
73
+ end
74
+
75
+ def render_interaction_history(steps)
76
+ steps.map do |step|
77
+ step['type'].to_s.start_with?('mcp_server_') ? step.except('signature') : step
78
+ end
79
+ end
80
+
81
+ def render_interaction_message(message)
82
+ steps = []
83
+ if message.content || message.attachments.any?
84
+ steps << { type: message.role == :assistant ? 'model_output' : 'user_input',
85
+ content: render_interaction_content(message.content, message.attachments) }
86
+ end
87
+ message.tool_calls&.each_value do |call|
88
+ steps << { type: 'function_call', id: call.id, name: call.name, arguments: call.arguments }
89
+ end
90
+ steps
91
+ end
92
+
93
+ def interaction_state?(content)
94
+ content.is_a?(Hash) && content.dig('response', 'object') == 'interaction'
95
+ end
96
+
97
+ def parse_completion_body(data, raw:, model: data['model'] || @model&.id, cost: nil)
98
+ unless %w[completed requires_action incomplete].include?(data['status'])
99
+ message = Array(data['errors']).filter_map { |error| error['message'] }.join('; ')
100
+ raise Error.new(message.empty? ? "Gemini interaction ended with status #{data['status']}" : message,
101
+ response: raw)
102
+ end
103
+
104
+ steps = data.fetch('steps', [])
105
+ content = parse_interaction_content(steps)
106
+ calls = parse_interaction_calls(steps)
107
+ if data['status'] == 'requires_action' && calls.empty?
108
+ raise Error.new('Gemini interaction requires an unsupported action', response: raw)
109
+ end
110
+
111
+ Message.new(role: :assistant, content: content[:text], attachments: content[:attachments],
112
+ citations: content[:citations], thinking: parse_interaction_thinking(steps),
113
+ tool_calls: calls, server_tool_calls: parse_interaction_server_calls(steps),
114
+ raw_content: { 'response' => data },
115
+ model: model, raw: raw, cost: cost,
116
+ finish_reason: interaction_finish_reason(data['status'], calls),
117
+ **parse_interaction_usage(data['usage'] || {}))
118
+ end
119
+
120
+ def interaction_finish_reason(status, calls)
121
+ return :tool_calls unless calls.empty?
122
+
123
+ finish_reasons.fetch(status)
124
+ end
125
+
126
+ def parse_interaction_thinking(steps)
127
+ thoughts = steps.select { |step| step['type'] == 'thought' }
128
+ text = thoughts.flat_map { |step| Array(step['summary']) }.filter_map { |part| part['text'] }.join
129
+ Thinking.build(text: text.empty? ? nil : text, signature: thoughts.last&.dig('signature'))
130
+ end
131
+
132
+ def parse_interaction_usage(usage)
133
+ prompt = usage['total_input_tokens']
134
+ cached = usage['total_cached_tokens'].to_i
135
+ thoughts = usage['total_thought_tokens'].to_i
136
+ {
137
+ input_tokens: prompt && [prompt + usage['total_tool_use_tokens'].to_i - cached, 0].max,
138
+ output_tokens: usage['total_output_tokens'] && (usage['total_output_tokens'] + thoughts),
139
+ cache_read_tokens: usage['total_cached_tokens'], thinking_tokens: usage['total_thought_tokens']
140
+ }
141
+ end
142
+ end
143
+ end
144
+ end
145
+ end
@@ -0,0 +1,90 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'stringio'
4
+
5
+ module RubyLLM
6
+ module Protocols
7
+ class Interactions
8
+ module Content # :nodoc:
9
+ CITATION_TYPES = %w[url_citation file_citation place_citation].freeze
10
+
11
+ module_function
12
+
13
+ def render_interaction_content(text, attachments)
14
+ parts = text.nil? ? [] : [{ type: 'text', text: text.to_s }]
15
+ parts + attachments.map { |attachment| render_interaction_attachment(attachment) }
16
+ end
17
+
18
+ def render_interaction_attachment(attachment)
19
+ return { type: 'text', text: attachment.content } if attachment.text?
20
+
21
+ type = interaction_attachment_type(attachment)
22
+ raise ArgumentError, "Gemini Interactions does not support #{attachment.mime_type} input" unless type
23
+
24
+ part = { type: type, mime_type: attachment.mime_type }
25
+ if attachment.provider_file?
26
+ part.merge(uri: attachment.provider_file_uri)
27
+ elsif attachment.url?
28
+ part.merge(uri: attachment.source.to_s)
29
+ else
30
+ part.merge(data: attachment.encoded)
31
+ end
32
+ end
33
+
34
+ def interaction_attachment_type(attachment)
35
+ return 'image' if attachment.image?
36
+ return 'audio' if attachment.audio?
37
+ return 'video' if attachment.video?
38
+
39
+ 'document' if attachment.pdf?
40
+ end
41
+
42
+ def parse_interaction_content(steps)
43
+ result = { text: +'', attachments: [], citations: [] }
44
+ steps.select { |step| step['type'] == 'model_output' }.each do |step|
45
+ Array(step['content']).each { |part| parse_interaction_part(part, result) }
46
+ end
47
+ result
48
+ end
49
+
50
+ def parse_interaction_part(part, result)
51
+ if part['type'] == 'text'
52
+ text = part['text'].to_s
53
+ result[:citations].concat(parse_interaction_citations(part, result[:text].length))
54
+ result[:text] << text
55
+ elsif part['data'] || part['uri']
56
+ source = part['uri'] || StringIO.new(Base64.decode64(part['data']))
57
+ result[:attachments] << Attachment.new(source, config: @config)
58
+ end
59
+ end
60
+
61
+ def parse_interaction_citations(part, offset)
62
+ Array(part['annotations']).filter_map do |annotation|
63
+ next unless CITATION_TYPES.include?(annotation['type'])
64
+
65
+ Citation.new(**interaction_citation_source(annotation),
66
+ **interaction_citation_span(part['text'].to_s, annotation, offset))
67
+ end
68
+ end
69
+
70
+ def interaction_citation_source(annotation)
71
+ { url: annotation['url'] || annotation['document_uri'],
72
+ title: annotation['title'] || annotation['file_name'] || annotation['name'],
73
+ source_id: annotation['media_id'] || annotation['place_id'], cited_text: annotation['source'],
74
+ start_page: annotation['page_number'], end_page: annotation['page_number'] }
75
+ end
76
+
77
+ def interaction_citation_span(text, annotation, offset)
78
+ start_index = interaction_citation_index(text, annotation['start_index'])
79
+ end_index = interaction_citation_index(text, annotation['end_index'])
80
+ { start_index: start_index && (offset + start_index), end_index: end_index && (offset + end_index),
81
+ text: start_index && end_index && text[start_index...end_index] }
82
+ end
83
+
84
+ def interaction_citation_index(text, bytes)
85
+ text.byteslice(0, bytes)&.length unless bytes.nil?
86
+ end
87
+ end
88
+ end
89
+ end
90
+ end
@@ -0,0 +1,91 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Interactions
6
+ module Streaming # :nodoc:
7
+ module_function
8
+
9
+ def stream_response(payload, additional_headers = {})
10
+ @interaction_steps = {}
11
+ @interaction_response = {}
12
+ @interaction_done = false
13
+ response = stream_events(completion_url, payload, additional_headers) { |data| yield build_chunk(data) }
14
+ raise Error.new('Gemini interaction stream ended before completion', response:) unless @interaction_done
15
+
16
+ parse_completion_body(streamed_interaction, raw: response)
17
+ end
18
+
19
+ def build_chunk(data)
20
+ @interaction_steps ||= {}
21
+ @interaction_response ||= {}
22
+ case data['event_type']
23
+ when 'interaction.created'
24
+ @interaction_response.merge!(data.fetch('interaction'))
25
+ when 'interaction.completed'
26
+ @interaction_done = true
27
+ @interaction_response.merge!(data.fetch('interaction'))
28
+ return final_interaction_chunk
29
+ when 'error'
30
+ raise Error, data.dig('error', 'message') || 'Gemini interaction failed'
31
+ when 'step.start'
32
+ @interaction_steps[data.fetch('index')] = Support::Utils.deep_dup(data.fetch('step'))
33
+ when 'step.delta'
34
+ step = @interaction_steps.fetch(data.fetch('index'))
35
+ return append_interaction_delta(step, data.fetch('delta'))
36
+ end
37
+ Chunk.new(role: :assistant, content: nil)
38
+ end
39
+
40
+ def append_interaction_delta(step, delta)
41
+ type = delta['type']
42
+ case type
43
+ when 'text'
44
+ append_interaction_text(step, delta['text'])
45
+ return Chunk.new(role: :assistant, content: delta['text'])
46
+ when 'text_annotation'
47
+ append_interaction_text(step, '')
48
+ (step['content'].last['annotations'] ||= []) << delta['annotation']
49
+ when 'thought_summary'
50
+ (step['summary'] ||= []) << delta['content']
51
+ return Chunk.new(role: :assistant, content: nil,
52
+ thinking: Thinking.build(text: delta.dig('content', 'text')))
53
+ when 'thought_signature'
54
+ step['signature'] = delta['signature']
55
+ when 'arguments_delta'
56
+ step['arguments'] = +'' unless step['arguments'].is_a?(String)
57
+ step['arguments'] << delta['arguments'].to_s
58
+ when 'image', 'audio', 'video', 'document'
59
+ (step['content'] ||= []) << Support::Utils.deep_dup(delta)
60
+ else
61
+ step.merge!(delta.except('type'))
62
+ end
63
+ Chunk.new(role: :assistant, content: nil)
64
+ end
65
+
66
+ def append_interaction_text(step, text)
67
+ content = step['content'] ||= []
68
+ content << { 'type' => 'text', 'text' => +'' } unless content.last&.dig('type') == 'text'
69
+ content.last['text'] << text.to_s
70
+ end
71
+
72
+ def streamed_interaction
73
+ steps = @interaction_steps.sort.map do |_index, step|
74
+ next step unless step['type'] == 'function_call'
75
+
76
+ step.merge('arguments' => parse_interaction_arguments(step['arguments']))
77
+ end
78
+ @interaction_response.merge('steps' => steps)
79
+ end
80
+
81
+ def final_interaction_chunk
82
+ message = parse_completion_body(streamed_interaction, raw: nil)
83
+ Chunk.new(role: :assistant, content: nil, model: message.model, tokens: message.tokens,
84
+ citations: message.citations, tool_calls: message.tool_calls,
85
+ server_tool_calls: message.server_tool_calls, raw_content: message.raw_content,
86
+ attachments: message.attachments, finish_reason: message.finish_reason)
87
+ end
88
+ end
89
+ end
90
+ end
91
+ end
@@ -0,0 +1,51 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Interactions
6
+ module Tools # :nodoc:
7
+ module_function
8
+
9
+ def render_interaction_tools(tools)
10
+ tools.values.map do |tool|
11
+ parameters = tool.parameters_schema ||
12
+ Tool::SchemaDefinition.from_parameters(tool.declared_parameters)&.json_schema
13
+ Support::Utils.deep_merge({ type: 'function', name: tool.name, description: tool.description,
14
+ parameters: parameters }.compact, tool.provider_options)
15
+ end
16
+ end
17
+
18
+ def render_interaction_choice(choice)
19
+ return 'any' if choice == :required
20
+ return choice.to_s if %i[auto none].include?(choice)
21
+
22
+ { allowed_tools: { mode: 'any', tools: [choice.to_s] } }
23
+ end
24
+
25
+ def parse_interaction_calls(steps)
26
+ steps.select { |step| step['type'] == 'function_call' }.to_h do |step|
27
+ [step.fetch('id'), ToolCall.new(id: step.fetch('id'), name: step.fetch('name'),
28
+ arguments: parse_interaction_arguments(step['arguments']),
29
+ thought_signature: step['signature'])]
30
+ end
31
+ end
32
+
33
+ def parse_interaction_arguments(arguments)
34
+ arguments.is_a?(String) ? JSON.parse(arguments) : arguments || {}
35
+ rescue JSON::ParserError => e
36
+ raise ToolCallParseError.new(finish_reason: :tool_calls), cause: e
37
+ end
38
+
39
+ def parse_interaction_server_calls(steps)
40
+ steps.filter_map do |step|
41
+ type = step['type'].to_s
42
+ next unless type.end_with?('_call', '_result') && !type.start_with?('function_')
43
+
44
+ ServerToolCall.new(type: type, id: step['id'] || step['call_id'], name: step['name'],
45
+ input: step['arguments'], result: step['result'], raw: step)
46
+ end
47
+ end
48
+ end
49
+ end
50
+ end
51
+ end
@@ -0,0 +1,58 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Interactions
6
+ module Transcription # :nodoc: all
7
+ def render_transcription_options(timestamps:, **)
8
+ return {} if timestamps.nil?
9
+ raise ArgumentError, 'Gemini transcription timestamps must be word' unless timestamps == :word
10
+
11
+ { generation_config: { transcription_config: { mode: { type: 'verbatim',
12
+ timestamp_granularities: ['word'] } } } }
13
+ end
14
+
15
+ include Gemini::FileTranscription
16
+
17
+ def transcription_url(_model)
18
+ 'interactions'
19
+ end
20
+
21
+ def render_transcription_payload(attachment, model:, language:, speaker_names:, provider_options:, prompt:)
22
+ config = { language_codes: language && Array(language), custom_vocabulary: prompt && Array(prompt) }.compact
23
+ config[:mode] = { type: 'verbatim', diarization_mode: 'speaker' } if speaker_names
24
+ payload = { model:, store: false,
25
+ input: [{ type: 'audio', mime_type: attachment.mime_type, data: attachment.encoded }],
26
+ generation_config: { transcription_config: config } }
27
+ payload = Support::Utils.deep_merge(payload, provider_options)
28
+ validate_transcription_config(payload.dig(:generation_config, :transcription_config))
29
+ payload
30
+ end
31
+
32
+ def validate_transcription_config(config)
33
+ mode = config[:mode]
34
+ return unless config[:custom_vocabulary] && mode.is_a?(Hash)
35
+ return unless mode[:diarization_mode] || mode[:timestamp_granularities]
36
+
37
+ raise ArgumentError, 'Gemini custom vocabulary cannot be combined with diarization or word timestamps'
38
+ end
39
+
40
+ def parse_transcription_response(response, model:)
41
+ data = response.body
42
+ message = parse_completion_body(data, raw: response)
43
+ annotations = Array(data['steps']).flat_map { |step| Array(step['content']) }
44
+ .flat_map { |part| Array(part['annotations']) }
45
+ words = annotations.filter_map { |item| parse_transcription_word(item) if item['type'] == 'word_info' }
46
+ RubyLLM::Transcription.new(text: message.content, model:, words: words.empty? ? nil : words,
47
+ **parse_interaction_usage(data['usage'] || {}))
48
+ end
49
+
50
+ def parse_transcription_word(word)
51
+ { 'word' => word['text'], 'speaker' => word['speaker'],
52
+ 'start' => word['start_offset'] && Float(word['start_offset'].delete_suffix('s')),
53
+ 'end' => word['end_offset'] && Float(word['end_offset'].delete_suffix('s')) }.compact
54
+ end
55
+ end
56
+ end
57
+ end
58
+ end