ruby_llm 1.16.0 → 2.0.0.rc2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (475) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +85 -32
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
  91. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
  92. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
  93. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
  94. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  96. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
  97. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
  98. data/lib/ruby_llm/accounting/usage.rb +245 -0
  99. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  100. data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
  101. data/lib/ruby_llm/active_record/batch.rb +97 -0
  102. data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
  103. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  104. data/lib/ruby_llm/active_record/model.rb +135 -0
  105. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  106. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  107. data/lib/ruby_llm/active_record/usage.rb +61 -0
  108. data/lib/ruby_llm/agent.rb +1065 -151
  109. data/lib/ruby_llm/aliases.json +268 -100
  110. data/lib/ruby_llm/attachment.rb +187 -48
  111. data/lib/ruby_llm/batch.rb +432 -0
  112. data/lib/ruby_llm/cached_content.rb +112 -0
  113. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  114. data/lib/ruby_llm/chat.rb +1127 -198
  115. data/lib/ruby_llm/chunk.rb +10 -0
  116. data/lib/ruby_llm/citation.rb +105 -0
  117. data/lib/ruby_llm/configuration.rb +261 -24
  118. data/lib/ruby_llm/context.rb +128 -6
  119. data/lib/ruby_llm/cost.rb +217 -80
  120. data/lib/ruby_llm/downloaded_file.rb +33 -0
  121. data/lib/ruby_llm/embedding.rb +121 -17
  122. data/lib/ruby_llm/embedding_request.rb +53 -0
  123. data/lib/ruby_llm/error.rb +159 -23
  124. data/lib/ruby_llm/fallback.rb +133 -0
  125. data/lib/ruby_llm/files/mime_type.rb +97 -0
  126. data/lib/ruby_llm/image.rb +153 -32
  127. data/lib/ruby_llm/message.rb +233 -54
  128. data/lib/ruby_llm/model/modalities.rb +17 -4
  129. data/lib/ruby_llm/model/pricing.rb +24 -5
  130. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  131. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  132. data/lib/ruby_llm/model.rb +244 -2
  133. data/lib/ruby_llm/models/aliases.rb +41 -0
  134. data/lib/ruby_llm/models/registry.rb +165 -0
  135. data/lib/ruby_llm/models/schema.rb +99 -0
  136. data/lib/ruby_llm/models.json +73261 -33733
  137. data/lib/ruby_llm/models.rb +477 -215
  138. data/lib/ruby_llm/moderation.rb +139 -26
  139. data/lib/ruby_llm/ocr.rb +112 -0
  140. data/lib/ruby_llm/prompt.rb +79 -0
  141. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  142. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  143. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  144. data/lib/ruby_llm/protocol.rb +662 -0
  145. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  146. data/lib/ruby_llm/protocols/anthropic/chat.rb +548 -0
  147. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  148. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  149. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  150. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  151. data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
  152. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  153. data/lib/ruby_llm/protocols/anthropic.rb +101 -0
  154. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  155. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
  156. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  157. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  158. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  159. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  160. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  161. data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
  162. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  163. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  164. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  165. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  166. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  167. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  168. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
  169. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  170. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  171. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  172. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  173. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  174. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  175. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  176. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  177. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  178. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  179. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  180. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  181. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  182. data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
  183. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  184. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  185. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  186. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  187. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  188. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  189. data/lib/ruby_llm/protocols/converse/chat.rb +695 -0
  190. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  191. data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
  192. data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
  193. data/lib/ruby_llm/protocols/converse.rb +54 -0
  194. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  195. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  196. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  197. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  198. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  199. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  200. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  201. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  202. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  203. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  204. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  208. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  209. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  210. data/lib/ruby_llm/protocols/files.rb +119 -0
  211. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  212. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  213. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  214. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
  215. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  216. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  217. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  218. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  219. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  220. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  221. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  222. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  223. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  224. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  225. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  226. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  227. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  228. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  229. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  230. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  231. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  232. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  233. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  234. data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
  235. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  236. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  237. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  238. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  239. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  240. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  241. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  242. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  243. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  244. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  245. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  246. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  247. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  248. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  249. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  250. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  251. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  252. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  253. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  254. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  255. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  256. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  257. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  258. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  259. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  260. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  261. data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
  262. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  263. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  264. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  265. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  266. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  267. data/lib/ruby_llm/protocols/responses.rb +35 -0
  268. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  269. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  270. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  271. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  272. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  273. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  274. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  275. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  276. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  277. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  278. data/lib/ruby_llm/provider.rb +560 -128
  279. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  280. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  281. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  282. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  283. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  284. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  285. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  286. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  287. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  288. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  289. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  290. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  291. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  292. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  293. data/lib/ruby_llm/providers/azure.rb +77 -78
  294. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  295. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  296. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  297. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  298. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  299. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  300. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  301. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  302. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  303. data/lib/ruby_llm/providers/cohere.rb +31 -0
  304. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  305. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  306. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  307. data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
  308. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  309. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  310. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  311. data/lib/ruby_llm/providers/gemini.rb +15 -8
  312. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  313. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  314. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  315. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  316. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  317. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  318. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  319. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  320. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  321. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  322. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  323. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  324. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  325. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  326. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  327. data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
  328. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  329. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  330. data/lib/ruby_llm/providers/mistral.rb +16 -4
  331. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  332. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  333. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  334. data/lib/ruby_llm/providers/ollama.rb +9 -8
  335. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  336. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  337. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  338. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  339. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  340. data/lib/ruby_llm/providers/openai.rb +92 -11
  341. data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
  342. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  343. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  344. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  345. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  346. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  347. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  348. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  349. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  350. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  351. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  352. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  353. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  354. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  355. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  356. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  357. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  358. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  359. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  360. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  361. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  362. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  363. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  364. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  365. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  366. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  367. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  368. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  369. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  370. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  371. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  372. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  373. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  374. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  375. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  376. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  377. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  378. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  379. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  380. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  381. data/lib/ruby_llm/providers/xai.rb +15 -5
  382. data/lib/ruby_llm/railtie.rb +7 -16
  383. data/lib/ruby_llm/rerank.rb +105 -0
  384. data/lib/ruby_llm/research_job.rb +241 -0
  385. data/lib/ruby_llm/search_results.rb +68 -0
  386. data/lib/ruby_llm/server_tool_call.rb +73 -0
  387. data/lib/ruby_llm/speech.rb +159 -0
  388. data/lib/ruby_llm/speech_chunk.rb +33 -0
  389. data/lib/ruby_llm/support/deprecator.rb +22 -0
  390. data/lib/ruby_llm/support/inspectable.rb +49 -0
  391. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  392. data/lib/ruby_llm/support/utils.rb +147 -0
  393. data/lib/ruby_llm/thinking.rb +127 -20
  394. data/lib/ruby_llm/tokenization.rb +59 -0
  395. data/lib/ruby_llm/tokens.rb +103 -33
  396. data/lib/ruby_llm/tool.rb +266 -91
  397. data/lib/ruby_llm/tool_call.rb +36 -3
  398. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  399. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  400. data/lib/ruby_llm/transcription.rb +138 -13
  401. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  402. data/lib/ruby_llm/transport/connection.rb +193 -0
  403. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  404. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  405. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  406. data/lib/ruby_llm/uploaded_file.rb +144 -0
  407. data/lib/ruby_llm/version.rb +2 -1
  408. data/lib/ruby_llm/video.rb +136 -0
  409. data/lib/ruby_llm/video_job.rb +150 -0
  410. data/lib/ruby_llm/workflow.rb +91 -0
  411. data/lib/ruby_llm.rb +380 -6
  412. data/lib/tasks/ruby_llm.rake +21 -16
  413. data/skills/rubyllm/SKILL.md +81 -0
  414. data/skills/rubyllm/agents/openai.yaml +4 -0
  415. metadata +339 -97
  416. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  417. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  418. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  419. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  420. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  421. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  422. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  423. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  424. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  425. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  426. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  427. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  428. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  429. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  430. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  431. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  432. data/lib/ruby_llm/aliases.rb +0 -41
  433. data/lib/ruby_llm/connection.rb +0 -159
  434. data/lib/ruby_llm/content.rb +0 -91
  435. data/lib/ruby_llm/deprecator.rb +0 -24
  436. data/lib/ruby_llm/error_middleware.rb +0 -81
  437. data/lib/ruby_llm/instrumentation.rb +0 -36
  438. data/lib/ruby_llm/mime_type.rb +0 -96
  439. data/lib/ruby_llm/model/info.rb +0 -164
  440. data/lib/ruby_llm/model_registry.rb +0 -39
  441. data/lib/ruby_llm/models_schema.json +0 -171
  442. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  443. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  444. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  445. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  446. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  447. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  448. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  449. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  450. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  451. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  452. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  453. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  454. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  455. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  456. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  457. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  458. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  459. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  460. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  461. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  462. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  463. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  464. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  465. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  466. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  467. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  468. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  469. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  470. data/lib/ruby_llm/streaming.rb +0 -179
  471. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  472. data/lib/ruby_llm/utils.rb +0 -130
  473. data/lib/tasks/models.rake +0 -593
  474. data/lib/tasks/release.rake +0 -94
  475. data/lib/tasks/vcr.rake +0 -124
@@ -1,91 +1,101 @@
1
1
  # frozen_string_literal: true
2
2
 
3
- require 'time'
4
-
5
3
  module RubyLLM
6
4
  module Providers
7
5
  class GPUStack
8
- # Models methods of the GPUStack API integration
6
+ # Models methods of the GPUStack API integration. GPUStack 2.x serves
7
+ # the plain OpenAI list format at /v1/models; categories arrive through
8
+ # per-category queries because the list response doesn't include them.
9
9
  module Models
10
- module_function
10
+ CATEGORIES = %w[llm embedding image reranker speech_to_text text_to_speech unknown].freeze
11
11
 
12
12
  def models_url
13
- 'models'
13
+ 'models?with_meta=true'
14
14
  end
15
15
 
16
- def parse_list_models_response(response, slug, _capabilities)
17
- items = response.body['items'] || []
18
- items.map do |model|
19
- Model::Info.new(
20
- id: model['name'],
21
- name: model['name'],
22
- created_at: model['created_at'] ? Time.parse(model['created_at']) : nil,
23
- provider: slug,
24
- family: 'gpustack',
25
- metadata: {
26
- description: model['description'],
27
- source: model['source'],
28
- huggingface_repo_id: model['huggingface_repo_id'],
29
- ollama_library_model_name: model['ollama_library_model_name'],
30
- backend: model['backend'],
31
- meta: model['meta'],
32
- categories: model['categories']
33
- },
34
- context_window: model.dig('meta', 'n_ctx'),
35
- max_output_tokens: model.dig('meta', 'n_ctx'),
36
- capabilities: build_capabilities(model),
37
- modalities: build_modalities(model),
38
- pricing: {}
39
- )
16
+ def list_models
17
+ models = {}
18
+ CATEGORIES.each do |category|
19
+ data = @connection.get("#{models_url}&categories=#{category}").body['data'] || []
20
+ data.each do |model|
21
+ entry = models[model['id']] ||= model.merge('categories' => [])
22
+ entry['categories'] << category
23
+ end
40
24
  end
25
+ models.values.map { |model| build_model(model, @provider.slug) }
26
+ end
27
+
28
+ def parse_list_models_response(response, slug)
29
+ data = response.body['data'] || []
30
+ data.map { |model| build_model(model, slug) }
41
31
  end
42
32
 
43
33
  private
44
34
 
45
- def determine_model_type(model)
46
- return 'embedding' if model['categories']&.include?('embedding')
47
- return 'chat' if model['categories']&.include?('llm')
35
+ def build_model(model, slug)
36
+ meta = model['meta'] || {}
37
+ categories = model['categories'] || []
48
38
 
49
- 'other'
39
+ Model.new(
40
+ id: model['id'],
41
+ name: model['id'],
42
+ created_at: model['created'] ? Time.at(model['created']) : nil,
43
+ provider: slug,
44
+ family: 'gpustack',
45
+ context_window: context_window(meta),
46
+ max_output_tokens: context_window(meta),
47
+ capabilities: build_capabilities(categories, meta),
48
+ modalities: build_modalities(categories, meta),
49
+ pricing: {},
50
+ metadata: {
51
+ owned_by: model['owned_by'],
52
+ categories: categories,
53
+ meta: model['meta']
54
+ }
55
+ )
50
56
  end
51
57
 
52
- def build_capabilities(model) # rubocop:disable Metrics/PerceivedComplexity
53
- capabilities = []
54
-
55
- # Add streaming by default for LLM models
56
- capabilities << 'streaming' if model['categories']&.include?('llm')
57
-
58
- # Map GPUStack metadata to standard capabilities
59
- capabilities << 'function_calling' if model.dig('meta', 'support_tool_calls')
60
- capabilities << 'vision' if model.dig('meta', 'support_vision')
61
- capabilities << 'reasoning' if model.dig('meta', 'support_reasoning')
58
+ def context_window(meta)
59
+ meta['n_ctx'] || meta['max_model_len']
60
+ end
62
61
 
63
- # GPUStack models generally support structured output and json mode
64
- capabilities << 'structured_output' if model['categories']&.include?('llm')
65
- capabilities << 'json_mode' if model['categories']&.include?('llm')
62
+ def build_capabilities(categories, meta)
63
+ return [] unless categories.include?('llm')
66
64
 
65
+ capabilities = %w[streaming structured_output json_mode]
66
+ capabilities << 'function_calling' if meta['support_tool_calls']
67
+ capabilities << 'vision' if meta['support_vision']
68
+ capabilities << 'reasoning' if meta['support_reasoning']
67
69
  capabilities
68
70
  end
69
71
 
70
- def build_modalities(model)
71
- input_modalities = []
72
- output_modalities = []
73
-
74
- if model['categories']&.include?('llm')
75
- input_modalities << 'text'
76
- input_modalities << 'image' if model.dig('meta', 'support_vision')
77
- input_modalities << 'audio' if model.dig('meta', 'support_audio')
78
- output_modalities << 'text'
79
- elsif model['categories']&.include?('embedding')
80
- input_modalities << 'text'
81
- output_modalities << 'embeddings'
82
- end
83
-
72
+ def build_modalities(categories, meta)
84
73
  {
85
- input: input_modalities,
86
- output: output_modalities
74
+ input: input_modalities(categories, meta),
75
+ output: output_modalities(categories)
87
76
  }
88
77
  end
78
+
79
+ def input_modalities(categories, meta)
80
+ inputs = []
81
+ inputs << 'text' if categories.intersect?(%w[llm embedding image reranker text_to_speech])
82
+ inputs << 'image' if categories.include?('llm') && meta['support_vision']
83
+ inputs << 'audio' if audio_input?(categories, meta)
84
+ inputs
85
+ end
86
+
87
+ def audio_input?(categories, meta)
88
+ categories.include?('speech_to_text') || (categories.include?('llm') && meta['support_audio'])
89
+ end
90
+
91
+ def output_modalities(categories)
92
+ outputs = []
93
+ outputs << 'text' if categories.intersect?(%w[llm speech_to_text reranker])
94
+ outputs << 'embeddings' if categories.include?('embedding')
95
+ outputs << 'image' if categories.include?('image')
96
+ outputs << 'audio' if categories.include?('text_to_speech')
97
+ outputs
98
+ end
89
99
  end
90
100
  end
91
101
  end
@@ -0,0 +1,15 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class GPUStack
6
+ # GPUStack's vLLM speech endpoint requires an explicit stream flag.
7
+ module Speech
8
+ def stream_speech(payload, model:, voice:, format:, &)
9
+ format ||= 'pcm'
10
+ super(payload.merge(stream: true, response_format: format), model:, voice:, format:, &)
11
+ end
12
+ end
13
+ end
14
+ end
15
+ end
@@ -0,0 +1,29 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class GPUStack
6
+ # GPUStack's vLLM transcription stream uses Chat Completions deltas.
7
+ module Transcription
8
+ module_function
9
+
10
+ def stream_transcription(payload, model:, &)
11
+ super({ stream_include_usage: 'true' }.merge(payload), model:, &)
12
+ end
13
+
14
+ def build_transcription_chunk(data)
15
+ return super unless data.key?('choices')
16
+
17
+ choice = data['choices'].first || {}
18
+ finished = choice['finish_reason'] || data['choices'].empty?
19
+
20
+ RubyLLM::TranscriptionChunk.new(
21
+ type: finished ? RubyLLM::TranscriptionChunk::DONE : RubyLLM::TranscriptionChunk::DELTA,
22
+ delta: choice.dig('delta', 'content'),
23
+ raw: data
24
+ )
25
+ end
26
+ end
27
+ end
28
+ end
29
+ end
@@ -2,16 +2,44 @@
2
2
 
3
3
  module RubyLLM
4
4
  module Providers
5
- # GPUStack API integration based on Ollama.
6
- class GPUStack < OpenAI
7
- include GPUStack::Chat
8
- include GPUStack::Models
9
- include GPUStack::Media
5
+ # GPUStack API integration.
6
+ class GPUStack < Provider
7
+ # GPUStack's dialect of the Chat Completions API.
8
+ class ChatCompletions < Protocols::ChatCompletions
9
+ include GPUStack::Chat
10
+ include GPUStack::Embeddings
11
+ include GPUStack::Media
12
+ include GPUStack::Models
13
+ include GPUStack::Speech
14
+ include GPUStack::Transcription
15
+ include Protocols::ChatCompletions::Rerank
16
+ include Protocols::GPUStack::Tokenization
17
+ include Protocols::GPUStack::Videos
18
+ end
19
+
20
+ protocol :chat_completions, ChatCompletions
21
+ protocol :responses, Protocols::GPUStack::Responses
22
+
23
+ def resolve_protocol(name, model, **request)
24
+ return fetch_protocol(:chat_completions) if !name && request[:operation]
25
+
26
+ super
27
+ end
10
28
 
11
29
  def api_base
12
30
  @config.gpustack_api_base
13
31
  end
14
32
 
33
+ def backend_api_base # :nodoc:
34
+ uri = URI(api_base)
35
+ unless uri.path.match?(%r{/model/proxy/\d+/v1/?\z})
36
+ raise Error, 'This GPUStack operation requires gpustack_api_base to end in /model/proxy/ROUTE_ID/v1'
37
+ end
38
+
39
+ uri.path = uri.path.sub(%r{/v1/?\z}, '')
40
+ uri.to_s
41
+ end
42
+
15
43
  def headers
16
44
  return {} unless @config.gpustack_api_key
17
45
 
@@ -25,16 +53,12 @@ module RubyLLM
25
53
  %i[gpustack_api_base gpustack_api_key]
26
54
  end
27
55
 
28
- def local?
29
- true
30
- end
31
-
32
56
  def configuration_requirements
33
57
  %i[gpustack_api_base]
34
58
  end
35
59
 
36
- def capabilities
37
- GPUStack::Capabilities
60
+ def local?
61
+ true
38
62
  end
39
63
  end
40
64
  end
@@ -3,164 +3,16 @@
3
3
  module RubyLLM
4
4
  module Providers
5
5
  class Mistral
6
- # Determines capabilities for Mistral models
6
+ # Feature capability gaps not represented in upstream model catalogs.
7
7
  module Capabilities
8
- module_function
8
+ TOOL_CAPABILITIES = %w[tool_choice parallel_tool_calls].freeze
9
+ STRUCTURED_OUTPUT_MODELS = %w[mistral-small-2603 mistral-small-latest].freeze
9
10
 
10
- def supports_streaming?(model_id)
11
- !model_id.match?(/embed|moderation|ocr|transcriptions/)
12
- end
13
-
14
- def supports_tools?(model_id)
15
- !model_id.match?(/embed|moderation|ocr|voxtral|transcriptions|mistral-(tiny|small)-(2312|2402)/)
16
- end
17
-
18
- def supports_tool_choice?(_model_id)
19
- true
20
- end
21
-
22
- def supports_tool_parallel_control?(_model_id)
23
- true
24
- end
25
-
26
- def supports_vision?(model_id)
27
- model_id.match?(/pixtral|mistral-small-(2503|2506)|mistral-medium/)
28
- end
29
-
30
- def supports_json_mode?(model_id)
31
- !model_id.match?(/embed|moderation|ocr|voxtral|transcriptions/) && supports_tools?(model_id)
32
- end
33
-
34
- def supports_reasoning?(model_id)
35
- model_id.match?(/magistral/) ||
36
- model_id.match?(/\Amistral-(?:small-latest|medium-(?:3(?:[.-]5)?|latest))\z/)
37
- end
38
-
39
- def format_display_name(model_id)
40
- case model_id
41
- when /mistral-large/ then 'Mistral Large'
42
- when /mistral-medium/ then 'Mistral Medium'
43
- when /mistral-small/ then 'Mistral Small'
44
- when /ministral-3b/ then 'Ministral 3B'
45
- when /ministral-8b/ then 'Ministral 8B'
46
- when /codestral/ then 'Codestral'
47
- when /pixtral-large/ then 'Pixtral Large'
48
- when /pixtral-12b/ then 'Pixtral 12B'
49
- when /mistral-embed/ then 'Mistral Embed'
50
- when /mistral-moderation/ then 'Mistral Moderation'
51
- else model_id.split('-').map(&:capitalize).join(' ')
52
- end
53
- end
54
-
55
- def model_family(model_id)
56
- case model_id
57
- when /mistral-large/ then 'mistral-large'
58
- when /mistral-medium/ then 'mistral-medium'
59
- when /mistral-small/ then 'mistral-small'
60
- when /ministral/ then 'ministral'
61
- when /codestral/ then 'codestral'
62
- when /pixtral/ then 'pixtral'
63
- when /mistral-embed/ then 'mistral-embed'
64
- when /mistral-moderation/ then 'mistral-moderation'
65
- else 'mistral'
66
- end
67
- end
68
-
69
- def context_window_for(_model_id)
70
- 32_768
71
- end
72
-
73
- def max_tokens_for(_model_id)
74
- 8192
75
- end
76
-
77
- def modalities_for(model_id)
78
- case model_id
79
- when /pixtral/
80
- {
81
- input: %w[text image],
82
- output: ['text']
83
- }
84
- when /embed/
85
- {
86
- input: ['text'],
87
- output: ['embeddings']
88
- }
89
- else
90
- {
91
- input: ['text'],
92
- output: ['text']
93
- }
94
- end
95
- end
96
-
97
- def capabilities_for(model_id) # rubocop:disable Metrics/PerceivedComplexity
98
- case model_id
99
- when /moderation/ then ['moderation']
100
- when /voxtral.*transcribe/ then ['transcription']
101
- when /ocr/ then ['vision']
102
- else
103
- capabilities = []
104
- capabilities << 'streaming' if supports_streaming?(model_id)
105
- capabilities << 'function_calling' if supports_tools?(model_id)
106
- capabilities << 'structured_output' if supports_json_mode?(model_id)
107
- capabilities << 'vision' if supports_vision?(model_id)
108
-
109
- capabilities << 'reasoning' if supports_reasoning?(model_id)
110
- capabilities << 'batch' unless model_id.match?(/voxtral|ocr|embed|moderation/)
111
- capabilities << 'fine_tuning' if model_id.match?(/mistral-(small|medium|large)|devstral/)
112
- capabilities << 'distillation' if model_id.match?(/ministral/)
113
- capabilities << 'predicted_outputs' if model_id.match?(/codestral/)
114
-
115
- capabilities.uniq
116
- end
117
- end
118
-
119
- def pricing_for(_model_id)
120
- {
121
- input: 0.0,
122
- output: 0.0
123
- }
124
- end
125
-
126
- def release_date_for(model_id) # rubocop:disable Metrics/CyclomaticComplexity
127
- case model_id
128
- when 'open-mistral-7b', 'mistral-tiny' then '2023-09-27'
129
- when 'mistral-medium-2312', 'mistral-small-2312', 'mistral-small',
130
- 'open-mixtral-8x7b', 'mistral-tiny-2312' then '2023-12-11'
131
-
132
- when 'mistral-embed' then '2024-01-11'
133
- when 'mistral-large-2402', 'mistral-small-2402' then '2024-02-26'
134
- when 'open-mixtral-8x22b', 'open-mixtral-8x22b-2404' then '2024-04-17'
135
- when 'codestral-2405' then '2024-05-22'
136
- when 'codestral-mamba-2407', 'codestral-mamba-latest', 'open-codestral-mamba' then '2024-07-16'
137
- when 'open-mistral-nemo', 'open-mistral-nemo-2407', 'mistral-tiny-2407',
138
- 'mistral-tiny-latest' then '2024-07-18'
139
- when 'mistral-large-2407' then '2024-07-24'
140
- when 'pixtral-12b-2409', 'pixtral-12b-latest', 'pixtral-12b' then '2024-09-17'
141
- when 'mistral-small-2409' then '2024-09-18'
142
- when 'ministral-3b-2410', 'ministral-3b-latest', 'ministral-8b-2410',
143
- 'ministral-8b-latest' then '2024-10-16'
144
- when 'pixtral-large-2411', 'pixtral-large-latest', 'mistral-large-pixtral-2411' then '2024-11-12'
145
- when 'mistral-large-2411', 'mistral-large-latest', 'mistral-large' then '2024-11-20'
146
- when 'codestral-2411-rc5', 'mistral-moderation-2411', 'mistral-moderation-latest' then '2024-11-26'
147
- when 'codestral-2412' then '2024-12-17'
11
+ def self.augment(capabilities, model_id:, **)
12
+ capabilities |= TOOL_CAPABILITIES if capabilities.include?('function_calling')
13
+ capabilities |= ['structured_output'] if STRUCTURED_OUTPUT_MODELS.include?(model_id)
148
14
 
149
- when 'mistral-small-2501' then '2025-01-13'
150
- when 'codestral-2501' then '2025-01-14'
151
- when 'mistral-saba-2502', 'mistral-saba-latest' then '2025-02-18'
152
- when 'mistral-small-2503' then '2025-03-03'
153
- when 'mistral-ocr-2503' then '2025-03-21'
154
- when 'mistral-medium', 'mistral-medium-latest', 'mistral-medium-2505' then '2025-05-06'
155
- when 'codestral-embed', 'codestral-embed-2505' then '2025-05-21'
156
- when 'mistral-ocr-2505', 'mistral-ocr-latest' then '2025-05-23'
157
- when 'devstral-small-2505' then '2025-05-28'
158
- when 'mistral-small-2506', 'mistral-small-latest', 'magistral-medium-2506',
159
- 'magistral-medium-latest' then '2025-06-10'
160
- when 'devstral-small-2507', 'devstral-small-latest', 'devstral-medium-2507',
161
- 'devstral-medium-latest' then '2025-07-09'
162
- when 'codestral-2508', 'codestral-latest' then '2025-08-30'
163
- end
15
+ capabilities
164
16
  end
165
17
  end
166
18
  end
@@ -5,38 +5,51 @@ module RubyLLM
5
5
  class Mistral
6
6
  # Chat methods for Mistral API
7
7
  module Chat
8
+ PROMPT_CACHE_OPTIONS = %i[key].freeze
9
+
8
10
  module_function
9
11
 
10
12
  def format_role(role)
11
13
  role.to_s
12
14
  end
13
15
 
14
- def format_messages(messages)
15
- messages.map do |msg|
16
- {
17
- role: format_role(msg.role),
18
- content: format_content_with_thinking(msg),
19
- tool_calls: OpenAI::Tools.format_tool_calls(msg.tool_calls),
20
- tool_call_id: msg.tool_call_id
21
- }.compact
22
- end
23
- end
24
-
25
- # rubocop:disable Metrics/ParameterLists
26
- def render_payload(messages, tools:, temperature:, model:, stream: false,
27
- schema: nil, thinking: nil, tool_prefs: nil)
16
+ def render_payload(messages, tools:, temperature:, model:, stream: false, max_output_tokens: nil,
17
+ schema: nil, thinking: nil, citations: false, caching: nil, tool_prefs: nil)
28
18
  payload = super
29
19
  payload.delete(:stream_options)
30
- configure_thinking_payload(payload, model, thinking)
31
20
  normalize_required_tool_choice(payload)
21
+ payload.merge!(prompt_cache_params(caching)) if caching
32
22
  payload
33
23
  end
34
- # rubocop:enable Metrics/ParameterLists
24
+
25
+ def openai_prompt_caching?
26
+ false
27
+ end
28
+
29
+ def prompt_cache_params(caching)
30
+ options = prompt_cache_options(caching)
31
+
32
+ {}.tap do |params|
33
+ params[:prompt_cache_key] = options[:key] if options[:key]
34
+ end
35
+ end
36
+
37
+ def prompt_cache_options(caching)
38
+ options = caching.to_h.transform_keys(&:to_sym)
39
+ unsupported = options.keys - PROMPT_CACHE_OPTIONS
40
+ return options if unsupported.empty?
41
+
42
+ raise ArgumentError, "Mistral prompt caching accepts :key, got #{format_cache_option_keys(unsupported)}"
43
+ end
44
+
45
+ def format_cache_option_keys(keys)
46
+ keys.map { |key| ":#{key}" }.join(', ')
47
+ end
35
48
 
36
49
  def build_tool_choice(tool_choice)
37
50
  return 'any' if tool_choice == :required
38
51
 
39
- OpenAI::Tools.build_tool_choice(tool_choice)
52
+ Protocols::ChatCompletions::Tools.build_tool_choice(tool_choice)
40
53
  end
41
54
 
42
55
  def normalize_required_tool_choice(payload)
@@ -51,8 +64,10 @@ module RubyLLM
51
64
  }
52
65
  end
53
66
 
54
- def format_content_with_thinking(msg)
55
- formatted_content = Mistral::Media.format_content(msg.content)
67
+ # Mistral carries reasoning in content blocks, not in the top-level
68
+ # reasoning fields the rest of the Chat Completions family uses.
69
+ def format_message_content(msg, **)
70
+ formatted_content = super
56
71
  return formatted_content unless msg.role == :assistant && msg.thinking
57
72
 
58
73
  content_blocks = build_thinking_blocks(msg.thinking)
@@ -61,47 +76,8 @@ module RubyLLM
61
76
  content_blocks
62
77
  end
63
78
 
64
- def warn_on_unsupported_thinking(model, thinking)
65
- return unless thinking&.enabled?
66
- return if native_reasoning_model?(model.id) || adjustable_reasoning_model?(model.id)
67
-
68
- RubyLLM.logger.warn(
69
- 'Mistral thinking is only supported on Magistral and adjustable-reasoning models. ' \
70
- "Ignoring thinking settings for #{model.id}."
71
- )
72
- end
73
-
74
- def configure_thinking_payload(payload, model, thinking)
75
- return unless thinking&.enabled?
76
-
77
- if native_reasoning_model?(model.id)
78
- configure_native_reasoning_payload(payload, thinking)
79
- elsif adjustable_reasoning_model?(model.id)
80
- payload[:reasoning_effort] = reasoning_effort_for(thinking)
81
- else
82
- payload.delete(:reasoning_effort)
83
- warn_on_unsupported_thinking(model, thinking)
84
- end
85
- end
86
-
87
- def configure_native_reasoning_payload(payload, thinking)
88
- payload.delete(:reasoning_effort)
89
- payload[:prompt_mode] = thinking.effort == 'none' ? nil : 'reasoning'
90
- end
91
-
92
- def reasoning_effort_for(thinking)
93
- effort = thinking.respond_to?(:effort) ? thinking.effort : nil
94
- return effort if %w[high none].include?(effort)
95
-
96
- 'high'
97
- end
98
-
99
- def native_reasoning_model?(model_id)
100
- model_id.to_s.include?('magistral')
101
- end
102
-
103
- def adjustable_reasoning_model?(model_id)
104
- model_id.to_s.match?(/\Amistral-(?:small-latest|medium-(?:3(?:[.-]5)?|latest))\z/)
79
+ def format_thinking(_msg)
80
+ {}
105
81
  end
106
82
 
107
83
  def build_thinking_blocks(thinking)
@@ -123,7 +99,7 @@ module RubyLLM
123
99
  def append_formatted_content(content_blocks, formatted_content)
124
100
  if formatted_content.is_a?(Array)
125
101
  content_blocks.concat(formatted_content)
126
- elsif formatted_content
102
+ elsif formatted_content && !formatted_content.empty?
127
103
  content_blocks << { type: 'text', text: formatted_content }
128
104
  end
129
105
  end