ruby_llm 1.16.0 → 2.0.0.rc2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (475) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +85 -32
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
  91. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
  92. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
  93. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
  94. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  96. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
  97. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
  98. data/lib/ruby_llm/accounting/usage.rb +245 -0
  99. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  100. data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
  101. data/lib/ruby_llm/active_record/batch.rb +97 -0
  102. data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
  103. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  104. data/lib/ruby_llm/active_record/model.rb +135 -0
  105. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  106. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  107. data/lib/ruby_llm/active_record/usage.rb +61 -0
  108. data/lib/ruby_llm/agent.rb +1065 -151
  109. data/lib/ruby_llm/aliases.json +268 -100
  110. data/lib/ruby_llm/attachment.rb +187 -48
  111. data/lib/ruby_llm/batch.rb +432 -0
  112. data/lib/ruby_llm/cached_content.rb +112 -0
  113. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  114. data/lib/ruby_llm/chat.rb +1127 -198
  115. data/lib/ruby_llm/chunk.rb +10 -0
  116. data/lib/ruby_llm/citation.rb +105 -0
  117. data/lib/ruby_llm/configuration.rb +261 -24
  118. data/lib/ruby_llm/context.rb +128 -6
  119. data/lib/ruby_llm/cost.rb +217 -80
  120. data/lib/ruby_llm/downloaded_file.rb +33 -0
  121. data/lib/ruby_llm/embedding.rb +121 -17
  122. data/lib/ruby_llm/embedding_request.rb +53 -0
  123. data/lib/ruby_llm/error.rb +159 -23
  124. data/lib/ruby_llm/fallback.rb +133 -0
  125. data/lib/ruby_llm/files/mime_type.rb +97 -0
  126. data/lib/ruby_llm/image.rb +153 -32
  127. data/lib/ruby_llm/message.rb +233 -54
  128. data/lib/ruby_llm/model/modalities.rb +17 -4
  129. data/lib/ruby_llm/model/pricing.rb +24 -5
  130. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  131. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  132. data/lib/ruby_llm/model.rb +244 -2
  133. data/lib/ruby_llm/models/aliases.rb +41 -0
  134. data/lib/ruby_llm/models/registry.rb +165 -0
  135. data/lib/ruby_llm/models/schema.rb +99 -0
  136. data/lib/ruby_llm/models.json +73261 -33733
  137. data/lib/ruby_llm/models.rb +477 -215
  138. data/lib/ruby_llm/moderation.rb +139 -26
  139. data/lib/ruby_llm/ocr.rb +112 -0
  140. data/lib/ruby_llm/prompt.rb +79 -0
  141. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  142. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  143. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  144. data/lib/ruby_llm/protocol.rb +662 -0
  145. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  146. data/lib/ruby_llm/protocols/anthropic/chat.rb +548 -0
  147. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  148. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  149. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  150. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  151. data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
  152. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  153. data/lib/ruby_llm/protocols/anthropic.rb +101 -0
  154. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  155. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
  156. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  157. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  158. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  159. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  160. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  161. data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
  162. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  163. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  164. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  165. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  166. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  167. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  168. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
  169. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  170. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  171. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  172. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  173. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  174. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  175. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  176. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  177. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  178. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  179. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  180. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  181. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  182. data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
  183. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  184. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  185. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  186. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  187. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  188. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  189. data/lib/ruby_llm/protocols/converse/chat.rb +695 -0
  190. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  191. data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
  192. data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
  193. data/lib/ruby_llm/protocols/converse.rb +54 -0
  194. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  195. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  196. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  197. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  198. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  199. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  200. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  201. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  202. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  203. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  204. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  208. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  209. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  210. data/lib/ruby_llm/protocols/files.rb +119 -0
  211. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  212. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  213. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  214. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
  215. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  216. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  217. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  218. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  219. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  220. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  221. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  222. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  223. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  224. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  225. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  226. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  227. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  228. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  229. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  230. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  231. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  232. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  233. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  234. data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
  235. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  236. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  237. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  238. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  239. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  240. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  241. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  242. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  243. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  244. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  245. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  246. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  247. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  248. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  249. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  250. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  251. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  252. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  253. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  254. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  255. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  256. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  257. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  258. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  259. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  260. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  261. data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
  262. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  263. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  264. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  265. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  266. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  267. data/lib/ruby_llm/protocols/responses.rb +35 -0
  268. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  269. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  270. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  271. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  272. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  273. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  274. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  275. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  276. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  277. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  278. data/lib/ruby_llm/provider.rb +560 -128
  279. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  280. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  281. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  282. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  283. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  284. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  285. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  286. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  287. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  288. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  289. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  290. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  291. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  292. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  293. data/lib/ruby_llm/providers/azure.rb +77 -78
  294. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  295. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  296. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  297. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  298. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  299. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  300. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  301. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  302. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  303. data/lib/ruby_llm/providers/cohere.rb +31 -0
  304. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  305. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  306. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  307. data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
  308. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  309. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  310. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  311. data/lib/ruby_llm/providers/gemini.rb +15 -8
  312. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  313. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  314. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  315. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  316. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  317. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  318. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  319. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  320. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  321. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  322. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  323. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  324. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  325. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  326. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  327. data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
  328. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  329. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  330. data/lib/ruby_llm/providers/mistral.rb +16 -4
  331. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  332. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  333. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  334. data/lib/ruby_llm/providers/ollama.rb +9 -8
  335. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  336. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  337. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  338. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  339. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  340. data/lib/ruby_llm/providers/openai.rb +92 -11
  341. data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
  342. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  343. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  344. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  345. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  346. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  347. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  348. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  349. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  350. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  351. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  352. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  353. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  354. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  355. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  356. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  357. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  358. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  359. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  360. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  361. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  362. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  363. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  364. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  365. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  366. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  367. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  368. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  369. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  370. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  371. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  372. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  373. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  374. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  375. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  376. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  377. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  378. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  379. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  380. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  381. data/lib/ruby_llm/providers/xai.rb +15 -5
  382. data/lib/ruby_llm/railtie.rb +7 -16
  383. data/lib/ruby_llm/rerank.rb +105 -0
  384. data/lib/ruby_llm/research_job.rb +241 -0
  385. data/lib/ruby_llm/search_results.rb +68 -0
  386. data/lib/ruby_llm/server_tool_call.rb +73 -0
  387. data/lib/ruby_llm/speech.rb +159 -0
  388. data/lib/ruby_llm/speech_chunk.rb +33 -0
  389. data/lib/ruby_llm/support/deprecator.rb +22 -0
  390. data/lib/ruby_llm/support/inspectable.rb +49 -0
  391. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  392. data/lib/ruby_llm/support/utils.rb +147 -0
  393. data/lib/ruby_llm/thinking.rb +127 -20
  394. data/lib/ruby_llm/tokenization.rb +59 -0
  395. data/lib/ruby_llm/tokens.rb +103 -33
  396. data/lib/ruby_llm/tool.rb +266 -91
  397. data/lib/ruby_llm/tool_call.rb +36 -3
  398. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  399. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  400. data/lib/ruby_llm/transcription.rb +138 -13
  401. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  402. data/lib/ruby_llm/transport/connection.rb +193 -0
  403. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  404. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  405. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  406. data/lib/ruby_llm/uploaded_file.rb +144 -0
  407. data/lib/ruby_llm/version.rb +2 -1
  408. data/lib/ruby_llm/video.rb +136 -0
  409. data/lib/ruby_llm/video_job.rb +150 -0
  410. data/lib/ruby_llm/workflow.rb +91 -0
  411. data/lib/ruby_llm.rb +380 -6
  412. data/lib/tasks/ruby_llm.rake +21 -16
  413. data/skills/rubyllm/SKILL.md +81 -0
  414. data/skills/rubyllm/agents/openai.yaml +4 -0
  415. metadata +339 -97
  416. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  417. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  418. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  419. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  420. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  421. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  422. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  423. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  424. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  425. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  426. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  427. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  428. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  429. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  430. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  431. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  432. data/lib/ruby_llm/aliases.rb +0 -41
  433. data/lib/ruby_llm/connection.rb +0 -159
  434. data/lib/ruby_llm/content.rb +0 -91
  435. data/lib/ruby_llm/deprecator.rb +0 -24
  436. data/lib/ruby_llm/error_middleware.rb +0 -81
  437. data/lib/ruby_llm/instrumentation.rb +0 -36
  438. data/lib/ruby_llm/mime_type.rb +0 -96
  439. data/lib/ruby_llm/model/info.rb +0 -164
  440. data/lib/ruby_llm/model_registry.rb +0 -39
  441. data/lib/ruby_llm/models_schema.json +0 -171
  442. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  443. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  444. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  445. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  446. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  447. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  448. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  449. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  450. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  451. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  452. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  453. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  454. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  455. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  456. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  457. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  458. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  459. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  460. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  461. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  462. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  463. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  464. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  465. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  466. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  467. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  468. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  469. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  470. data/lib/ruby_llm/streaming.rb +0 -179
  471. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  472. data/lib/ruby_llm/utils.rb +0 -130
  473. data/lib/tasks/models.rake +0 -593
  474. data/lib/tasks/release.rake +0 -94
  475. data/lib/tasks/vcr.rake +0 -124
@@ -5,142 +5,85 @@ module RubyLLM
5
5
  class OpenRouter
6
6
  # Chat methods of the OpenRouter API integration
7
7
  module Chat
8
- module_function
9
-
10
- # rubocop:disable Metrics/ParameterLists,Metrics/PerceivedComplexity
11
- def render_payload(messages, tools:, temperature:, model:, stream: false, schema: nil,
12
- thinking: nil, tool_prefs: nil)
13
- tool_prefs ||= {}
14
- payload = {
15
- model: model.id,
16
- messages: format_messages(messages),
17
- stream: stream
18
- }
19
-
20
- payload[:temperature] = temperature unless temperature.nil?
21
- if tools.any?
22
- payload[:tools] = tools.map { |_, tool| OpenAI::Tools.tool_for(tool) }
23
- payload[:tool_choice] = OpenAI::Tools.build_tool_choice(tool_prefs[:choice]) unless tool_prefs[:choice].nil?
24
- payload[:parallel_tool_calls] = tool_prefs[:calls] == :many unless tool_prefs[:calls].nil?
25
- end
8
+ OPENROUTER_INLINE_FILE_THRESHOLD = 50 * 1024 * 1024
9
+ OPENROUTER_FILE_UPLOAD_LIMIT = 100 * 1024 * 1024
10
+ CACHE_CONTROL_TYPE = 'ephemeral'
11
+ PROMPT_CACHE_OPTIONS = %i[ttl].freeze
12
+ COMPACTION_PLUGIN_ID = 'context-compression'
26
13
 
27
- if schema
28
- schema_name = schema[:name]
29
- schema_def = RubyLLM::Utils.deep_dup(schema[:schema])
30
- if schema_def.is_a?(Hash)
31
- schema_def.delete(:strict)
32
- schema_def.delete('strict')
33
- end
34
- strict = schema[:strict]
35
- payload[:response_format] = {
36
- type: 'json_schema',
37
- json_schema: {
38
- name: schema_name,
39
- schema: schema_def,
40
- strict: strict
41
- }
42
- }
43
- end
44
-
45
- reasoning = build_reasoning(thinking)
46
- payload[:reasoning] = reasoning if reasoning
14
+ module_function
47
15
 
48
- payload[:stream_options] = { include_usage: true } if stream
49
- payload
16
+ def apply_end_user(payload, identifier)
17
+ payload.merge(user: identifier)
50
18
  end
51
- # rubocop:enable Metrics/ParameterLists,Metrics/PerceivedComplexity
52
-
53
- def parse_completion_response(response)
54
- data = response.body
55
- return if data.nil? || data.empty?
56
-
57
- raise Error.new(response, data.dig('error', 'message')) if data.dig('error', 'message')
58
19
 
59
- message_data = data.dig('choices', 0, 'message')
60
- return unless message_data
61
-
62
- usage = data['usage'] || {}
63
- thinking_tokens = thinking_tokens(usage)
64
- thinking_text = extract_thinking_text(message_data)
65
- thinking_signature = extract_thinking_signature(message_data)
66
-
67
- Message.new(
68
- role: :assistant,
69
- content: message_data['content'],
70
- thinking: Thinking.build(text: thinking_text, signature: thinking_signature),
71
- tool_calls: OpenAI::Tools.parse_tool_calls(message_data['tool_calls']),
72
- input_tokens: input_tokens(usage),
73
- output_tokens: output_tokens(usage),
74
- cached_tokens: cache_read_tokens(usage),
75
- cache_creation_tokens: cache_write_tokens(usage),
76
- thinking_tokens: thinking_tokens,
77
- model_id: data['model'],
78
- raw: response
79
- )
20
+ # OpenRouter compacts through its context-compression plugin, which
21
+ # drops messages from the middle of the conversation once the prompt
22
+ # would overflow the model's context window. It summarizes nothing
23
+ # and takes no threshold of its own, so the portable options have
24
+ # nowhere to go.
25
+ def apply_compaction(payload, compaction)
26
+ log_ignored_compaction_options(compaction)
27
+ payload.merge(plugins: Array(payload[:plugins]) + [{ id: COMPACTION_PLUGIN_ID }])
80
28
  end
81
29
 
82
- def input_tokens(usage)
83
- return usage['prompt_cache_miss_tokens'] if usage['prompt_cache_miss_tokens']
30
+ def log_ignored_compaction_options(compaction)
31
+ return if compaction.empty?
84
32
 
85
- prompt_tokens = usage['prompt_tokens']
86
- return unless prompt_tokens
87
-
88
- [prompt_tokens.to_i - cache_read_tokens(usage).to_i - cache_write_tokens(usage).to_i, 0].max
89
- end
90
-
91
- def output_tokens(usage)
92
- OpenAI::Chat.output_tokens(usage)
93
- end
94
-
95
- def cache_read_tokens(usage)
96
- usage.dig('prompt_tokens_details', 'cached_tokens') || usage['prompt_cache_hit_tokens']
33
+ RubyLLM.logger.debug do
34
+ "#{@provider.name} compresses context at the model's own limit, dropping #{compaction.inspect}"
35
+ end
97
36
  end
98
37
 
99
- def cache_write_tokens(usage)
100
- usage.dig('prompt_tokens_details', 'cache_write_tokens') || 0
38
+ def format_content(content, attachments = [])
39
+ OpenRouter::Media.format_content(content, attachments)
101
40
  end
102
41
 
103
- def thinking_tokens(usage)
104
- OpenAI::Chat.thinking_tokens(usage)
105
- end
42
+ def render_payload(messages, tools:, temperature:, model:, stream: false, max_output_tokens: nil, schema: nil,
43
+ thinking: nil, citations: false, caching: nil, tool_prefs: nil)
44
+ payload = super
45
+ payload.delete(:reasoning_effort)
46
+ strip_schema_strict(payload)
106
47
 
107
- def format_messages(messages)
108
- messages.map do |msg|
109
- {
110
- role: format_role(msg.role),
111
- content: format_content(msg.content),
112
- tool_calls: OpenAI::Tools.format_tool_calls(msg.tool_calls),
113
- tool_call_id: msg.tool_call_id
114
- }.compact.merge(format_thinking(msg))
115
- end
48
+ reasoning = build_reasoning(thinking)
49
+ payload[:reasoning] = reasoning if reasoning
50
+ payload[:cache_control] = prompt_cache_control(caching) if caching && !cache_boundaries?(messages)
51
+ payload
116
52
  end
117
53
 
118
- def format_content(content)
119
- OpenAI::Media.format_content(content)
120
- end
54
+ def strip_schema_strict(payload)
55
+ schema_def = payload.dig(:response_format, :json_schema, :schema)
56
+ return unless schema_def.is_a?(Hash)
121
57
 
122
- def format_role(role)
123
- case role
124
- when :system
125
- @config.openai_use_system_role ? 'system' : 'developer'
126
- else
127
- role.to_s
128
- end
58
+ schema_def = RubyLLM::Support::Utils.deep_dup(schema_def)
59
+ schema_def.delete(:strict)
60
+ schema_def.delete('strict')
61
+ payload[:response_format][:json_schema][:schema] = schema_def
129
62
  end
130
63
 
131
64
  def build_reasoning(thinking)
132
65
  return nil unless thinking&.enabled?
133
66
 
134
67
  reasoning = {}
135
- reasoning[:effort] = thinking.effort if thinking.respond_to?(:effort) && thinking.effort
68
+ reasoning[:effort] = thinking.effort.to_s if thinking.respond_to?(:effort) && thinking.effort
136
69
  reasoning[:max_tokens] = thinking.budget if thinking.respond_to?(:budget) && thinking.budget
70
+ add_reasoning_toggle(reasoning, thinking)
137
71
  reasoning[:enabled] = true if reasoning.empty?
138
72
  reasoning
139
73
  end
140
74
 
75
+ def add_reasoning_toggle(reasoning, thinking)
76
+ return unless thinking.respond_to?(:enabled) && !thinking.enabled.nil?
77
+
78
+ reasoning[:enabled] = thinking.enabled
79
+ end
80
+
141
81
  def format_thinking(msg)
82
+ return {} unless msg.role == :assistant
83
+ return { reasoning_details: msg.raw_reasoning } if msg.raw_reasoning.is_a?(Array)
84
+
142
85
  thinking = msg.thinking
143
- return {} unless thinking && msg.role == :assistant
86
+ return {} unless thinking
144
87
 
145
88
  details = []
146
89
  if thinking.text
@@ -159,6 +102,78 @@ module RubyLLM
159
102
  details.empty? ? {} : { reasoning_details: details }
160
103
  end
161
104
 
105
+ def format_message_content(msg, caching: nil)
106
+ content = super
107
+ caching != false && msg.cache_until_here? ? inject_cache_control(content, caching:) : content
108
+ end
109
+
110
+ def inject_cache_control(content, caching: nil)
111
+ blocks = content.is_a?(Array) ? content.dup : [{ type: 'text', text: content }]
112
+ return blocks if blocks.empty?
113
+
114
+ last = blocks.last
115
+ return blocks unless last.is_a?(Hash)
116
+ return blocks if last[:cache_control] || last['cache_control']
117
+
118
+ blocks[-1] = last.merge(cache_control: prompt_cache_control(caching))
119
+ blocks
120
+ end
121
+
122
+ def prompt_cache_control(caching = nil)
123
+ options = prompt_cache_options(caching)
124
+
125
+ { type: CACHE_CONTROL_TYPE }.tap do |control|
126
+ control[:ttl] = options[:ttl] if options[:ttl]
127
+ end
128
+ end
129
+
130
+ def prompt_cache_options(caching)
131
+ return {} unless caching
132
+
133
+ options = caching.to_h.transform_keys(&:to_sym)
134
+ unsupported = options.keys - PROMPT_CACHE_OPTIONS
135
+ return options if unsupported.empty?
136
+
137
+ raise ArgumentError,
138
+ "OpenRouter prompt caching accepts :ttl, got #{format_cache_option_keys(unsupported)}"
139
+ end
140
+
141
+ def format_cache_option_keys(keys)
142
+ keys.map { |key| ":#{key}" }.join(', ')
143
+ end
144
+
145
+ def cache_boundaries?(messages)
146
+ messages.any?(&:cache_until_here?)
147
+ end
148
+
149
+ def openai_prompt_caching?
150
+ false
151
+ end
152
+
153
+ def reported_cost(usage)
154
+ cost = usage['cost']
155
+ return nil unless cost
156
+
157
+ cost += usage.dig('cost_details', 'upstream_inference_cost').to_f if usage['is_byok']
158
+ cost
159
+ end
160
+
161
+ def supports_provider_file_references?
162
+ true
163
+ end
164
+
165
+ def default_large_file_upload_threshold
166
+ OPENROUTER_INLINE_FILE_THRESHOLD
167
+ end
168
+
169
+ def provider_file_upload_limit
170
+ OPENROUTER_FILE_UPLOAD_LIMIT
171
+ end
172
+
173
+ def provider_file_attachable?(attachment)
174
+ attachment.pdf?
175
+ end
176
+
162
177
  def extract_thinking_text(message_data)
163
178
  candidate = message_data['reasoning']
164
179
  return candidate if candidate.is_a?(String)
@@ -190,6 +205,13 @@ module RubyLLM
190
205
  encrypted = details.find { |detail| detail['type'] == 'reasoning.encrypted' && detail['data'].is_a?(String) }
191
206
  encrypted&.dig('data')
192
207
  end
208
+
209
+ # OpenRouter requires the reasoning_details array back untouched for
210
+ # signed reasoning to survive multi-turn tool calls.
211
+ def extract_raw_reasoning(message_data)
212
+ details = message_data['reasoning_details']
213
+ details if details.is_a?(Array) && !details.empty?
214
+ end
193
215
  end
194
216
  end
195
217
  end
@@ -0,0 +1,51 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class OpenRouter
6
+ # Text and multimodal embedding requests for OpenRouter.
7
+ module Embeddings
8
+ EMBEDDING_MEDIA_TYPES = { audio: :input_audio, video: :input_video, pdf: :input_file }.freeze
9
+
10
+ module_function
11
+
12
+ def supports_embedding_media?
13
+ true
14
+ end
15
+
16
+ def render_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil, with: [],
17
+ provider_options: {})
18
+ input = if with.any?
19
+ raise ArgumentError, 'embed one text at a time when embedding attachments' if text.is_a?(Array)
20
+
21
+ [{ content: format_embedding_content(text, with) }]
22
+ else
23
+ text
24
+ end
25
+
26
+ payload = super(input, model:, dimensions:, task_type:, title:, provider_options: {})
27
+ payload[:input_type] = task_type if task_type
28
+ payload.merge(provider_options)
29
+ end
30
+
31
+ def format_embedding_content(text, attachments)
32
+ Protocols::ChatCompletions::Media.format_parts(text, attachments) do |attachment|
33
+ format_embedding_attachment(attachment)
34
+ end
35
+ end
36
+
37
+ def format_embedding_attachment(attachment)
38
+ raise UnsupportedAttachmentError, attachment.mime_type if attachment.provider_file?
39
+
40
+ if (type = EMBEDDING_MEDIA_TYPES[attachment.type])
41
+ { type: type.to_s, type => { data: attachment.for_llm, format: attachment.format } }
42
+ else
43
+ Protocols::ChatCompletions::Media.format_attachment(
44
+ attachment, document_attachments: :none, image_attachments: true, audio_attachments: false
45
+ )
46
+ end
47
+ end
48
+ end
49
+ end
50
+ end
51
+ end
@@ -4,64 +4,65 @@ module RubyLLM
4
4
  module Providers
5
5
  class OpenRouter
6
6
  # Image generation methods for the OpenRouter API integration.
7
- # OpenRouter uses the chat completions endpoint for image generation
8
- # instead of a dedicated images endpoint.
7
+ # OpenRouter has a unified images endpoint that generates and edits
8
+ # images across providers and reports the exact cost of each call.
9
9
  module Images
10
10
  module_function
11
11
 
12
12
  def images_url(with: nil, mask: nil) # rubocop:disable Lint/UnusedMethodArgument
13
- 'chat/completions'
13
+ 'images'
14
14
  end
15
15
 
16
- def render_image_payload(prompt, model:, size:, with: nil, mask: nil, params: {}) # rubocop:disable Lint/UnusedMethodArgument,Metrics/ParameterLists
17
- RubyLLM.logger.debug { "Ignoring size #{size}. OpenRouter image generation does not support size parameter." }
18
- {
19
- model: model,
20
- messages: [
21
- {
22
- role: 'user',
23
- content: prompt
24
- }
25
- ],
26
- modalities: %w[image text]
27
- }
16
+ def render_image_payload(prompt, model:, size:, count: nil, with: nil, mask: nil, provider_options: {}) # rubocop:disable Lint/UnusedMethodArgument
17
+ RubyLLM.logger.debug { "Ignoring size #{size}. Use aspect_ratio/resolution provider options instead." }
18
+ if count && count > 1
19
+ RubyLLM.logger.debug do
20
+ "Ignoring count #{count}. OpenRouter generates one image per request."
21
+ end
22
+ end
23
+ payload = { model: model, prompt: prompt }
24
+ references = build_input_references(with)
25
+ payload[:input_references] = references if references.any?
26
+ payload.merge(provider_options)
28
27
  end
29
28
 
30
- def parse_image_response(response, model:)
31
- data = response.body
32
- message = data.dig('choices', 0, 'message')
29
+ def build_input_references(with)
30
+ Attachment.wrap(with, config: @config).map do |attachment|
31
+ raise UnsupportedAttachmentError, attachment.mime_type unless attachment.image?
33
32
 
34
- unless message&.key?('images') && message['images']&.any?
35
- raise Error.new(nil, 'Unexpected response format from OpenRouter image generation API')
33
+ Protocols::ChatCompletions::Media.format_image(attachment)
36
34
  end
35
+ end
37
36
 
38
- image_data = message['images'].first
39
- image_url = image_data.dig('image_url', 'url') || image_data['url']
40
-
41
- raise Error.new(nil, 'No image URL found in OpenRouter response') unless image_url
37
+ def validate_paint_inputs!(with:, mask:) # rubocop:disable Lint/UnusedMethodArgument
38
+ raise UnsupportedAttachmentError, 'image mask' unless mask.nil?
39
+ end
42
40
 
43
- build_image_from_url(image_url, model)
41
+ # OpenRouter's images endpoint returns one image per request.
42
+ def parse_image_responses(response, model:)
43
+ [parse_image_response(response, model:)]
44
44
  end
45
45
 
46
- def build_image_from_url(image_url, model)
47
- if image_url.start_with?('data:')
48
- # Parse data URL format: data:image/png;base64,<data>
49
- match = image_url.match(/^data:([^;]+);base64,(.+)$/)
50
- raise Error.new(nil, 'Invalid data URL format from OpenRouter') unless match
46
+ def parse_image_response(response, model:)
47
+ data = response.body
48
+ image_data = Array(data['data']).first
51
49
 
52
- Image.new(
53
- data: match[2],
54
- mime_type: match[1],
55
- model_id: model
56
- )
57
- else
58
- # Regular URL
59
- Image.new(
60
- url: image_url,
61
- mime_type: 'image/png',
62
- model_id: model
63
- )
64
- end
50
+ raise Error, 'Unexpected response format from OpenRouter image API' unless image_data
51
+
52
+ Image.new(
53
+ data: image_data['b64_json'],
54
+ mime_type: image_data['media_type'] || 'image/png',
55
+ model: model,
56
+ usage: image_usage(data['usage'] || {})
57
+ )
58
+ end
59
+
60
+ def image_usage(usage)
61
+ {
62
+ 'input_tokens' => usage['prompt_tokens'],
63
+ 'output_tokens' => usage['completion_tokens'],
64
+ 'cost' => reported_cost(usage)
65
+ }.compact
65
66
  end
66
67
  end
67
68
  end
@@ -0,0 +1,34 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class OpenRouter
6
+ # Handles media content for OpenRouter, which adds video input parts
7
+ # to the standard Chat Completions vocabulary.
8
+ module Media
9
+ module_function
10
+
11
+ def format_content(content, attachments = [])
12
+ Protocols::ChatCompletions::Media.format_parts(content, attachments) do |attachment|
13
+ if attachment.type == :video
14
+ format_video(attachment)
15
+ else
16
+ Protocols::ChatCompletions::Media.format_attachment(
17
+ attachment, document_attachments: :pdf, image_attachments: true, audio_attachments: true
18
+ )
19
+ end
20
+ end
21
+ end
22
+
23
+ def format_video(video)
24
+ {
25
+ type: 'video_url',
26
+ video_url: {
27
+ url: video.url? ? video.source.to_s : video.for_llm
28
+ }
29
+ }
30
+ end
31
+ end
32
+ end
33
+ end
34
+ end
@@ -3,19 +3,56 @@
3
3
  module RubyLLM
4
4
  module Providers
5
5
  class OpenRouter
6
- # Models methods of the OpenRouter API integration
6
+ # Models methods of the OpenRouter API integration. Chat models come
7
+ # from the main catalog; embedding, speech, transcription, and image
8
+ # models live in separate catalogs that are merged in.
7
9
  module Models
10
+ CATALOG_URLS = [
11
+ 'models',
12
+ 'embeddings/models',
13
+ 'models?output_modalities=speech',
14
+ 'models?output_modalities=transcription',
15
+ 'models?output_modalities=rerank',
16
+ 'images/models'
17
+ ].freeze
18
+
19
+ OUTPUT_MODALITY_MAP = {
20
+ 'speech' => 'audio',
21
+ 'transcription' => 'text'
22
+ }.freeze
23
+
24
+ CAPABILITY_BY_OUTPUT_MODALITY = {
25
+ 'speech' => 'speech_generation',
26
+ 'transcription' => 'transcription',
27
+ 'image' => 'image_generation'
28
+ }.freeze
29
+
30
+ CAPABILITY_PARAMETERS = {
31
+ 'function_calling' => %w[tools tool_choice],
32
+ 'tool_choice' => %w[tool_choice],
33
+ 'parallel_tool_calls' => %w[parallel_tool_calls],
34
+ 'structured_output' => %w[response_format structured_outputs],
35
+ 'batch' => %w[batch]
36
+ }.freeze
37
+
38
+ def list_models
39
+ CATALOG_URLS.flat_map do |url|
40
+ parse_list_models_response @connection.get(url), @provider.slug
41
+ end.uniq(&:id)
42
+ end
43
+
8
44
  module_function
9
45
 
10
46
  def models_url
11
47
  'models'
12
48
  end
13
49
 
14
- def parse_list_models_response(response, slug, _capabilities)
50
+ def parse_list_models_response(response, slug)
15
51
  Array(response.body['data']).map do |model_data| # rubocop:disable Metrics/BlockLength
52
+ output_modalities = Array(model_data.dig('architecture', 'output_modalities'))
16
53
  modalities = {
17
54
  input: Array(model_data.dig('architecture', 'input_modalities')),
18
- output: Array(model_data.dig('architecture', 'output_modalities'))
55
+ output: output_modalities.map { |modality| OUTPUT_MODALITY_MAP.fetch(modality, modality) }.uniq
19
56
  }
20
57
 
21
58
  pricing = { text_tokens: { standard: {} } }
@@ -24,6 +61,7 @@ module RubyLLM
24
61
  prompt: :input_per_million,
25
62
  completion: :output_per_million,
26
63
  input_cache_read: :cache_read_input_per_million,
64
+ input_cache_write: :cache_write_input_per_million,
27
65
  internal_reasoning: :reasoning_output_per_million
28
66
  }
29
67
 
@@ -32,9 +70,10 @@ module RubyLLM
32
70
  pricing[:text_tokens][:standard][target_key] = value * 1_000_000 if value.positive?
33
71
  end
34
72
 
35
- capabilities = supported_parameters_to_capabilities(model_data['supported_parameters'])
73
+ capabilities = supported_parameters_to_capabilities(model_data['supported_parameters']) |
74
+ output_modalities.filter_map { |modality| CAPABILITY_BY_OUTPUT_MODALITY[modality] }
36
75
 
37
- Model::Info.new(
76
+ Model.new(
38
77
  id: model_data['id'],
39
78
  name: model_data['name'],
40
79
  provider: slug,
@@ -42,6 +81,7 @@ module RubyLLM
42
81
  created_at: model_data['created'] ? Time.at(model_data['created']) : nil,
43
82
  context_window: model_data['context_length'],
44
83
  max_output_tokens: model_data.dig('top_provider', 'max_completion_tokens'),
84
+ knowledge_cutoff: model_data['knowledge_cutoff'],
45
85
  modalities: modalities,
46
86
  capabilities: capabilities,
47
87
  pricing: pricing,
@@ -50,7 +90,8 @@ module RubyLLM
50
90
  architecture: model_data['architecture'],
51
91
  top_provider: model_data['top_provider'],
52
92
  per_request_limits: model_data['per_request_limits'],
53
- supported_parameters: model_data['supported_parameters']
93
+ supported_parameters: model_data['supported_parameters'],
94
+ expiration_date: model_data['expiration_date']
54
95
  }
55
96
  )
56
97
  end
@@ -59,11 +100,9 @@ module RubyLLM
59
100
  def supported_parameters_to_capabilities(params)
60
101
  return [] unless params
61
102
 
62
- capabilities = []
63
- capabilities << 'streaming'
64
- capabilities << 'function_calling' if params.include?('tools') || params.include?('tool_choice')
65
- capabilities << 'structured_output' if params.include?('response_format')
66
- capabilities << 'batch' if params.include?('batch')
103
+ capabilities = ['streaming'] + CAPABILITY_PARAMETERS.filter_map do |capability, parameters|
104
+ capability if parameters.any? { |parameter| params.include?(parameter) }
105
+ end
67
106
  capabilities << 'predicted_outputs' if params.include?('logit_bias') && params.include?('top_k')
68
107
  capabilities
69
108
  end
@@ -0,0 +1,32 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class OpenRouter
6
+ # Speech generation methods for the OpenRouter API integration.
7
+ # Voices are model-specific across OpenRouter's TTS catalog, so no
8
+ # default voice is assumed.
9
+ module Speech
10
+ module_function
11
+
12
+ def render_speech_payload(input, model:, voice:, format:, provider_options: {})
13
+ {
14
+ model: model,
15
+ input: input,
16
+ voice: voice,
17
+ response_format: format || 'mp3'
18
+ }.compact.merge(provider_options)
19
+ end
20
+
21
+ def parse_speech_response(response, model:, voice:, format:)
22
+ RubyLLM::Speech.new(
23
+ data: response.body,
24
+ model: model,
25
+ voice: voice,
26
+ format: (format || 'mp3').to_s
27
+ )
28
+ end
29
+ end
30
+ end
31
+ end
32
+ end