ruby_llm 1.16.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (480) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +108 -44
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +109 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +3 -1
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +15 -9
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/legacy_content_sql.rb +34 -0
  91. data/lib/generators/ruby_llm/upgrade/online_copy_migration/data.rb +341 -0
  92. data/lib/generators/ruby_llm/upgrade/online_copy_migration/journal.rb +171 -0
  93. data/lib/generators/ruby_llm/upgrade/online_copy_migration/verification.rb +197 -0
  94. data/lib/generators/ruby_llm/upgrade/online_copy_migration.rb +174 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +468 -0
  96. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
  97. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +247 -0
  98. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +662 -0
  99. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +251 -0
  100. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  101. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +188 -0
  102. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +359 -0
  103. data/lib/ruby_llm/accounting/usage.rb +254 -0
  104. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  105. data/lib/ruby_llm/active_record/attachment_helpers.rb +264 -0
  106. data/lib/ruby_llm/active_record/batch.rb +97 -0
  107. data/lib/ruby_llm/active_record/chat_methods.rb +831 -305
  108. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  109. data/lib/ruby_llm/active_record/model.rb +135 -0
  110. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  111. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  112. data/lib/ruby_llm/active_record/usage.rb +61 -0
  113. data/lib/ruby_llm/agent.rb +1066 -151
  114. data/lib/ruby_llm/aliases.json +291 -101
  115. data/lib/ruby_llm/attachment.rb +192 -48
  116. data/lib/ruby_llm/batch.rb +432 -0
  117. data/lib/ruby_llm/cached_content.rb +112 -0
  118. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  119. data/lib/ruby_llm/chat.rb +1131 -198
  120. data/lib/ruby_llm/chunk.rb +10 -0
  121. data/lib/ruby_llm/citation.rb +105 -0
  122. data/lib/ruby_llm/configuration.rb +261 -24
  123. data/lib/ruby_llm/context.rb +128 -6
  124. data/lib/ruby_llm/cost.rb +217 -80
  125. data/lib/ruby_llm/downloaded_file.rb +33 -0
  126. data/lib/ruby_llm/embedding.rb +121 -17
  127. data/lib/ruby_llm/embedding_request.rb +53 -0
  128. data/lib/ruby_llm/error.rb +159 -23
  129. data/lib/ruby_llm/fallback.rb +133 -0
  130. data/lib/ruby_llm/files/mime_type.rb +97 -0
  131. data/lib/ruby_llm/image.rb +153 -32
  132. data/lib/ruby_llm/message.rb +255 -53
  133. data/lib/ruby_llm/model/modalities.rb +17 -4
  134. data/lib/ruby_llm/model/pricing.rb +24 -5
  135. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  136. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  137. data/lib/ruby_llm/model.rb +244 -2
  138. data/lib/ruby_llm/models/aliases.rb +41 -0
  139. data/lib/ruby_llm/models/registry.rb +165 -0
  140. data/lib/ruby_llm/models/schema.rb +99 -0
  141. data/lib/ruby_llm/models.json +76038 -34173
  142. data/lib/ruby_llm/models.rb +477 -215
  143. data/lib/ruby_llm/moderation.rb +139 -26
  144. data/lib/ruby_llm/ocr.rb +112 -0
  145. data/lib/ruby_llm/prompt.rb +79 -0
  146. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  147. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  148. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  149. data/lib/ruby_llm/protocol.rb +685 -0
  150. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  151. data/lib/ruby_llm/protocols/anthropic/chat.rb +543 -0
  152. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  153. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  154. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  155. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  156. data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
  157. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  158. data/lib/ruby_llm/protocols/anthropic.rb +101 -0
  159. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  160. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +110 -0
  161. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  162. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  163. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  164. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  165. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  166. data/lib/ruby_llm/protocols/chat_completions/chat.rb +484 -0
  167. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  168. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  169. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  170. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  171. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  172. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  173. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +63 -0
  174. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  175. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  176. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  177. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  178. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  179. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  180. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  181. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  182. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  183. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  184. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  185. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  186. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  187. data/lib/ruby_llm/protocols/cohere/rerank.rb +59 -0
  188. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  189. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  190. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  191. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  192. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  193. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  194. data/lib/ruby_llm/protocols/converse/chat.rb +694 -0
  195. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  196. data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
  197. data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
  198. data/lib/ruby_llm/protocols/converse.rb +54 -0
  199. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  200. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  201. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  202. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  203. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  204. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  208. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  209. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  210. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  211. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  212. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  213. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  214. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  215. data/lib/ruby_llm/protocols/files.rb +119 -0
  216. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  217. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  218. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  219. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +91 -0
  220. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  221. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  222. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  223. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  224. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  225. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  226. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  227. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  228. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  229. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  230. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  231. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  232. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  233. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  234. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  235. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  236. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  237. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  238. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  239. data/lib/ruby_llm/protocols/interactions/tools.rb +53 -0
  240. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  241. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  242. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  243. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  244. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  245. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  246. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  247. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  248. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  249. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  250. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  251. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  252. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  253. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  254. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  255. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  256. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  257. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  258. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  259. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  260. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  261. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  262. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  263. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  264. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  265. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  266. data/lib/ruby_llm/protocols/responses/chat.rb +466 -0
  267. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  268. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  269. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  270. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  271. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  272. data/lib/ruby_llm/protocols/responses.rb +35 -0
  273. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  274. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  275. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  276. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  277. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  278. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  279. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  280. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  281. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  282. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  283. data/lib/ruby_llm/provider.rb +560 -128
  284. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  285. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  286. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  287. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  288. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  289. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  290. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  291. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  292. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  293. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  294. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  295. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  296. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  297. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  298. data/lib/ruby_llm/providers/azure.rb +77 -78
  299. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  300. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  301. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  302. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  303. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  304. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  305. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  306. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  307. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  308. data/lib/ruby_llm/providers/cohere.rb +31 -0
  309. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  310. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  311. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  312. data/lib/ruby_llm/providers/deepseek/responses.rb +67 -0
  313. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  314. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  315. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  316. data/lib/ruby_llm/providers/gemini.rb +15 -8
  317. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  318. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  319. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  320. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  321. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  322. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  323. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  324. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  325. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  326. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  327. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  328. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  329. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  330. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  331. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  332. data/lib/ruby_llm/providers/mistral/ocr.rb +51 -0
  333. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  334. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  335. data/lib/ruby_llm/providers/mistral.rb +16 -4
  336. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  337. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  338. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  339. data/lib/ruby_llm/providers/ollama.rb +9 -8
  340. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  341. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  342. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  343. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  344. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  345. data/lib/ruby_llm/providers/openai.rb +92 -11
  346. data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
  347. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  348. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  349. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  350. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  351. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  352. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  353. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  354. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  355. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  356. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  357. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  358. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  359. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  360. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  361. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  362. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  363. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  364. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  365. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  366. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  367. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  368. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  369. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  370. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  371. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  372. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  373. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  374. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  375. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  376. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  377. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  378. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  379. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  380. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  381. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  382. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  383. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  384. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  385. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  386. data/lib/ruby_llm/providers/xai.rb +15 -5
  387. data/lib/ruby_llm/railtie.rb +7 -16
  388. data/lib/ruby_llm/rerank.rb +105 -0
  389. data/lib/ruby_llm/research_job.rb +241 -0
  390. data/lib/ruby_llm/search_results.rb +68 -0
  391. data/lib/ruby_llm/server_tool_call.rb +73 -0
  392. data/lib/ruby_llm/speech.rb +159 -0
  393. data/lib/ruby_llm/speech_chunk.rb +33 -0
  394. data/lib/ruby_llm/support/deprecator.rb +22 -0
  395. data/lib/ruby_llm/support/inspectable.rb +49 -0
  396. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  397. data/lib/ruby_llm/support/utils.rb +147 -0
  398. data/lib/ruby_llm/thinking.rb +127 -20
  399. data/lib/ruby_llm/tokenization.rb +59 -0
  400. data/lib/ruby_llm/tokens.rb +103 -33
  401. data/lib/ruby_llm/tool.rb +266 -91
  402. data/lib/ruby_llm/tool_call.rb +36 -3
  403. data/lib/ruby_llm/tools/provider_tools.rb +109 -0
  404. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  405. data/lib/ruby_llm/transcription.rb +138 -13
  406. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  407. data/lib/ruby_llm/transport/connection.rb +193 -0
  408. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  409. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  410. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  411. data/lib/ruby_llm/uploaded_file.rb +144 -0
  412. data/lib/ruby_llm/version.rb +2 -1
  413. data/lib/ruby_llm/video.rb +136 -0
  414. data/lib/ruby_llm/video_job.rb +150 -0
  415. data/lib/ruby_llm/workflow.rb +91 -0
  416. data/lib/ruby_llm.rb +380 -6
  417. data/lib/tasks/ruby_llm.rake +21 -16
  418. data/skills/rubyllm/SKILL.md +81 -0
  419. data/skills/rubyllm/agents/openai.yaml +4 -0
  420. metadata +344 -97
  421. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  422. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  423. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  424. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  425. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  426. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  427. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  428. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  429. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  430. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  431. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  432. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  433. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  434. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  435. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  436. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  437. data/lib/ruby_llm/aliases.rb +0 -41
  438. data/lib/ruby_llm/connection.rb +0 -159
  439. data/lib/ruby_llm/content.rb +0 -91
  440. data/lib/ruby_llm/deprecator.rb +0 -24
  441. data/lib/ruby_llm/error_middleware.rb +0 -81
  442. data/lib/ruby_llm/instrumentation.rb +0 -36
  443. data/lib/ruby_llm/mime_type.rb +0 -96
  444. data/lib/ruby_llm/model/info.rb +0 -164
  445. data/lib/ruby_llm/model_registry.rb +0 -39
  446. data/lib/ruby_llm/models_schema.json +0 -171
  447. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  448. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  449. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  450. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  451. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  452. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  453. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  454. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  455. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  456. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  457. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  458. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  459. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  460. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  461. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  462. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  463. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  464. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  465. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  466. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  467. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  468. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  469. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  470. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  471. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  472. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  473. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  474. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  475. data/lib/ruby_llm/streaming.rb +0 -179
  476. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  477. data/lib/ruby_llm/utils.rb +0 -130
  478. data/lib/tasks/models.rake +0 -593
  479. data/lib/tasks/release.rake +0 -94
  480. data/lib/tasks/vcr.rake +0 -124
@@ -5,142 +5,89 @@ module RubyLLM
5
5
  class OpenRouter
6
6
  # Chat methods of the OpenRouter API integration
7
7
  module Chat
8
- module_function
9
-
10
- # rubocop:disable Metrics/ParameterLists,Metrics/PerceivedComplexity
11
- def render_payload(messages, tools:, temperature:, model:, stream: false, schema: nil,
12
- thinking: nil, tool_prefs: nil)
13
- tool_prefs ||= {}
14
- payload = {
15
- model: model.id,
16
- messages: format_messages(messages),
17
- stream: stream
18
- }
19
-
20
- payload[:temperature] = temperature unless temperature.nil?
21
- if tools.any?
22
- payload[:tools] = tools.map { |_, tool| OpenAI::Tools.tool_for(tool) }
23
- payload[:tool_choice] = OpenAI::Tools.build_tool_choice(tool_prefs[:choice]) unless tool_prefs[:choice].nil?
24
- payload[:parallel_tool_calls] = tool_prefs[:calls] == :many unless tool_prefs[:calls].nil?
25
- end
26
-
27
- if schema
28
- schema_name = schema[:name]
29
- schema_def = RubyLLM::Utils.deep_dup(schema[:schema])
30
- if schema_def.is_a?(Hash)
31
- schema_def.delete(:strict)
32
- schema_def.delete('strict')
33
- end
34
- strict = schema[:strict]
35
- payload[:response_format] = {
36
- type: 'json_schema',
37
- json_schema: {
38
- name: schema_name,
39
- schema: schema_def,
40
- strict: strict
41
- }
42
- }
43
- end
44
-
45
- reasoning = build_reasoning(thinking)
46
- payload[:reasoning] = reasoning if reasoning
47
-
48
- payload[:stream_options] = { include_usage: true } if stream
49
- payload
50
- end
51
- # rubocop:enable Metrics/ParameterLists,Metrics/PerceivedComplexity
52
-
53
- def parse_completion_response(response)
54
- data = response.body
55
- return if data.nil? || data.empty?
56
-
57
- raise Error.new(response, data.dig('error', 'message')) if data.dig('error', 'message')
58
-
59
- message_data = data.dig('choices', 0, 'message')
60
- return unless message_data
61
-
62
- usage = data['usage'] || {}
63
- thinking_tokens = thinking_tokens(usage)
64
- thinking_text = extract_thinking_text(message_data)
65
- thinking_signature = extract_thinking_signature(message_data)
8
+ OPENROUTER_INLINE_FILE_THRESHOLD = 50 * 1024 * 1024
9
+ OPENROUTER_FILE_UPLOAD_LIMIT = 100 * 1024 * 1024
10
+ CACHE_CONTROL_TYPE = 'ephemeral'
11
+ PROMPT_CACHE_OPTIONS = %i[ttl].freeze
12
+ COMPACTION_PLUGIN_ID = 'context-compression'
66
13
 
67
- Message.new(
68
- role: :assistant,
69
- content: message_data['content'],
70
- thinking: Thinking.build(text: thinking_text, signature: thinking_signature),
71
- tool_calls: OpenAI::Tools.parse_tool_calls(message_data['tool_calls']),
72
- input_tokens: input_tokens(usage),
73
- output_tokens: output_tokens(usage),
74
- cached_tokens: cache_read_tokens(usage),
75
- cache_creation_tokens: cache_write_tokens(usage),
76
- thinking_tokens: thinking_tokens,
77
- model_id: data['model'],
78
- raw: response
79
- )
80
- end
81
-
82
- def input_tokens(usage)
83
- return usage['prompt_cache_miss_tokens'] if usage['prompt_cache_miss_tokens']
84
-
85
- prompt_tokens = usage['prompt_tokens']
86
- return unless prompt_tokens
14
+ module_function
87
15
 
88
- [prompt_tokens.to_i - cache_read_tokens(usage).to_i - cache_write_tokens(usage).to_i, 0].max
16
+ def apply_end_user(payload, identifier)
17
+ payload.merge(user: identifier)
89
18
  end
90
19
 
91
- def output_tokens(usage)
92
- OpenAI::Chat.output_tokens(usage)
20
+ # OpenRouter compacts through its context-compression plugin, which
21
+ # drops messages from the middle of the conversation once the prompt
22
+ # would overflow the model's context window. It summarizes nothing
23
+ # and takes no threshold of its own, so the portable options have
24
+ # nowhere to go.
25
+ def apply_compaction(payload, compaction)
26
+ log_ignored_compaction_options(compaction)
27
+ payload.merge(plugins: Array(payload[:plugins]) + [{ id: COMPACTION_PLUGIN_ID }])
93
28
  end
94
29
 
95
- def cache_read_tokens(usage)
96
- usage.dig('prompt_tokens_details', 'cached_tokens') || usage['prompt_cache_hit_tokens']
97
- end
30
+ def log_ignored_compaction_options(compaction)
31
+ return if compaction.empty?
98
32
 
99
- def cache_write_tokens(usage)
100
- usage.dig('prompt_tokens_details', 'cache_write_tokens') || 0
33
+ RubyLLM.logger.debug do
34
+ "#{@provider.name} compresses context at the model's own limit, dropping #{compaction.inspect}"
35
+ end
101
36
  end
102
37
 
103
- def thinking_tokens(usage)
104
- OpenAI::Chat.thinking_tokens(usage)
38
+ def format_content(content, attachments = [])
39
+ OpenRouter::Media.format_content(content, attachments)
105
40
  end
106
41
 
107
- def format_messages(messages)
108
- messages.map do |msg|
109
- {
110
- role: format_role(msg.role),
111
- content: format_content(msg.content),
112
- tool_calls: OpenAI::Tools.format_tool_calls(msg.tool_calls),
113
- tool_call_id: msg.tool_call_id
114
- }.compact.merge(format_thinking(msg))
42
+ def render_payload(messages, tools:, temperature:, model:, stream: false, max_output_tokens: nil, schema: nil,
43
+ thinking: nil, citations: false, caching: nil, tool_prefs: nil)
44
+ payload = super
45
+ payload.delete(:reasoning_effort)
46
+ strip_schema_strict(payload)
47
+ if tool_prefs&.dig(:choice) == :none
48
+ payload.delete(:tools)
49
+ payload.delete(:parallel_tool_calls)
115
50
  end
116
- end
117
51
 
118
- def format_content(content)
119
- OpenAI::Media.format_content(content)
52
+ reasoning = build_reasoning(thinking)
53
+ payload[:reasoning] = reasoning if reasoning
54
+ payload[:cache_control] = prompt_cache_control(caching) if caching
55
+ payload
120
56
  end
121
57
 
122
- def format_role(role)
123
- case role
124
- when :system
125
- @config.openai_use_system_role ? 'system' : 'developer'
126
- else
127
- role.to_s
128
- end
58
+ def strip_schema_strict(payload)
59
+ schema_def = payload.dig(:response_format, :json_schema, :schema)
60
+ return unless schema_def.is_a?(Hash)
61
+
62
+ schema_def = RubyLLM::Support::Utils.deep_dup(schema_def)
63
+ schema_def.delete(:strict)
64
+ schema_def.delete('strict')
65
+ payload[:response_format][:json_schema][:schema] = schema_def
129
66
  end
130
67
 
131
68
  def build_reasoning(thinking)
132
69
  return nil unless thinking&.enabled?
133
70
 
134
71
  reasoning = {}
135
- reasoning[:effort] = thinking.effort if thinking.respond_to?(:effort) && thinking.effort
72
+ reasoning[:effort] = thinking.effort.to_s if thinking.respond_to?(:effort) && thinking.effort
136
73
  reasoning[:max_tokens] = thinking.budget if thinking.respond_to?(:budget) && thinking.budget
74
+ add_reasoning_toggle(reasoning, thinking)
137
75
  reasoning[:enabled] = true if reasoning.empty?
138
76
  reasoning
139
77
  end
140
78
 
79
+ def add_reasoning_toggle(reasoning, thinking)
80
+ return unless thinking.respond_to?(:enabled) && !thinking.enabled.nil?
81
+
82
+ reasoning[:enabled] = thinking.enabled
83
+ end
84
+
141
85
  def format_thinking(msg)
86
+ return {} unless msg.role == :assistant
87
+ return { reasoning_details: msg.raw_reasoning } if msg.raw_reasoning.is_a?(Array)
88
+
142
89
  thinking = msg.thinking
143
- return {} unless thinking && msg.role == :assistant
90
+ return {} unless thinking
144
91
 
145
92
  details = []
146
93
  if thinking.text
@@ -159,6 +106,74 @@ module RubyLLM
159
106
  details.empty? ? {} : { reasoning_details: details }
160
107
  end
161
108
 
109
+ def format_message_content(msg, caching: nil)
110
+ content = super
111
+ caching != false && msg.cache_until_here? ? inject_cache_control(content, caching:) : content
112
+ end
113
+
114
+ def inject_cache_control(content, caching: nil)
115
+ blocks = content.is_a?(Array) ? content.dup : [{ type: 'text', text: content }]
116
+ return blocks if blocks.empty?
117
+
118
+ last = blocks.last
119
+ return blocks unless last.is_a?(Hash)
120
+ return blocks if last[:cache_control] || last['cache_control']
121
+
122
+ blocks[-1] = last.merge(cache_control: prompt_cache_control(caching))
123
+ blocks
124
+ end
125
+
126
+ def prompt_cache_control(caching = nil)
127
+ options = prompt_cache_options(caching)
128
+
129
+ { type: CACHE_CONTROL_TYPE }.tap do |control|
130
+ control[:ttl] = options[:ttl] if options[:ttl]
131
+ end
132
+ end
133
+
134
+ def prompt_cache_options(caching)
135
+ return {} unless caching
136
+
137
+ options = caching.to_h.transform_keys(&:to_sym)
138
+ unsupported = options.keys - PROMPT_CACHE_OPTIONS
139
+ return options if unsupported.empty?
140
+
141
+ raise ArgumentError,
142
+ "OpenRouter prompt caching accepts :ttl, got #{format_cache_option_keys(unsupported)}"
143
+ end
144
+
145
+ def format_cache_option_keys(keys)
146
+ keys.map { |key| ":#{key}" }.join(', ')
147
+ end
148
+
149
+ def openai_prompt_caching?
150
+ false
151
+ end
152
+
153
+ def reported_cost(usage)
154
+ cost = usage['cost']
155
+ return nil unless cost
156
+
157
+ cost += usage.dig('cost_details', 'upstream_inference_cost').to_f if usage['is_byok']
158
+ cost
159
+ end
160
+
161
+ def supports_provider_file_references?
162
+ true
163
+ end
164
+
165
+ def default_large_file_upload_threshold
166
+ OPENROUTER_INLINE_FILE_THRESHOLD
167
+ end
168
+
169
+ def provider_file_upload_limit
170
+ OPENROUTER_FILE_UPLOAD_LIMIT
171
+ end
172
+
173
+ def provider_file_attachable?(attachment)
174
+ attachment.pdf?
175
+ end
176
+
162
177
  def extract_thinking_text(message_data)
163
178
  candidate = message_data['reasoning']
164
179
  return candidate if candidate.is_a?(String)
@@ -190,6 +205,13 @@ module RubyLLM
190
205
  encrypted = details.find { |detail| detail['type'] == 'reasoning.encrypted' && detail['data'].is_a?(String) }
191
206
  encrypted&.dig('data')
192
207
  end
208
+
209
+ # OpenRouter requires the reasoning_details array back untouched for
210
+ # signed reasoning to survive multi-turn tool calls.
211
+ def extract_raw_reasoning(message_data)
212
+ details = message_data['reasoning_details']
213
+ details if details.is_a?(Array) && !details.empty?
214
+ end
193
215
  end
194
216
  end
195
217
  end
@@ -0,0 +1,51 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class OpenRouter
6
+ # Text and multimodal embedding requests for OpenRouter.
7
+ module Embeddings
8
+ EMBEDDING_MEDIA_TYPES = { audio: :input_audio, video: :input_video, pdf: :input_file }.freeze
9
+
10
+ module_function
11
+
12
+ def supports_embedding_media?
13
+ true
14
+ end
15
+
16
+ def render_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil, with: [],
17
+ provider_options: {})
18
+ input = if with.any?
19
+ raise ArgumentError, 'embed one text at a time when embedding attachments' if text.is_a?(Array)
20
+
21
+ [{ content: format_embedding_content(text, with) }]
22
+ else
23
+ text
24
+ end
25
+
26
+ payload = super(input, model:, dimensions:, task_type:, title:, provider_options: {})
27
+ payload[:input_type] = task_type if task_type
28
+ payload.merge(provider_options)
29
+ end
30
+
31
+ def format_embedding_content(text, attachments)
32
+ Protocols::ChatCompletions::Media.format_parts(text, attachments) do |attachment|
33
+ format_embedding_attachment(attachment)
34
+ end
35
+ end
36
+
37
+ def format_embedding_attachment(attachment)
38
+ raise UnsupportedAttachmentError, attachment.mime_type if attachment.provider_file?
39
+
40
+ if (type = EMBEDDING_MEDIA_TYPES[attachment.type])
41
+ { type: type.to_s, type => { data: attachment.for_llm, format: attachment.format } }
42
+ else
43
+ Protocols::ChatCompletions::Media.format_attachment(
44
+ attachment, document_attachments: :none, image_attachments: true, audio_attachments: false
45
+ )
46
+ end
47
+ end
48
+ end
49
+ end
50
+ end
51
+ end
@@ -4,64 +4,65 @@ module RubyLLM
4
4
  module Providers
5
5
  class OpenRouter
6
6
  # Image generation methods for the OpenRouter API integration.
7
- # OpenRouter uses the chat completions endpoint for image generation
8
- # instead of a dedicated images endpoint.
7
+ # OpenRouter has a unified images endpoint that generates and edits
8
+ # images across providers and reports the exact cost of each call.
9
9
  module Images
10
10
  module_function
11
11
 
12
12
  def images_url(with: nil, mask: nil) # rubocop:disable Lint/UnusedMethodArgument
13
- 'chat/completions'
13
+ 'images'
14
14
  end
15
15
 
16
- def render_image_payload(prompt, model:, size:, with: nil, mask: nil, params: {}) # rubocop:disable Lint/UnusedMethodArgument,Metrics/ParameterLists
17
- RubyLLM.logger.debug { "Ignoring size #{size}. OpenRouter image generation does not support size parameter." }
18
- {
19
- model: model,
20
- messages: [
21
- {
22
- role: 'user',
23
- content: prompt
24
- }
25
- ],
26
- modalities: %w[image text]
27
- }
16
+ def render_image_payload(prompt, model:, size:, count: nil, with: nil, mask: nil, provider_options: {}) # rubocop:disable Lint/UnusedMethodArgument
17
+ RubyLLM.logger.debug { "Ignoring size #{size}. Use aspect_ratio/resolution provider options instead." }
18
+ if count && count > 1
19
+ RubyLLM.logger.debug do
20
+ "Ignoring count #{count}. OpenRouter generates one image per request."
21
+ end
22
+ end
23
+ payload = { model: model, prompt: prompt }
24
+ references = build_input_references(with)
25
+ payload[:input_references] = references if references.any?
26
+ payload.merge(provider_options)
28
27
  end
29
28
 
30
- def parse_image_response(response, model:)
31
- data = response.body
32
- message = data.dig('choices', 0, 'message')
29
+ def build_input_references(with)
30
+ Attachment.wrap(with, config: @config).map do |attachment|
31
+ raise UnsupportedAttachmentError, attachment.mime_type unless attachment.image?
33
32
 
34
- unless message&.key?('images') && message['images']&.any?
35
- raise Error.new(nil, 'Unexpected response format from OpenRouter image generation API')
33
+ Protocols::ChatCompletions::Media.format_image(attachment)
36
34
  end
35
+ end
37
36
 
38
- image_data = message['images'].first
39
- image_url = image_data.dig('image_url', 'url') || image_data['url']
40
-
41
- raise Error.new(nil, 'No image URL found in OpenRouter response') unless image_url
37
+ def validate_paint_inputs!(with:, mask:) # rubocop:disable Lint/UnusedMethodArgument
38
+ raise UnsupportedAttachmentError, 'image mask' unless mask.nil?
39
+ end
42
40
 
43
- build_image_from_url(image_url, model)
41
+ # OpenRouter's images endpoint returns one image per request.
42
+ def parse_image_responses(response, model:)
43
+ [parse_image_response(response, model:)]
44
44
  end
45
45
 
46
- def build_image_from_url(image_url, model)
47
- if image_url.start_with?('data:')
48
- # Parse data URL format: data:image/png;base64,<data>
49
- match = image_url.match(/^data:([^;]+);base64,(.+)$/)
50
- raise Error.new(nil, 'Invalid data URL format from OpenRouter') unless match
46
+ def parse_image_response(response, model:)
47
+ data = response.body
48
+ image_data = Array(data['data']).first
51
49
 
52
- Image.new(
53
- data: match[2],
54
- mime_type: match[1],
55
- model_id: model
56
- )
57
- else
58
- # Regular URL
59
- Image.new(
60
- url: image_url,
61
- mime_type: 'image/png',
62
- model_id: model
63
- )
64
- end
50
+ raise Error, 'Unexpected response format from OpenRouter image API' unless image_data
51
+
52
+ Image.new(
53
+ data: image_data['b64_json'],
54
+ mime_type: image_data['media_type'] || 'image/png',
55
+ model: model,
56
+ usage: image_usage(data['usage'] || {})
57
+ )
58
+ end
59
+
60
+ def image_usage(usage)
61
+ {
62
+ 'input_tokens' => usage['prompt_tokens'],
63
+ 'output_tokens' => usage['completion_tokens'],
64
+ 'cost' => reported_cost(usage)
65
+ }.compact
65
66
  end
66
67
  end
67
68
  end
@@ -0,0 +1,34 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class OpenRouter
6
+ # Handles media content for OpenRouter, which adds video input parts
7
+ # to the standard Chat Completions vocabulary.
8
+ module Media
9
+ module_function
10
+
11
+ def format_content(content, attachments = [])
12
+ Protocols::ChatCompletions::Media.format_parts(content, attachments) do |attachment|
13
+ if attachment.type == :video
14
+ format_video(attachment)
15
+ else
16
+ Protocols::ChatCompletions::Media.format_attachment(
17
+ attachment, document_attachments: :pdf, image_attachments: true, audio_attachments: true
18
+ )
19
+ end
20
+ end
21
+ end
22
+
23
+ def format_video(video)
24
+ {
25
+ type: 'video_url',
26
+ video_url: {
27
+ url: video.url? ? video.source.to_s : video.for_llm
28
+ }
29
+ }
30
+ end
31
+ end
32
+ end
33
+ end
34
+ end
@@ -3,19 +3,56 @@
3
3
  module RubyLLM
4
4
  module Providers
5
5
  class OpenRouter
6
- # Models methods of the OpenRouter API integration
6
+ # Models methods of the OpenRouter API integration. Chat models come
7
+ # from the main catalog; embedding, speech, transcription, and image
8
+ # models live in separate catalogs that are merged in.
7
9
  module Models
10
+ CATALOG_URLS = [
11
+ 'models',
12
+ 'embeddings/models',
13
+ 'models?output_modalities=speech',
14
+ 'models?output_modalities=transcription',
15
+ 'models?output_modalities=rerank',
16
+ 'images/models'
17
+ ].freeze
18
+
19
+ OUTPUT_MODALITY_MAP = {
20
+ 'speech' => 'audio',
21
+ 'transcription' => 'text'
22
+ }.freeze
23
+
24
+ CAPABILITY_BY_OUTPUT_MODALITY = {
25
+ 'speech' => 'speech_generation',
26
+ 'transcription' => 'transcription',
27
+ 'image' => 'image_generation'
28
+ }.freeze
29
+
30
+ CAPABILITY_PARAMETERS = {
31
+ 'function_calling' => %w[tools tool_choice],
32
+ 'tool_choice' => %w[tool_choice],
33
+ 'parallel_tool_calls' => %w[parallel_tool_calls],
34
+ 'structured_output' => %w[response_format structured_outputs],
35
+ 'batch' => %w[batch]
36
+ }.freeze
37
+
38
+ def list_models
39
+ CATALOG_URLS.flat_map do |url|
40
+ parse_list_models_response @connection.get(url), @provider.slug
41
+ end.uniq(&:id)
42
+ end
43
+
8
44
  module_function
9
45
 
10
46
  def models_url
11
47
  'models'
12
48
  end
13
49
 
14
- def parse_list_models_response(response, slug, _capabilities)
50
+ def parse_list_models_response(response, slug)
15
51
  Array(response.body['data']).map do |model_data| # rubocop:disable Metrics/BlockLength
52
+ output_modalities = Array(model_data.dig('architecture', 'output_modalities'))
16
53
  modalities = {
17
54
  input: Array(model_data.dig('architecture', 'input_modalities')),
18
- output: Array(model_data.dig('architecture', 'output_modalities'))
55
+ output: output_modalities.map { |modality| OUTPUT_MODALITY_MAP.fetch(modality, modality) }.uniq
19
56
  }
20
57
 
21
58
  pricing = { text_tokens: { standard: {} } }
@@ -24,6 +61,7 @@ module RubyLLM
24
61
  prompt: :input_per_million,
25
62
  completion: :output_per_million,
26
63
  input_cache_read: :cache_read_input_per_million,
64
+ input_cache_write: :cache_write_input_per_million,
27
65
  internal_reasoning: :reasoning_output_per_million
28
66
  }
29
67
 
@@ -32,9 +70,10 @@ module RubyLLM
32
70
  pricing[:text_tokens][:standard][target_key] = value * 1_000_000 if value.positive?
33
71
  end
34
72
 
35
- capabilities = supported_parameters_to_capabilities(model_data['supported_parameters'])
73
+ capabilities = supported_parameters_to_capabilities(model_data['supported_parameters']) |
74
+ output_modalities.filter_map { |modality| CAPABILITY_BY_OUTPUT_MODALITY[modality] }
36
75
 
37
- Model::Info.new(
76
+ Model.new(
38
77
  id: model_data['id'],
39
78
  name: model_data['name'],
40
79
  provider: slug,
@@ -42,6 +81,7 @@ module RubyLLM
42
81
  created_at: model_data['created'] ? Time.at(model_data['created']) : nil,
43
82
  context_window: model_data['context_length'],
44
83
  max_output_tokens: model_data.dig('top_provider', 'max_completion_tokens'),
84
+ knowledge_cutoff: model_data['knowledge_cutoff'],
45
85
  modalities: modalities,
46
86
  capabilities: capabilities,
47
87
  pricing: pricing,
@@ -50,7 +90,8 @@ module RubyLLM
50
90
  architecture: model_data['architecture'],
51
91
  top_provider: model_data['top_provider'],
52
92
  per_request_limits: model_data['per_request_limits'],
53
- supported_parameters: model_data['supported_parameters']
93
+ supported_parameters: model_data['supported_parameters'],
94
+ expiration_date: model_data['expiration_date']
54
95
  }
55
96
  )
56
97
  end
@@ -59,11 +100,9 @@ module RubyLLM
59
100
  def supported_parameters_to_capabilities(params)
60
101
  return [] unless params
61
102
 
62
- capabilities = []
63
- capabilities << 'streaming'
64
- capabilities << 'function_calling' if params.include?('tools') || params.include?('tool_choice')
65
- capabilities << 'structured_output' if params.include?('response_format')
66
- capabilities << 'batch' if params.include?('batch')
103
+ capabilities = ['streaming'] + CAPABILITY_PARAMETERS.filter_map do |capability, parameters|
104
+ capability if parameters.any? { |parameter| params.include?(parameter) }
105
+ end
67
106
  capabilities << 'predicted_outputs' if params.include?('logit_bias') && params.include?('top_k')
68
107
  capabilities
69
108
  end
@@ -0,0 +1,32 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Providers
5
+ class OpenRouter
6
+ # Speech generation methods for the OpenRouter API integration.
7
+ # Voices are model-specific across OpenRouter's TTS catalog, so no
8
+ # default voice is assumed.
9
+ module Speech
10
+ module_function
11
+
12
+ def render_speech_payload(input, model:, voice:, format:, provider_options: {})
13
+ {
14
+ model: model,
15
+ input: input,
16
+ voice: voice,
17
+ response_format: format || 'mp3'
18
+ }.compact.merge(provider_options)
19
+ end
20
+
21
+ def parse_speech_response(response, model:, voice:, format:)
22
+ RubyLLM::Speech.new(
23
+ data: response.body,
24
+ model: model,
25
+ voice: voice,
26
+ format: (format || 'mp3').to_s
27
+ )
28
+ end
29
+ end
30
+ end
31
+ end
32
+ end