ruby_llm 1.16.0 → 2.0.0.rc4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (480) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +108 -44
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/legacy_content_sql.rb +34 -0
  91. data/lib/generators/ruby_llm/upgrade/online_copy_migration/data.rb +341 -0
  92. data/lib/generators/ruby_llm/upgrade/online_copy_migration/journal.rb +171 -0
  93. data/lib/generators/ruby_llm/upgrade/online_copy_migration/verification.rb +197 -0
  94. data/lib/generators/ruby_llm/upgrade/online_copy_migration.rb +174 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +468 -0
  96. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
  97. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +247 -0
  98. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +662 -0
  99. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +251 -0
  100. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  101. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +188 -0
  102. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +359 -0
  103. data/lib/ruby_llm/accounting/usage.rb +254 -0
  104. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  105. data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
  106. data/lib/ruby_llm/active_record/batch.rb +97 -0
  107. data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
  108. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  109. data/lib/ruby_llm/active_record/model.rb +135 -0
  110. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  111. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  112. data/lib/ruby_llm/active_record/usage.rb +61 -0
  113. data/lib/ruby_llm/agent.rb +1066 -151
  114. data/lib/ruby_llm/aliases.json +291 -101
  115. data/lib/ruby_llm/attachment.rb +192 -48
  116. data/lib/ruby_llm/batch.rb +432 -0
  117. data/lib/ruby_llm/cached_content.rb +112 -0
  118. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  119. data/lib/ruby_llm/chat.rb +1131 -198
  120. data/lib/ruby_llm/chunk.rb +10 -0
  121. data/lib/ruby_llm/citation.rb +105 -0
  122. data/lib/ruby_llm/configuration.rb +261 -24
  123. data/lib/ruby_llm/context.rb +128 -6
  124. data/lib/ruby_llm/cost.rb +217 -80
  125. data/lib/ruby_llm/downloaded_file.rb +33 -0
  126. data/lib/ruby_llm/embedding.rb +121 -17
  127. data/lib/ruby_llm/embedding_request.rb +53 -0
  128. data/lib/ruby_llm/error.rb +159 -23
  129. data/lib/ruby_llm/fallback.rb +133 -0
  130. data/lib/ruby_llm/files/mime_type.rb +97 -0
  131. data/lib/ruby_llm/image.rb +153 -32
  132. data/lib/ruby_llm/message.rb +242 -54
  133. data/lib/ruby_llm/model/modalities.rb +17 -4
  134. data/lib/ruby_llm/model/pricing.rb +24 -5
  135. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  136. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  137. data/lib/ruby_llm/model.rb +244 -2
  138. data/lib/ruby_llm/models/aliases.rb +41 -0
  139. data/lib/ruby_llm/models/registry.rb +165 -0
  140. data/lib/ruby_llm/models/schema.rb +99 -0
  141. data/lib/ruby_llm/models.json +76038 -34173
  142. data/lib/ruby_llm/models.rb +477 -215
  143. data/lib/ruby_llm/moderation.rb +139 -26
  144. data/lib/ruby_llm/ocr.rb +112 -0
  145. data/lib/ruby_llm/prompt.rb +79 -0
  146. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  147. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  148. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  149. data/lib/ruby_llm/protocol.rb +662 -0
  150. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  151. data/lib/ruby_llm/protocols/anthropic/chat.rb +543 -0
  152. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  153. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  154. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  155. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  156. data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
  157. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  158. data/lib/ruby_llm/protocols/anthropic.rb +101 -0
  159. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  160. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +110 -0
  161. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  162. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  163. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  164. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  165. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  166. data/lib/ruby_llm/protocols/chat_completions/chat.rb +484 -0
  167. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  168. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  169. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  170. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  171. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  172. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  173. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +63 -0
  174. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  175. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  176. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  177. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  178. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  179. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  180. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  181. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  182. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  183. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  184. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  185. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  186. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  187. data/lib/ruby_llm/protocols/cohere/rerank.rb +59 -0
  188. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  189. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  190. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  191. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  192. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  193. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  194. data/lib/ruby_llm/protocols/converse/chat.rb +694 -0
  195. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  196. data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
  197. data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
  198. data/lib/ruby_llm/protocols/converse.rb +54 -0
  199. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  200. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  201. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  202. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  203. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  204. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  208. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  209. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  210. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  211. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  212. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  213. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  214. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  215. data/lib/ruby_llm/protocols/files.rb +119 -0
  216. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  217. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  218. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  219. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +91 -0
  220. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  221. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  222. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  223. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  224. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  225. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  226. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  227. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  228. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  229. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  230. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  231. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  232. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  233. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  234. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  235. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  236. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  237. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  238. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  239. data/lib/ruby_llm/protocols/interactions/tools.rb +53 -0
  240. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  241. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  242. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  243. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  244. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  245. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  246. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  247. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  248. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  249. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  250. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  251. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  252. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  253. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  254. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  255. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  256. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  257. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  258. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  259. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  260. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  261. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  262. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  263. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  264. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  265. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  266. data/lib/ruby_llm/protocols/responses/chat.rb +466 -0
  267. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  268. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  269. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  270. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  271. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  272. data/lib/ruby_llm/protocols/responses.rb +35 -0
  273. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  274. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  275. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  276. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  277. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  278. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  279. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  280. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  281. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  282. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  283. data/lib/ruby_llm/provider.rb +560 -128
  284. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  285. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  286. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  287. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  288. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  289. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  290. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  291. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  292. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  293. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  294. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  295. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  296. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  297. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  298. data/lib/ruby_llm/providers/azure.rb +77 -78
  299. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  300. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  301. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  302. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  303. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  304. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  305. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  306. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  307. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  308. data/lib/ruby_llm/providers/cohere.rb +31 -0
  309. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  310. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  311. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  312. data/lib/ruby_llm/providers/deepseek/responses.rb +67 -0
  313. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  314. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  315. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  316. data/lib/ruby_llm/providers/gemini.rb +15 -8
  317. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  318. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  319. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  320. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  321. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  322. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  323. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  324. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  325. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  326. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  327. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  328. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  329. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  330. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  331. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  332. data/lib/ruby_llm/providers/mistral/ocr.rb +51 -0
  333. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  334. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  335. data/lib/ruby_llm/providers/mistral.rb +16 -4
  336. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  337. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  338. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  339. data/lib/ruby_llm/providers/ollama.rb +9 -8
  340. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  341. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  342. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  343. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  344. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  345. data/lib/ruby_llm/providers/openai.rb +92 -11
  346. data/lib/ruby_llm/providers/openrouter/chat.rb +126 -108
  347. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  348. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  349. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  350. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  351. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  352. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  353. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  354. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  355. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  356. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  357. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  358. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  359. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  360. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  361. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  362. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  363. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  364. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  365. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  366. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  367. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  368. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  369. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  370. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  371. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  372. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  373. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  374. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  375. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  376. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  377. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  378. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  379. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  380. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  381. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  382. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  383. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  384. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  385. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  386. data/lib/ruby_llm/providers/xai.rb +15 -5
  387. data/lib/ruby_llm/railtie.rb +7 -16
  388. data/lib/ruby_llm/rerank.rb +105 -0
  389. data/lib/ruby_llm/research_job.rb +241 -0
  390. data/lib/ruby_llm/search_results.rb +68 -0
  391. data/lib/ruby_llm/server_tool_call.rb +73 -0
  392. data/lib/ruby_llm/speech.rb +159 -0
  393. data/lib/ruby_llm/speech_chunk.rb +33 -0
  394. data/lib/ruby_llm/support/deprecator.rb +22 -0
  395. data/lib/ruby_llm/support/inspectable.rb +49 -0
  396. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  397. data/lib/ruby_llm/support/utils.rb +147 -0
  398. data/lib/ruby_llm/thinking.rb +127 -20
  399. data/lib/ruby_llm/tokenization.rb +59 -0
  400. data/lib/ruby_llm/tokens.rb +103 -33
  401. data/lib/ruby_llm/tool.rb +266 -91
  402. data/lib/ruby_llm/tool_call.rb +36 -3
  403. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  404. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  405. data/lib/ruby_llm/transcription.rb +138 -13
  406. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  407. data/lib/ruby_llm/transport/connection.rb +193 -0
  408. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  409. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  410. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  411. data/lib/ruby_llm/uploaded_file.rb +144 -0
  412. data/lib/ruby_llm/version.rb +2 -1
  413. data/lib/ruby_llm/video.rb +136 -0
  414. data/lib/ruby_llm/video_job.rb +150 -0
  415. data/lib/ruby_llm/workflow.rb +91 -0
  416. data/lib/ruby_llm.rb +380 -6
  417. data/lib/tasks/ruby_llm.rake +21 -16
  418. data/skills/rubyllm/SKILL.md +81 -0
  419. data/skills/rubyllm/agents/openai.yaml +4 -0
  420. metadata +344 -97
  421. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  422. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  423. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  424. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  425. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  426. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  427. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  428. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  429. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  430. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  431. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  432. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  433. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  434. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  435. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  436. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  437. data/lib/ruby_llm/aliases.rb +0 -41
  438. data/lib/ruby_llm/connection.rb +0 -159
  439. data/lib/ruby_llm/content.rb +0 -91
  440. data/lib/ruby_llm/deprecator.rb +0 -24
  441. data/lib/ruby_llm/error_middleware.rb +0 -81
  442. data/lib/ruby_llm/instrumentation.rb +0 -36
  443. data/lib/ruby_llm/mime_type.rb +0 -96
  444. data/lib/ruby_llm/model/info.rb +0 -164
  445. data/lib/ruby_llm/model_registry.rb +0 -39
  446. data/lib/ruby_llm/models_schema.json +0 -171
  447. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  448. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  449. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  450. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  451. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  452. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  453. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  454. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  455. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  456. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  457. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  458. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  459. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  460. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  461. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  462. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  463. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  464. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  465. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  466. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  467. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  468. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  469. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  470. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  471. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  472. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  473. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  474. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  475. data/lib/ruby_llm/streaming.rb +0 -179
  476. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  477. data/lib/ruby_llm/utils.rb +0 -130
  478. data/lib/tasks/models.rake +0 -593
  479. data/lib/tasks/release.rake +0 -94
  480. data/lib/tasks/vcr.rake +0 -124
@@ -0,0 +1,484 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class ChatCompletions
6
+ # Chat methods of the OpenAI API integration
7
+ module Chat
8
+ FINISH_REASONS = {
9
+ 'stop' => :stop, 'length' => :max_tokens, 'tool_calls' => :tool_calls,
10
+ 'function_call' => :tool_calls, 'content_filter' => :content_filter
11
+ }.freeze
12
+
13
+ OPENAI_INLINE_FILE_LIMIT = 50 * 1024 * 1024
14
+ OPENAI_FILE_UPLOAD_LIMIT = 512 * 1024 * 1024
15
+ PROMPT_CACHE_OPTIONS = %i[key ttl mode retention].freeze
16
+
17
+ def completion_url
18
+ 'chat/completions'
19
+ end
20
+
21
+ module_function
22
+
23
+ def finish_reasons = FINISH_REASONS
24
+
25
+ def normalize_finish_reason(reason)
26
+ return nil if reason.nil?
27
+
28
+ finish_reasons.fetch(reason.to_s) { reason.to_s.to_sym }
29
+ end
30
+
31
+ # OpenAI strict mode rejects an object whose properties are not all
32
+ # required, so a schema with optional properties goes out non-strict
33
+ # unless the caller asked for strict explicitly.
34
+ def schema_strict(schema)
35
+ schema.key?(:strict) ? schema[:strict] : strict_schema?(schema[:schema])
36
+ end
37
+
38
+ def strict_schema?(node)
39
+ case node
40
+ when Hash
41
+ return false if optional_properties?(node)
42
+
43
+ node.values.all? { |value| strict_schema?(value) }
44
+ when Array then node.all? { |value| strict_schema?(value) }
45
+ else true
46
+ end
47
+ end
48
+
49
+ def optional_properties?(node)
50
+ properties = node[:properties]
51
+ return false unless properties.is_a?(Hash)
52
+
53
+ (properties.keys.map(&:to_s) - Array(node[:required]).map(&:to_s)).any?
54
+ end
55
+
56
+ # rubocop:disable-next Metrics/PerceivedComplexity
57
+ def render_payload(messages, tools:, temperature:, model:, stream: false, max_output_tokens: nil, schema: nil,
58
+ thinking: nil, citations: false, caching: nil, tool_prefs: nil)
59
+ warn_unsupported_citations(model) if citations && !model.supports?(:citations)
60
+ tool_prefs ||= {}
61
+ payload = {
62
+ model: model.id,
63
+ messages: format_messages(messages, caching: caching),
64
+ stream: stream
65
+ }
66
+
67
+ payload[:temperature] = temperature unless temperature.nil?
68
+ payload[max_output_tokens_field(model)] = max_output_tokens unless max_output_tokens.nil?
69
+ if tools.any?
70
+ payload[:tools] = tools.map { |_, tool| tool_for(tool) }
71
+ payload[:tool_choice] = build_tool_choice(tool_prefs[:choice]) unless tool_prefs[:choice].nil?
72
+ payload[:parallel_tool_calls] = tool_prefs[:calls] == :many unless tool_prefs[:calls].nil?
73
+ end
74
+
75
+ if schema
76
+ payload[:response_format] = {
77
+ type: 'json_schema',
78
+ json_schema: {
79
+ name: schema[:name],
80
+ schema: schema[:schema],
81
+ strict: schema_strict(schema)
82
+ }
83
+ }
84
+ end
85
+
86
+ effort = resolve_effort(thinking)
87
+ payload[:reasoning_effort] = effort if effort
88
+
89
+ payload[:stream_options] = { include_usage: true } if stream
90
+ apply_prompt_cache_params(payload, caching)
91
+ payload
92
+ end
93
+
94
+ # OpenAI and Azure reject max_tokens on their reasoning models and
95
+ # accept max_completion_tokens on every model; the rest of the wire
96
+ # format only knows max_tokens.
97
+ def max_output_tokens_field(_model)
98
+ %w[openai azure].include?(@provider.slug) ? :max_completion_tokens : :max_tokens
99
+ end
100
+
101
+ def warn_unsupported_citations(model)
102
+ RubyLLM.logger.warn(
103
+ "#{model.id} does not support citations according to the model registry. " \
104
+ 'with_citations may have no effect.'
105
+ )
106
+ end
107
+
108
+ def parse_completion_body(data, raw:)
109
+ raise Error.new(data.dig('error', 'message'), response: raw) if data.dig('error', 'message')
110
+
111
+ message_data = data.dig('choices', 0, 'message')
112
+ raise no_completion_message_error(data, raw) unless message_data
113
+
114
+ usage = data['usage'] || {}
115
+ thinking_tokens = thinking_tokens(usage)
116
+ content, thinking_from_blocks = extract_content_and_thinking(message_data['content'])
117
+ thinking_text = thinking_from_blocks || extract_thinking_text(message_data)
118
+ thinking_signature = extract_thinking_signature(message_data)
119
+
120
+ finish_reason = normalize_finish_reason(data.dig('choices', 0, 'finish_reason'))
121
+
122
+ Message.new(
123
+ role: :assistant,
124
+ content: content,
125
+ citations: extract_citations(message_data, data, content),
126
+ thinking: Thinking.build(text: thinking_text, signature: thinking_signature),
127
+ raw_reasoning: extract_raw_reasoning(message_data),
128
+ tool_calls: parse_tool_calls(message_data['tool_calls'], response: raw, finish_reason: finish_reason),
129
+ input_tokens: input_tokens(usage),
130
+ output_tokens: output_tokens(usage),
131
+ cache_read_tokens: cache_read_tokens(usage),
132
+ cache_write_tokens: cache_write_tokens(usage),
133
+ thinking_tokens: thinking_tokens,
134
+ server_tool_use: server_tool_use(usage),
135
+ reported_cost: reported_cost(usage),
136
+ finish_reason: finish_reason,
137
+ model: data['model'],
138
+ raw: raw
139
+ )
140
+ end
141
+
142
+ def reported_cost(_usage)
143
+ nil
144
+ end
145
+
146
+ def server_tool_use(usage)
147
+ usage['server_tool_use'] || usage['server_tool_use_details']
148
+ end
149
+
150
+ def no_completion_message_error(data, raw)
151
+ finish_reason = data.dig('choices', 0, 'finish_reason')
152
+ message = 'Provider returned no completion message'
153
+ message = "#{message} (finish_reason: #{finish_reason})" if finish_reason
154
+ Error.new(message, response: raw)
155
+ end
156
+
157
+ def input_tokens(usage)
158
+ return usage['prompt_cache_miss_tokens'] if usage['prompt_cache_miss_tokens']
159
+
160
+ prompt_tokens = usage['prompt_tokens']
161
+ return unless prompt_tokens
162
+
163
+ [prompt_tokens.to_i - cache_read_tokens(usage).to_i - cache_write_tokens(usage).to_i, 0].max
164
+ end
165
+
166
+ def output_tokens(usage)
167
+ completion_tokens = usage['completion_tokens']
168
+ return unless completion_tokens
169
+
170
+ completion_tokens = completion_tokens.to_i
171
+ generated_tokens = generated_tokens_from_total(usage)
172
+ return completion_tokens unless generated_tokens && generated_tokens > completion_tokens
173
+
174
+ generated_tokens
175
+ end
176
+
177
+ def generated_tokens_from_total(usage)
178
+ prompt_tokens = usage['prompt_tokens']
179
+ total_tokens = usage['total_tokens']
180
+ return unless prompt_tokens && total_tokens
181
+
182
+ [total_tokens.to_i - prompt_tokens.to_i, 0].max
183
+ end
184
+
185
+ def cache_read_tokens(usage)
186
+ usage.dig('prompt_tokens_details', 'cached_tokens') || usage['prompt_cache_hit_tokens']
187
+ end
188
+
189
+ def cache_write_tokens(usage)
190
+ usage.dig('prompt_tokens_details', 'cache_write_tokens') ||
191
+ usage.dig('input_tokens_details', 'cache_write_tokens') ||
192
+ 0
193
+ end
194
+
195
+ def thinking_tokens(usage)
196
+ usage.dig('completion_tokens_details', 'reasoning_tokens') || usage['reasoning_tokens']
197
+ end
198
+
199
+ def extract_citations(message_data, data, content)
200
+ annotations = parse_annotations(message_data['annotations'], content)
201
+ return annotations if annotations.any?
202
+
203
+ parse_root_citations(data)
204
+ end
205
+
206
+ def parse_annotations(annotations, content)
207
+ Array(annotations).filter_map do |annotation|
208
+ details = annotation['url_citation']
209
+ next unless details.is_a?(Hash)
210
+
211
+ start_index = details['start_index']
212
+ end_index = details['end_index']
213
+
214
+ Citation.new(
215
+ url: details['url'],
216
+ title: details['title'],
217
+ text: annotated_text(content, start_index, end_index),
218
+ start_index: start_index,
219
+ end_index: end_index
220
+ )
221
+ end
222
+ end
223
+
224
+ def annotated_text(content, start_index, end_index)
225
+ return nil unless content.is_a?(String) && start_index && end_index
226
+
227
+ content[start_index...end_index]
228
+ end
229
+
230
+ # Perplexity and xAI return search citations at the root of the response.
231
+ def parse_root_citations(data)
232
+ search_results = data['search_results']
233
+ return parse_search_results(search_results) if search_results.is_a?(Array) && search_results.any?
234
+
235
+ Array(data['citations']).each_with_index.filter_map do |url, index|
236
+ Citation.new(url: url, source_index: index) if url.is_a?(String)
237
+ end
238
+ end
239
+
240
+ def parse_search_results(results)
241
+ results.each_with_index.filter_map do |result, index|
242
+ next unless result.is_a?(Hash)
243
+
244
+ Citation.new(
245
+ url: result['url'],
246
+ title: result['title'],
247
+ cited_text: result['snippet'],
248
+ source_index: index
249
+ )
250
+ end
251
+ end
252
+
253
+ def apply_prompt_cache_params(payload, caching)
254
+ return unless openai_prompt_caching?
255
+
256
+ payload.merge!(prompt_cache_params(caching)) if caching
257
+ end
258
+
259
+ def openai_prompt_caching?
260
+ true
261
+ end
262
+
263
+ def prompt_cache_params(caching)
264
+ options = prompt_cache_options(caching)
265
+ cache_options = build_prompt_cache_options(options)
266
+
267
+ {}.tap do |params|
268
+ params[:prompt_cache_key] = options[:key] if options[:key]
269
+ params[:prompt_cache_options] = cache_options unless cache_options.empty?
270
+ end
271
+ end
272
+
273
+ def build_prompt_cache_options(options)
274
+ ttl = options[:ttl] || retention_ttl(options[:retention])
275
+
276
+ {}.tap do |cache_options|
277
+ cache_options[:mode] = options[:mode] if options[:mode]
278
+ cache_options[:ttl] = ttl if ttl
279
+ end
280
+ end
281
+
282
+ def retention_ttl(retention)
283
+ return unless retention
284
+
285
+ RubyLLM.logger.warn(
286
+ 'with_caching retention: is deprecated; OpenAI replaced prompt_cache_retention ' \
287
+ 'with prompt_cache_options. Use ttl: instead.'
288
+ )
289
+ retention
290
+ end
291
+
292
+ def prompt_cache_options(caching)
293
+ options = caching.to_h.transform_keys(&:to_sym)
294
+ unsupported = options.keys - PROMPT_CACHE_OPTIONS
295
+ return options if unsupported.empty?
296
+
297
+ raise ArgumentError,
298
+ 'Chat Completions prompt caching accepts :key, :ttl, and :mode, ' \
299
+ "got #{format_cache_option_keys(unsupported)}"
300
+ end
301
+
302
+ def format_cache_option_keys(keys)
303
+ keys.map { |key| ":#{key}" }.join(', ')
304
+ end
305
+
306
+ def format_messages(messages, caching: nil)
307
+ messages_for_provider(messages)
308
+ .chunk_while { |previous, current| previous.tool_result? && current.tool_result? }
309
+ .flat_map { |group| format_message_group(group, caching: caching) }
310
+ end
311
+
312
+ # An assistant turn's tool results must stay consecutive, so the
313
+ # attachment carriers of a parallel round follow the whole run.
314
+ def format_message_group(group, caching: nil)
315
+ formatted = group.map { |msg| format_message(msg, caching: caching) }
316
+ carriers = group.select { |msg| msg.tool_result? && msg.attachments.any? }
317
+
318
+ formatted + carriers.map { |msg| tool_attachment_message(msg) }
319
+ end
320
+
321
+ def format_message(msg, caching: nil)
322
+ {
323
+ role: format_role(msg.role),
324
+ content: format_message_content(msg, caching: caching),
325
+ tool_calls: format_tool_calls(msg.tool_calls),
326
+ tool_call_id: msg.tool_call_id
327
+ }.compact.merge(format_thinking(msg))
328
+ end
329
+
330
+ # Chat Completions tool messages are text-only on the wire, so tool
331
+ # attachments ride a user message spliced in after the results.
332
+ def tool_attachment_message(msg)
333
+ parts = [Media.format_text("Attachments from tool call #{msg.tool_call_id}:")]
334
+ parts.concat(format_content(nil, msg.attachments))
335
+ { role: 'user', content: parts }
336
+ end
337
+
338
+ def messages_for_provider(messages)
339
+ system_messages, other_messages = messages.partition { |msg| msg.role == :system }
340
+ system_messages + other_messages
341
+ end
342
+
343
+ def format_message_content(msg, caching: nil, **)
344
+ content = format_content(msg.content, msg.tool_result? ? [] : msg.attachments)
345
+ return '' if content.nil? && thinking_only_assistant_message?(msg)
346
+ return inject_cache_breakpoint(content) if caching != false && msg.cache_until_here? && openai_prompt_caching?
347
+
348
+ content
349
+ end
350
+
351
+ def inject_cache_breakpoint(content)
352
+ parts = cache_breakpoint_parts(content)
353
+ return content unless parts&.last.is_a?(Hash)
354
+
355
+ parts[-1] = parts.last.merge(prompt_cache_breakpoint: { mode: 'explicit' })
356
+ parts
357
+ end
358
+
359
+ def cache_breakpoint_parts(content)
360
+ case content
361
+ when Array then content.dup
362
+ when String then [Media.format_text(content)] unless content.empty?
363
+ end
364
+ end
365
+
366
+ def thinking_only_assistant_message?(msg)
367
+ msg.role == :assistant && msg.thinking && !msg.tool_call?
368
+ end
369
+
370
+ def format_content(content, attachments = [])
371
+ Media.format_content(content, attachments)
372
+ end
373
+
374
+ def format_role(role)
375
+ case role
376
+ when :system
377
+ @config.openai_use_system_role ? 'system' : 'developer'
378
+ else
379
+ role.to_s
380
+ end
381
+ end
382
+
383
+ def resolve_effort(thinking)
384
+ return nil unless thinking
385
+
386
+ effort = thinking.respond_to?(:effort) ? thinking.effort : thinking
387
+ effort&.to_s
388
+ end
389
+
390
+ # safety_identifier is an OpenAI parameter; the other services on
391
+ # this wire format reject or ignore it, so only OpenAI's own
392
+ # endpoints receive it.
393
+ def apply_end_user(payload, identifier)
394
+ return super unless %w[openai azure].include?(@provider.slug)
395
+
396
+ payload.merge(safety_identifier: identifier)
397
+ end
398
+
399
+ def supports_provider_file_references?
400
+ @provider.slug == 'openai'
401
+ end
402
+
403
+ def default_large_file_upload_threshold
404
+ OPENAI_INLINE_FILE_LIMIT
405
+ end
406
+
407
+ def provider_file_upload_limit
408
+ OPENAI_FILE_UPLOAD_LIMIT
409
+ end
410
+
411
+ def provider_file_attachable?(attachment)
412
+ attachment.pdf?
413
+ end
414
+
415
+ def provider_file_upload_options(_attachment)
416
+ { purpose: 'user_data' }
417
+ end
418
+
419
+ def format_thinking(msg)
420
+ return {} unless msg.role == :assistant
421
+
422
+ thinking = msg.thinking
423
+ return {} unless thinking
424
+
425
+ payload = {}
426
+ if thinking.text
427
+ payload[:reasoning] = thinking.text
428
+ payload[:reasoning_content] = thinking.text
429
+ end
430
+ payload[:reasoning_signature] = thinking.signature if thinking.signature
431
+ payload
432
+ end
433
+
434
+ def extract_thinking_text(message_data)
435
+ candidate = message_data['reasoning_content'] || message_data['reasoning'] || message_data['thinking']
436
+ candidate.is_a?(String) ? candidate : nil
437
+ end
438
+
439
+ def extract_thinking_signature(message_data)
440
+ candidate = message_data['reasoning_signature'] || message_data['signature']
441
+ candidate.is_a?(String) ? candidate : nil
442
+ end
443
+
444
+ def extract_raw_reasoning(_message_data)
445
+ nil
446
+ end
447
+
448
+ def extract_content_and_thinking(content)
449
+ return [content, nil] unless content.is_a?(Array)
450
+
451
+ text = extract_text_from_blocks(content)
452
+ thinking = extract_thinking_from_blocks(content)
453
+
454
+ [text.empty? ? nil : text, thinking.empty? ? nil : thinking]
455
+ end
456
+
457
+ def extract_text_from_blocks(blocks)
458
+ blocks.filter_map do |block|
459
+ block['text'] if block['type'] == 'text' && block['text'].is_a?(String)
460
+ end.join
461
+ end
462
+
463
+ def extract_thinking_from_blocks(blocks)
464
+ blocks.filter_map do |block|
465
+ next unless block['type'] == 'thinking'
466
+
467
+ extract_thinking_text_from_block(block)
468
+ end.join
469
+ end
470
+
471
+ def extract_thinking_text_from_block(block)
472
+ thinking_block = block['thinking']
473
+ return thinking_block if thinking_block.is_a?(String)
474
+
475
+ if thinking_block.is_a?(Array)
476
+ return thinking_block.filter_map { |item| item['text'] if item['type'] == 'text' }.join
477
+ end
478
+
479
+ block['text'] if block['text'].is_a?(String)
480
+ end
481
+ end
482
+ end
483
+ end
484
+ end
@@ -0,0 +1,35 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class ChatCompletions
6
+ # OpenAI-compatible file-backed Batch API for embeddings.
7
+ module EmbeddingBatches
8
+ include Protocols::OpenAI::Batches
9
+
10
+ Response = Struct.new(:body)
11
+ private_constant :Response
12
+
13
+ private
14
+
15
+ def batch_endpoint
16
+ '/v1/embeddings'
17
+ end
18
+
19
+ def validate_batch_requests!(requests)
20
+ return if requests.all? { |request| embedding_payload?(request.fetch(:payload)) }
21
+
22
+ raise Error, "#{@provider.slug} embedding batch requests require embedding payloads"
23
+ end
24
+
25
+ def embedding_payload?(payload)
26
+ payload.key?(:input) || payload.key?('input')
27
+ end
28
+
29
+ def parse_batch_completion_response(body)
30
+ parse_embedding_response(Response.new(body), model: body['model'], text: nil)
31
+ end
32
+ end
33
+ end
34
+ end
35
+ end
@@ -0,0 +1,60 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class ChatCompletions
6
+ # Embeddings methods of the OpenAI API integration
7
+ module Embeddings
8
+ module_function
9
+
10
+ def embedding_url(...)
11
+ 'embeddings'
12
+ end
13
+
14
+ # rubocop:disable-next Lint/UnusedMethodArgument
15
+ def render_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil, provider_options: {})
16
+ {
17
+ model: model,
18
+ input: text,
19
+ dimensions: dimensions
20
+ }.compact.merge(provider_options)
21
+ end
22
+
23
+ def parse_embedding_response(response, model:, text:)
24
+ data = response.body
25
+ input_tokens = data.dig('usage', 'prompt_tokens')
26
+ rows = data['data']
27
+ single = rows.length == 1 && !text.is_a?(Array)
28
+
29
+ vectors = rows.map { |row| row['embedding'] }
30
+ vectors = vectors.first if single
31
+ sparse_vectors = parse_sparse_vectors(rows, single: single)
32
+
33
+ Embedding.new(vectors:, sparse_vectors:, model:, input_tokens:,
34
+ reported_cost: reported_cost(data['usage'] || {}))
35
+ end
36
+
37
+ # Sparse-capable models return a token-to-weight map beside the dense
38
+ # vector, under lexical_weights on BGE-M3 and sparse_embedding
39
+ # elsewhere. It is an extension: dense-only servers send neither, and
40
+ # then there is nothing to report.
41
+ def parse_sparse_vectors(rows, single:)
42
+ sparse = rows.map { |row| normalize_sparse_vector(row['sparse_embedding'] || row['lexical_weights']) }
43
+ return nil if sparse.all?(&:nil?)
44
+
45
+ single ? sparse.first : sparse
46
+ end
47
+
48
+ def normalize_sparse_vector(weights)
49
+ return nil unless weights.is_a?(Hash)
50
+
51
+ weights.to_h { |token, weight| [Integer(token), Float(weight)] }
52
+ end
53
+
54
+ def reported_cost(_usage)
55
+ nil
56
+ end
57
+ end
58
+ end
59
+ end
60
+ end
@@ -0,0 +1,127 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'faraday'
4
+ require 'stringio'
5
+
6
+ module RubyLLM
7
+ module Protocols
8
+ class ChatCompletions
9
+ # Image generation methods for the OpenAI API integration
10
+ module Images
11
+ module_function
12
+
13
+ def images_url(with: nil, mask: nil)
14
+ editing?(with, mask) ? 'images/edits' : 'images/generations'
15
+ end
16
+
17
+ def render_image_payload(prompt, model:, size:, count: nil, with: nil, mask: nil, provider_options: {})
18
+ return render_edit_payload(prompt, model:, size:, count:, with:, mask:, provider_options:) if editing?(with,
19
+ mask)
20
+
21
+ {
22
+ model: model,
23
+ prompt: prompt,
24
+ n: count || 1,
25
+ size: size
26
+ }.merge(provider_options)
27
+ end
28
+
29
+ def parse_image_response(response, model:)
30
+ parse_image_responses(response, model:).first
31
+ end
32
+
33
+ def parse_image_responses(response, model:)
34
+ data = response.body
35
+ entries = Array(data['data'])
36
+
37
+ raise Error, 'Unexpected response format from OpenAI image API' if entries.empty?
38
+
39
+ entries.map.with_index do |image_data, index|
40
+ Image.new(
41
+ url: image_data['url'],
42
+ mime_type: 'image/png', # DALL-E typically returns PNGs
43
+ revised_prompt: image_data['revised_prompt'],
44
+ model: model,
45
+ data: image_data['b64_json'],
46
+ usage: index.zero? ? (data['usage'] || {}) : {}
47
+ )
48
+ end
49
+ end
50
+
51
+ def validate_paint_inputs!(with:, mask:)
52
+ return unless editing?(with, mask)
53
+
54
+ raise ArgumentError, 'with: is required when mask: is provided' if mask && !attachments?(with)
55
+ end
56
+
57
+ def render_edit_payload(prompt, model:, size:, with:, mask:, provider_options:, count: nil)
58
+ payload = { model: model, prompt: prompt, n: count || 1 }
59
+ if json_image_references?(model)
60
+ payload[:images] = build_image_references(with)
61
+ payload[:mask] = build_image_reference(mask) if mask
62
+ payload[:size] = size if size
63
+ else
64
+ payload[:image] = single_upload_part(build_upload_parts(with))
65
+ payload[:mask] = build_upload_part(mask) if mask
66
+ end
67
+ payload.merge(provider_options)
68
+ end
69
+
70
+ def json_image_references?(model)
71
+ model.match?(/\A(gpt-image|chatgpt-image)/)
72
+ end
73
+
74
+ def build_image_references(sources)
75
+ Array(sources).filter_map do |source|
76
+ next if blank_attachment?(source)
77
+
78
+ build_image_reference(source)
79
+ end
80
+ end
81
+
82
+ def build_image_reference(source)
83
+ attachment = Attachment.new(source, config: @config)
84
+ return { file_id: attachment.provider_file_id } if attachment.provider_file?
85
+ return { image_url: attachment.source.to_s } if attachment.url?
86
+
87
+ raise UnsupportedAttachmentError, attachment.mime_type unless attachment.image?
88
+
89
+ { image_url: attachment.for_llm }
90
+ end
91
+
92
+ def build_upload_parts(sources)
93
+ Array(sources).filter_map do |source|
94
+ next if blank_attachment?(source)
95
+
96
+ build_upload_part(source)
97
+ end
98
+ end
99
+
100
+ # Retries only rewind upload parts at the top level of the payload,
101
+ # so a lone image rides there instead of inside an array.
102
+ def single_upload_part(parts)
103
+ parts.one? ? parts.first : parts
104
+ end
105
+
106
+ def build_upload_part(source)
107
+ attachment = Attachment.new(source, config: @config)
108
+ raise UnsupportedAttachmentError, attachment.mime_type unless attachment.image?
109
+
110
+ Faraday::UploadIO.new(StringIO.new(attachment.content), attachment.mime_type, attachment.filename)
111
+ end
112
+
113
+ def editing?(with, mask)
114
+ attachments?(with) || !mask.nil?
115
+ end
116
+
117
+ def attachments?(value)
118
+ Array(value).any? { |item| !blank_attachment?(item) }
119
+ end
120
+
121
+ def blank_attachment?(value)
122
+ value.nil? || (value.is_a?(String) && value.strip.empty?)
123
+ end
124
+ end
125
+ end
126
+ end
127
+ end