ruby_llm 1.16.0 → 2.0.0.rc2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (475) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +85 -32
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
  91. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
  92. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
  93. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
  94. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  96. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
  97. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
  98. data/lib/ruby_llm/accounting/usage.rb +245 -0
  99. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  100. data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
  101. data/lib/ruby_llm/active_record/batch.rb +97 -0
  102. data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
  103. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  104. data/lib/ruby_llm/active_record/model.rb +135 -0
  105. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  106. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  107. data/lib/ruby_llm/active_record/usage.rb +61 -0
  108. data/lib/ruby_llm/agent.rb +1065 -151
  109. data/lib/ruby_llm/aliases.json +268 -100
  110. data/lib/ruby_llm/attachment.rb +187 -48
  111. data/lib/ruby_llm/batch.rb +432 -0
  112. data/lib/ruby_llm/cached_content.rb +112 -0
  113. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  114. data/lib/ruby_llm/chat.rb +1127 -198
  115. data/lib/ruby_llm/chunk.rb +10 -0
  116. data/lib/ruby_llm/citation.rb +105 -0
  117. data/lib/ruby_llm/configuration.rb +261 -24
  118. data/lib/ruby_llm/context.rb +128 -6
  119. data/lib/ruby_llm/cost.rb +217 -80
  120. data/lib/ruby_llm/downloaded_file.rb +33 -0
  121. data/lib/ruby_llm/embedding.rb +121 -17
  122. data/lib/ruby_llm/embedding_request.rb +53 -0
  123. data/lib/ruby_llm/error.rb +159 -23
  124. data/lib/ruby_llm/fallback.rb +133 -0
  125. data/lib/ruby_llm/files/mime_type.rb +97 -0
  126. data/lib/ruby_llm/image.rb +153 -32
  127. data/lib/ruby_llm/message.rb +233 -54
  128. data/lib/ruby_llm/model/modalities.rb +17 -4
  129. data/lib/ruby_llm/model/pricing.rb +24 -5
  130. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  131. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  132. data/lib/ruby_llm/model.rb +244 -2
  133. data/lib/ruby_llm/models/aliases.rb +41 -0
  134. data/lib/ruby_llm/models/registry.rb +165 -0
  135. data/lib/ruby_llm/models/schema.rb +99 -0
  136. data/lib/ruby_llm/models.json +73261 -33733
  137. data/lib/ruby_llm/models.rb +477 -215
  138. data/lib/ruby_llm/moderation.rb +139 -26
  139. data/lib/ruby_llm/ocr.rb +112 -0
  140. data/lib/ruby_llm/prompt.rb +79 -0
  141. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  142. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  143. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  144. data/lib/ruby_llm/protocol.rb +662 -0
  145. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  146. data/lib/ruby_llm/protocols/anthropic/chat.rb +548 -0
  147. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  148. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  149. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  150. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  151. data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
  152. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  153. data/lib/ruby_llm/protocols/anthropic.rb +101 -0
  154. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  155. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
  156. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  157. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  158. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  159. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  160. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  161. data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
  162. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  163. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  164. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  165. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  166. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  167. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  168. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
  169. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  170. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  171. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  172. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  173. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  174. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  175. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  176. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  177. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  178. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  179. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  180. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  181. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  182. data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
  183. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  184. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  185. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  186. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  187. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  188. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  189. data/lib/ruby_llm/protocols/converse/chat.rb +695 -0
  190. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  191. data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
  192. data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
  193. data/lib/ruby_llm/protocols/converse.rb +54 -0
  194. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  195. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  196. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  197. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  198. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  199. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  200. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  201. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  202. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  203. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  204. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  208. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  209. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  210. data/lib/ruby_llm/protocols/files.rb +119 -0
  211. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  212. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  213. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  214. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
  215. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  216. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  217. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  218. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  219. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  220. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  221. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  222. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  223. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  224. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  225. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  226. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  227. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  228. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  229. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  230. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  231. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  232. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  233. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  234. data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
  235. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  236. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  237. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  238. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  239. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  240. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  241. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  242. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  243. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  244. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  245. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  246. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  247. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  248. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  249. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  250. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  251. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  252. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  253. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  254. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  255. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  256. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  257. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  258. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  259. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  260. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  261. data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
  262. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  263. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  264. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  265. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  266. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  267. data/lib/ruby_llm/protocols/responses.rb +35 -0
  268. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  269. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  270. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  271. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  272. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  273. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  274. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  275. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  276. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  277. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  278. data/lib/ruby_llm/provider.rb +560 -128
  279. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  280. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  281. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  282. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  283. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  284. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  285. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  286. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  287. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  288. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  289. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  290. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  291. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  292. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  293. data/lib/ruby_llm/providers/azure.rb +77 -78
  294. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  295. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  296. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  297. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  298. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  299. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  300. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  301. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  302. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  303. data/lib/ruby_llm/providers/cohere.rb +31 -0
  304. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  305. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  306. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  307. data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
  308. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  309. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  310. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  311. data/lib/ruby_llm/providers/gemini.rb +15 -8
  312. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  313. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  314. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  315. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  316. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  317. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  318. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  319. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  320. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  321. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  322. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  323. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  324. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  325. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  326. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  327. data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
  328. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  329. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  330. data/lib/ruby_llm/providers/mistral.rb +16 -4
  331. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  332. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  333. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  334. data/lib/ruby_llm/providers/ollama.rb +9 -8
  335. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  336. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  337. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  338. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  339. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  340. data/lib/ruby_llm/providers/openai.rb +92 -11
  341. data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
  342. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  343. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  344. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  345. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  346. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  347. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  348. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  349. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  350. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  351. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  352. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  353. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  354. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  355. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  356. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  357. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  358. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  359. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  360. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  361. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  362. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  363. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  364. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  365. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  366. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  367. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  368. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  369. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  370. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  371. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  372. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  373. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  374. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  375. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  376. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  377. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  378. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  379. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  380. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  381. data/lib/ruby_llm/providers/xai.rb +15 -5
  382. data/lib/ruby_llm/railtie.rb +7 -16
  383. data/lib/ruby_llm/rerank.rb +105 -0
  384. data/lib/ruby_llm/research_job.rb +241 -0
  385. data/lib/ruby_llm/search_results.rb +68 -0
  386. data/lib/ruby_llm/server_tool_call.rb +73 -0
  387. data/lib/ruby_llm/speech.rb +159 -0
  388. data/lib/ruby_llm/speech_chunk.rb +33 -0
  389. data/lib/ruby_llm/support/deprecator.rb +22 -0
  390. data/lib/ruby_llm/support/inspectable.rb +49 -0
  391. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  392. data/lib/ruby_llm/support/utils.rb +147 -0
  393. data/lib/ruby_llm/thinking.rb +127 -20
  394. data/lib/ruby_llm/tokenization.rb +59 -0
  395. data/lib/ruby_llm/tokens.rb +103 -33
  396. data/lib/ruby_llm/tool.rb +266 -91
  397. data/lib/ruby_llm/tool_call.rb +36 -3
  398. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  399. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  400. data/lib/ruby_llm/transcription.rb +138 -13
  401. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  402. data/lib/ruby_llm/transport/connection.rb +193 -0
  403. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  404. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  405. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  406. data/lib/ruby_llm/uploaded_file.rb +144 -0
  407. data/lib/ruby_llm/version.rb +2 -1
  408. data/lib/ruby_llm/video.rb +136 -0
  409. data/lib/ruby_llm/video_job.rb +150 -0
  410. data/lib/ruby_llm/workflow.rb +91 -0
  411. data/lib/ruby_llm.rb +380 -6
  412. data/lib/tasks/ruby_llm.rake +21 -16
  413. data/skills/rubyllm/SKILL.md +81 -0
  414. data/skills/rubyllm/agents/openai.yaml +4 -0
  415. metadata +339 -97
  416. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  417. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  418. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  419. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  420. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  421. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  422. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  423. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  424. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  425. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  426. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  427. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  428. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  429. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  430. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  431. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  432. data/lib/ruby_llm/aliases.rb +0 -41
  433. data/lib/ruby_llm/connection.rb +0 -159
  434. data/lib/ruby_llm/content.rb +0 -91
  435. data/lib/ruby_llm/deprecator.rb +0 -24
  436. data/lib/ruby_llm/error_middleware.rb +0 -81
  437. data/lib/ruby_llm/instrumentation.rb +0 -36
  438. data/lib/ruby_llm/mime_type.rb +0 -96
  439. data/lib/ruby_llm/model/info.rb +0 -164
  440. data/lib/ruby_llm/model_registry.rb +0 -39
  441. data/lib/ruby_llm/models_schema.json +0 -171
  442. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  443. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  444. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  445. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  446. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  447. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  448. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  449. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  450. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  451. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  452. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  453. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  454. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  455. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  456. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  457. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  458. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  459. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  460. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  461. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  462. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  463. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  464. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  465. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  466. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  467. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  468. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  469. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  470. data/lib/ruby_llm/streaming.rb +0 -179
  471. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  472. data/lib/ruby_llm/utils.rb +0 -130
  473. data/lib/tasks/models.rake +0 -593
  474. data/lib/tasks/release.rake +0 -94
  475. data/lib/tasks/vcr.rake +0 -124
@@ -0,0 +1,493 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class ChatCompletions
6
+ # Chat methods of the OpenAI API integration
7
+ module Chat
8
+ FINISH_REASONS = {
9
+ 'stop' => :stop, 'length' => :max_tokens, 'tool_calls' => :tool_calls,
10
+ 'function_call' => :tool_calls, 'content_filter' => :content_filter
11
+ }.freeze
12
+
13
+ OPENAI_INLINE_FILE_LIMIT = 50 * 1024 * 1024
14
+ OPENAI_FILE_UPLOAD_LIMIT = 512 * 1024 * 1024
15
+ PROMPT_CACHE_OPTIONS = %i[key ttl mode retention].freeze
16
+
17
+ def completion_url
18
+ 'chat/completions'
19
+ end
20
+
21
+ module_function
22
+
23
+ def finish_reasons = FINISH_REASONS
24
+
25
+ def normalize_finish_reason(reason)
26
+ return nil if reason.nil?
27
+
28
+ finish_reasons.fetch(reason.to_s) { reason.to_s.to_sym }
29
+ end
30
+
31
+ # OpenAI strict mode rejects an object whose properties are not all
32
+ # required, so a schema with optional properties goes out non-strict
33
+ # unless the caller asked for strict explicitly.
34
+ def schema_strict(schema)
35
+ schema.key?(:strict) ? schema[:strict] : strict_schema?(schema[:schema])
36
+ end
37
+
38
+ def strict_schema?(node)
39
+ case node
40
+ when Hash
41
+ return false if optional_properties?(node)
42
+
43
+ node.values.all? { |value| strict_schema?(value) }
44
+ when Array then node.all? { |value| strict_schema?(value) }
45
+ else true
46
+ end
47
+ end
48
+
49
+ def optional_properties?(node)
50
+ properties = node[:properties]
51
+ return false unless properties.is_a?(Hash)
52
+
53
+ (properties.keys.map(&:to_s) - Array(node[:required]).map(&:to_s)).any?
54
+ end
55
+
56
+ # rubocop:disable-next Metrics/PerceivedComplexity
57
+ def render_payload(messages, tools:, temperature:, model:, stream: false, max_output_tokens: nil, schema: nil,
58
+ thinking: nil, citations: false, caching: nil, tool_prefs: nil)
59
+ warn_unsupported_citations(model) if citations && !model.supports?(:citations)
60
+ tool_prefs ||= {}
61
+ payload = {
62
+ model: model.id,
63
+ messages: format_messages(messages, caching: caching),
64
+ stream: stream
65
+ }
66
+
67
+ payload[:temperature] = temperature unless temperature.nil?
68
+ payload[max_output_tokens_field(model)] = max_output_tokens unless max_output_tokens.nil?
69
+ if tools.any?
70
+ payload[:tools] = tools.map { |_, tool| tool_for(tool) }
71
+ payload[:tool_choice] = build_tool_choice(tool_prefs[:choice]) unless tool_prefs[:choice].nil?
72
+ payload[:parallel_tool_calls] = tool_prefs[:calls] == :many unless tool_prefs[:calls].nil?
73
+ end
74
+
75
+ if schema
76
+ payload[:response_format] = {
77
+ type: 'json_schema',
78
+ json_schema: {
79
+ name: schema[:name],
80
+ schema: schema[:schema],
81
+ strict: schema_strict(schema)
82
+ }
83
+ }
84
+ end
85
+
86
+ effort = resolve_effort(thinking)
87
+ payload[:reasoning_effort] = effort if effort
88
+
89
+ payload[:stream_options] = { include_usage: true } if stream
90
+ apply_prompt_cache_params(payload, messages, caching)
91
+ payload
92
+ end
93
+
94
+ # OpenAI and Azure reject max_tokens on their reasoning models and
95
+ # accept max_completion_tokens on every model; the rest of the wire
96
+ # format only knows max_tokens.
97
+ def max_output_tokens_field(_model)
98
+ %w[openai azure].include?(@provider.slug) ? :max_completion_tokens : :max_tokens
99
+ end
100
+
101
+ def warn_unsupported_citations(model)
102
+ RubyLLM.logger.warn(
103
+ "#{model.id} does not support citations according to the model registry. " \
104
+ 'with_citations may have no effect.'
105
+ )
106
+ end
107
+
108
+ def parse_completion_body(data, raw:)
109
+ raise Error.new(data.dig('error', 'message'), response: raw) if data.dig('error', 'message')
110
+
111
+ message_data = data.dig('choices', 0, 'message')
112
+ raise no_completion_message_error(data, raw) unless message_data
113
+
114
+ usage = data['usage'] || {}
115
+ thinking_tokens = thinking_tokens(usage)
116
+ content, thinking_from_blocks = extract_content_and_thinking(message_data['content'])
117
+ thinking_text = thinking_from_blocks || extract_thinking_text(message_data)
118
+ thinking_signature = extract_thinking_signature(message_data)
119
+
120
+ finish_reason = normalize_finish_reason(data.dig('choices', 0, 'finish_reason'))
121
+
122
+ Message.new(
123
+ role: :assistant,
124
+ content: content,
125
+ citations: extract_citations(message_data, data, content),
126
+ thinking: Thinking.build(text: thinking_text, signature: thinking_signature),
127
+ raw_reasoning: extract_raw_reasoning(message_data),
128
+ tool_calls: parse_tool_calls(message_data['tool_calls'], response: raw, finish_reason: finish_reason),
129
+ input_tokens: input_tokens(usage),
130
+ output_tokens: output_tokens(usage),
131
+ cache_read_tokens: cache_read_tokens(usage),
132
+ cache_write_tokens: cache_write_tokens(usage),
133
+ thinking_tokens: thinking_tokens,
134
+ server_tool_use: server_tool_use(usage),
135
+ reported_cost: reported_cost(usage),
136
+ finish_reason: finish_reason,
137
+ model: data['model'],
138
+ raw: raw
139
+ )
140
+ end
141
+
142
+ def reported_cost(_usage)
143
+ nil
144
+ end
145
+
146
+ def server_tool_use(usage)
147
+ usage['server_tool_use'] || usage['server_tool_use_details']
148
+ end
149
+
150
+ def no_completion_message_error(data, raw)
151
+ finish_reason = data.dig('choices', 0, 'finish_reason')
152
+ message = 'Provider returned no completion message'
153
+ message = "#{message} (finish_reason: #{finish_reason})" if finish_reason
154
+ Error.new(message, response: raw)
155
+ end
156
+
157
+ def input_tokens(usage)
158
+ return usage['prompt_cache_miss_tokens'] if usage['prompt_cache_miss_tokens']
159
+
160
+ prompt_tokens = usage['prompt_tokens']
161
+ return unless prompt_tokens
162
+
163
+ [prompt_tokens.to_i - cache_read_tokens(usage).to_i - cache_write_tokens(usage).to_i, 0].max
164
+ end
165
+
166
+ def output_tokens(usage)
167
+ completion_tokens = usage['completion_tokens']
168
+ return unless completion_tokens
169
+
170
+ completion_tokens = completion_tokens.to_i
171
+ generated_tokens = generated_tokens_from_total(usage)
172
+ return completion_tokens unless generated_tokens && generated_tokens > completion_tokens
173
+
174
+ generated_tokens
175
+ end
176
+
177
+ def generated_tokens_from_total(usage)
178
+ prompt_tokens = usage['prompt_tokens']
179
+ total_tokens = usage['total_tokens']
180
+ return unless prompt_tokens && total_tokens
181
+
182
+ [total_tokens.to_i - prompt_tokens.to_i, 0].max
183
+ end
184
+
185
+ def cache_read_tokens(usage)
186
+ usage.dig('prompt_tokens_details', 'cached_tokens') || usage['prompt_cache_hit_tokens']
187
+ end
188
+
189
+ def cache_write_tokens(usage)
190
+ usage.dig('prompt_tokens_details', 'cache_write_tokens') ||
191
+ usage.dig('input_tokens_details', 'cache_write_tokens') ||
192
+ 0
193
+ end
194
+
195
+ def thinking_tokens(usage)
196
+ usage.dig('completion_tokens_details', 'reasoning_tokens') || usage['reasoning_tokens']
197
+ end
198
+
199
+ def extract_citations(message_data, data, content)
200
+ annotations = parse_annotations(message_data['annotations'], content)
201
+ return annotations if annotations.any?
202
+
203
+ parse_root_citations(data)
204
+ end
205
+
206
+ def parse_annotations(annotations, content)
207
+ Array(annotations).filter_map do |annotation|
208
+ details = annotation['url_citation']
209
+ next unless details.is_a?(Hash)
210
+
211
+ start_index = details['start_index']
212
+ end_index = details['end_index']
213
+
214
+ Citation.new(
215
+ url: details['url'],
216
+ title: details['title'],
217
+ text: annotated_text(content, start_index, end_index),
218
+ start_index: start_index,
219
+ end_index: end_index
220
+ )
221
+ end
222
+ end
223
+
224
+ def annotated_text(content, start_index, end_index)
225
+ return nil unless content.is_a?(String) && start_index && end_index
226
+
227
+ content[start_index...end_index]
228
+ end
229
+
230
+ # Perplexity and xAI return search citations at the root of the response.
231
+ def parse_root_citations(data)
232
+ search_results = data['search_results']
233
+ return parse_search_results(search_results) if search_results.is_a?(Array) && search_results.any?
234
+
235
+ Array(data['citations']).each_with_index.filter_map do |url, index|
236
+ Citation.new(url: url, source_index: index) if url.is_a?(String)
237
+ end
238
+ end
239
+
240
+ def parse_search_results(results)
241
+ results.each_with_index.filter_map do |result, index|
242
+ next unless result.is_a?(Hash)
243
+
244
+ Citation.new(
245
+ url: result['url'],
246
+ title: result['title'],
247
+ cited_text: result['snippet'],
248
+ source_index: index
249
+ )
250
+ end
251
+ end
252
+
253
+ def apply_prompt_cache_params(payload, messages, caching)
254
+ return unless openai_prompt_caching?
255
+
256
+ payload.merge!(prompt_cache_params(caching)) if caching
257
+ force_explicit_cache_mode(payload) if caching != false && cache_boundaries?(messages)
258
+ end
259
+
260
+ def openai_prompt_caching?
261
+ true
262
+ end
263
+
264
+ def prompt_cache_params(caching)
265
+ options = prompt_cache_options(caching)
266
+ cache_options = build_prompt_cache_options(options)
267
+
268
+ {}.tap do |params|
269
+ params[:prompt_cache_key] = options[:key] if options[:key]
270
+ params[:prompt_cache_options] = cache_options unless cache_options.empty?
271
+ end
272
+ end
273
+
274
+ def build_prompt_cache_options(options)
275
+ ttl = options[:ttl] || retention_ttl(options[:retention])
276
+
277
+ {}.tap do |cache_options|
278
+ cache_options[:mode] = options[:mode] if options[:mode]
279
+ cache_options[:ttl] = ttl if ttl
280
+ end
281
+ end
282
+
283
+ def retention_ttl(retention)
284
+ return unless retention
285
+
286
+ RubyLLM.logger.warn(
287
+ 'with_caching retention: is deprecated; OpenAI replaced prompt_cache_retention ' \
288
+ 'with prompt_cache_options. Use ttl: instead.'
289
+ )
290
+ retention
291
+ end
292
+
293
+ def force_explicit_cache_mode(payload)
294
+ payload[:prompt_cache_options] = { mode: 'explicit' }.merge(payload[:prompt_cache_options] || {})
295
+ end
296
+
297
+ def cache_boundaries?(messages)
298
+ messages.any?(&:cache_until_here?)
299
+ end
300
+
301
+ def prompt_cache_options(caching)
302
+ options = caching.to_h.transform_keys(&:to_sym)
303
+ unsupported = options.keys - PROMPT_CACHE_OPTIONS
304
+ return options if unsupported.empty?
305
+
306
+ raise ArgumentError,
307
+ 'Chat Completions prompt caching accepts :key, :ttl, and :mode, ' \
308
+ "got #{format_cache_option_keys(unsupported)}"
309
+ end
310
+
311
+ def format_cache_option_keys(keys)
312
+ keys.map { |key| ":#{key}" }.join(', ')
313
+ end
314
+
315
+ def format_messages(messages, caching: nil)
316
+ messages_for_provider(messages)
317
+ .chunk_while { |previous, current| previous.tool_result? && current.tool_result? }
318
+ .flat_map { |group| format_message_group(group, caching: caching) }
319
+ end
320
+
321
+ # An assistant turn's tool results must stay consecutive, so the
322
+ # attachment carriers of a parallel round follow the whole run.
323
+ def format_message_group(group, caching: nil)
324
+ formatted = group.map { |msg| format_message(msg, caching: caching) }
325
+ carriers = group.select { |msg| msg.tool_result? && msg.attachments.any? }
326
+
327
+ formatted + carriers.map { |msg| tool_attachment_message(msg) }
328
+ end
329
+
330
+ def format_message(msg, caching: nil)
331
+ {
332
+ role: format_role(msg.role),
333
+ content: format_message_content(msg, caching: caching),
334
+ tool_calls: format_tool_calls(msg.tool_calls),
335
+ tool_call_id: msg.tool_call_id
336
+ }.compact.merge(format_thinking(msg))
337
+ end
338
+
339
+ # Chat Completions tool messages are text-only on the wire, so tool
340
+ # attachments ride a user message spliced in after the results.
341
+ def tool_attachment_message(msg)
342
+ parts = [Media.format_text("Attachments from tool call #{msg.tool_call_id}:")]
343
+ parts.concat(format_content(nil, msg.attachments))
344
+ { role: 'user', content: parts }
345
+ end
346
+
347
+ def messages_for_provider(messages)
348
+ system_messages, other_messages = messages.partition { |msg| msg.role == :system }
349
+ system_messages + other_messages
350
+ end
351
+
352
+ def format_message_content(msg, caching: nil, **)
353
+ content = format_content(msg.content, msg.tool_result? ? [] : msg.attachments)
354
+ return '' if content.nil? && thinking_only_assistant_message?(msg)
355
+ return inject_cache_breakpoint(content) if caching != false && msg.cache_until_here? && openai_prompt_caching?
356
+
357
+ content
358
+ end
359
+
360
+ def inject_cache_breakpoint(content)
361
+ parts = cache_breakpoint_parts(content)
362
+ return content unless parts&.last.is_a?(Hash)
363
+
364
+ parts[-1] = parts.last.merge(prompt_cache_breakpoint: { mode: 'explicit' })
365
+ parts
366
+ end
367
+
368
+ def cache_breakpoint_parts(content)
369
+ case content
370
+ when Array then content.dup
371
+ when String then [Media.format_text(content)] unless content.empty?
372
+ end
373
+ end
374
+
375
+ def thinking_only_assistant_message?(msg)
376
+ msg.role == :assistant && msg.thinking && !msg.tool_call?
377
+ end
378
+
379
+ def format_content(content, attachments = [])
380
+ Media.format_content(content, attachments)
381
+ end
382
+
383
+ def format_role(role)
384
+ case role
385
+ when :system
386
+ @config.openai_use_system_role ? 'system' : 'developer'
387
+ else
388
+ role.to_s
389
+ end
390
+ end
391
+
392
+ def resolve_effort(thinking)
393
+ return nil unless thinking
394
+
395
+ effort = thinking.respond_to?(:effort) ? thinking.effort : thinking
396
+ effort&.to_s
397
+ end
398
+
399
+ # safety_identifier is an OpenAI parameter; the other services on
400
+ # this wire format reject or ignore it, so only OpenAI's own
401
+ # endpoints receive it.
402
+ def apply_end_user(payload, identifier)
403
+ return super unless %w[openai azure].include?(@provider.slug)
404
+
405
+ payload.merge(safety_identifier: identifier)
406
+ end
407
+
408
+ def supports_provider_file_references?
409
+ @provider.slug == 'openai'
410
+ end
411
+
412
+ def default_large_file_upload_threshold
413
+ OPENAI_INLINE_FILE_LIMIT
414
+ end
415
+
416
+ def provider_file_upload_limit
417
+ OPENAI_FILE_UPLOAD_LIMIT
418
+ end
419
+
420
+ def provider_file_attachable?(attachment)
421
+ attachment.pdf?
422
+ end
423
+
424
+ def provider_file_upload_options(_attachment)
425
+ { purpose: 'user_data' }
426
+ end
427
+
428
+ def format_thinking(msg)
429
+ return {} unless msg.role == :assistant
430
+
431
+ thinking = msg.thinking
432
+ return {} unless thinking
433
+
434
+ payload = {}
435
+ if thinking.text
436
+ payload[:reasoning] = thinking.text
437
+ payload[:reasoning_content] = thinking.text
438
+ end
439
+ payload[:reasoning_signature] = thinking.signature if thinking.signature
440
+ payload
441
+ end
442
+
443
+ def extract_thinking_text(message_data)
444
+ candidate = message_data['reasoning_content'] || message_data['reasoning'] || message_data['thinking']
445
+ candidate.is_a?(String) ? candidate : nil
446
+ end
447
+
448
+ def extract_thinking_signature(message_data)
449
+ candidate = message_data['reasoning_signature'] || message_data['signature']
450
+ candidate.is_a?(String) ? candidate : nil
451
+ end
452
+
453
+ def extract_raw_reasoning(_message_data)
454
+ nil
455
+ end
456
+
457
+ def extract_content_and_thinking(content)
458
+ return [content, nil] unless content.is_a?(Array)
459
+
460
+ text = extract_text_from_blocks(content)
461
+ thinking = extract_thinking_from_blocks(content)
462
+
463
+ [text.empty? ? nil : text, thinking.empty? ? nil : thinking]
464
+ end
465
+
466
+ def extract_text_from_blocks(blocks)
467
+ blocks.filter_map do |block|
468
+ block['text'] if block['type'] == 'text' && block['text'].is_a?(String)
469
+ end.join
470
+ end
471
+
472
+ def extract_thinking_from_blocks(blocks)
473
+ blocks.filter_map do |block|
474
+ next unless block['type'] == 'thinking'
475
+
476
+ extract_thinking_text_from_block(block)
477
+ end.join
478
+ end
479
+
480
+ def extract_thinking_text_from_block(block)
481
+ thinking_block = block['thinking']
482
+ return thinking_block if thinking_block.is_a?(String)
483
+
484
+ if thinking_block.is_a?(Array)
485
+ return thinking_block.filter_map { |item| item['text'] if item['type'] == 'text' }.join
486
+ end
487
+
488
+ block['text'] if block['text'].is_a?(String)
489
+ end
490
+ end
491
+ end
492
+ end
493
+ end
@@ -0,0 +1,35 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class ChatCompletions
6
+ # OpenAI-compatible file-backed Batch API for embeddings.
7
+ module EmbeddingBatches
8
+ include Protocols::OpenAI::Batches
9
+
10
+ Response = Struct.new(:body)
11
+ private_constant :Response
12
+
13
+ private
14
+
15
+ def batch_endpoint
16
+ '/v1/embeddings'
17
+ end
18
+
19
+ def validate_batch_requests!(requests)
20
+ return if requests.all? { |request| embedding_payload?(request.fetch(:payload)) }
21
+
22
+ raise Error, "#{@provider.slug} embedding batch requests require embedding payloads"
23
+ end
24
+
25
+ def embedding_payload?(payload)
26
+ payload.key?(:input) || payload.key?('input')
27
+ end
28
+
29
+ def parse_batch_completion_response(body)
30
+ parse_embedding_response(Response.new(body), model: body['model'], text: nil)
31
+ end
32
+ end
33
+ end
34
+ end
35
+ end
@@ -0,0 +1,60 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class ChatCompletions
6
+ # Embeddings methods of the OpenAI API integration
7
+ module Embeddings
8
+ module_function
9
+
10
+ def embedding_url(...)
11
+ 'embeddings'
12
+ end
13
+
14
+ # rubocop:disable-next Lint/UnusedMethodArgument
15
+ def render_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil, provider_options: {})
16
+ {
17
+ model: model,
18
+ input: text,
19
+ dimensions: dimensions
20
+ }.compact.merge(provider_options)
21
+ end
22
+
23
+ def parse_embedding_response(response, model:, text:)
24
+ data = response.body
25
+ input_tokens = data.dig('usage', 'prompt_tokens')
26
+ rows = data['data']
27
+ single = rows.length == 1 && !text.is_a?(Array)
28
+
29
+ vectors = rows.map { |row| row['embedding'] }
30
+ vectors = vectors.first if single
31
+ sparse_vectors = parse_sparse_vectors(rows, single: single)
32
+
33
+ Embedding.new(vectors:, sparse_vectors:, model:, input_tokens:,
34
+ reported_cost: reported_cost(data['usage'] || {}))
35
+ end
36
+
37
+ # Sparse-capable models return a token-to-weight map beside the dense
38
+ # vector, under lexical_weights on BGE-M3 and sparse_embedding
39
+ # elsewhere. It is an extension: dense-only servers send neither, and
40
+ # then there is nothing to report.
41
+ def parse_sparse_vectors(rows, single:)
42
+ sparse = rows.map { |row| normalize_sparse_vector(row['sparse_embedding'] || row['lexical_weights']) }
43
+ return nil if sparse.all?(&:nil?)
44
+
45
+ single ? sparse.first : sparse
46
+ end
47
+
48
+ def normalize_sparse_vector(weights)
49
+ return nil unless weights.is_a?(Hash)
50
+
51
+ weights.to_h { |token, weight| [Integer(token), Float(weight)] }
52
+ end
53
+
54
+ def reported_cost(_usage)
55
+ nil
56
+ end
57
+ end
58
+ end
59
+ end
60
+ end