ruby_llm 1.16.0 → 2.0.0.rc2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (475) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +85 -32
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
  91. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
  92. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
  93. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
  94. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  96. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
  97. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
  98. data/lib/ruby_llm/accounting/usage.rb +245 -0
  99. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  100. data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
  101. data/lib/ruby_llm/active_record/batch.rb +97 -0
  102. data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
  103. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  104. data/lib/ruby_llm/active_record/model.rb +135 -0
  105. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  106. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  107. data/lib/ruby_llm/active_record/usage.rb +61 -0
  108. data/lib/ruby_llm/agent.rb +1065 -151
  109. data/lib/ruby_llm/aliases.json +268 -100
  110. data/lib/ruby_llm/attachment.rb +187 -48
  111. data/lib/ruby_llm/batch.rb +432 -0
  112. data/lib/ruby_llm/cached_content.rb +112 -0
  113. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  114. data/lib/ruby_llm/chat.rb +1127 -198
  115. data/lib/ruby_llm/chunk.rb +10 -0
  116. data/lib/ruby_llm/citation.rb +105 -0
  117. data/lib/ruby_llm/configuration.rb +261 -24
  118. data/lib/ruby_llm/context.rb +128 -6
  119. data/lib/ruby_llm/cost.rb +217 -80
  120. data/lib/ruby_llm/downloaded_file.rb +33 -0
  121. data/lib/ruby_llm/embedding.rb +121 -17
  122. data/lib/ruby_llm/embedding_request.rb +53 -0
  123. data/lib/ruby_llm/error.rb +159 -23
  124. data/lib/ruby_llm/fallback.rb +133 -0
  125. data/lib/ruby_llm/files/mime_type.rb +97 -0
  126. data/lib/ruby_llm/image.rb +153 -32
  127. data/lib/ruby_llm/message.rb +233 -54
  128. data/lib/ruby_llm/model/modalities.rb +17 -4
  129. data/lib/ruby_llm/model/pricing.rb +24 -5
  130. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  131. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  132. data/lib/ruby_llm/model.rb +244 -2
  133. data/lib/ruby_llm/models/aliases.rb +41 -0
  134. data/lib/ruby_llm/models/registry.rb +165 -0
  135. data/lib/ruby_llm/models/schema.rb +99 -0
  136. data/lib/ruby_llm/models.json +73261 -33733
  137. data/lib/ruby_llm/models.rb +477 -215
  138. data/lib/ruby_llm/moderation.rb +139 -26
  139. data/lib/ruby_llm/ocr.rb +112 -0
  140. data/lib/ruby_llm/prompt.rb +79 -0
  141. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  142. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  143. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  144. data/lib/ruby_llm/protocol.rb +662 -0
  145. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  146. data/lib/ruby_llm/protocols/anthropic/chat.rb +548 -0
  147. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  148. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  149. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  150. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  151. data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
  152. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  153. data/lib/ruby_llm/protocols/anthropic.rb +101 -0
  154. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  155. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
  156. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  157. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  158. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  159. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  160. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  161. data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
  162. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  163. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  164. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  165. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  166. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  167. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  168. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
  169. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  170. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  171. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  172. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  173. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  174. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  175. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  176. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  177. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  178. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  179. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  180. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  181. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  182. data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
  183. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  184. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  185. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  186. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  187. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  188. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  189. data/lib/ruby_llm/protocols/converse/chat.rb +695 -0
  190. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  191. data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
  192. data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
  193. data/lib/ruby_llm/protocols/converse.rb +54 -0
  194. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  195. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  196. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  197. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  198. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  199. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  200. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  201. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  202. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  203. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  204. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  208. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  209. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  210. data/lib/ruby_llm/protocols/files.rb +119 -0
  211. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  212. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  213. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  214. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
  215. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  216. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  217. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  218. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  219. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  220. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  221. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  222. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  223. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  224. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  225. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  226. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  227. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  228. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  229. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  230. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  231. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  232. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  233. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  234. data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
  235. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  236. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  237. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  238. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  239. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  240. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  241. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  242. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  243. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  244. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  245. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  246. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  247. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  248. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  249. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  250. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  251. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  252. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  253. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  254. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  255. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  256. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  257. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  258. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  259. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  260. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  261. data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
  262. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  263. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  264. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  265. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  266. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  267. data/lib/ruby_llm/protocols/responses.rb +35 -0
  268. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  269. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  270. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  271. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  272. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  273. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  274. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  275. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  276. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  277. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  278. data/lib/ruby_llm/provider.rb +560 -128
  279. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  280. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  281. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  282. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  283. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  284. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  285. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  286. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  287. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  288. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  289. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  290. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  291. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  292. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  293. data/lib/ruby_llm/providers/azure.rb +77 -78
  294. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  295. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  296. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  297. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  298. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  299. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  300. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  301. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  302. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  303. data/lib/ruby_llm/providers/cohere.rb +31 -0
  304. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  305. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  306. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  307. data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
  308. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  309. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  310. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  311. data/lib/ruby_llm/providers/gemini.rb +15 -8
  312. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  313. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  314. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  315. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  316. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  317. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  318. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  319. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  320. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  321. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  322. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  323. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  324. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  325. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  326. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  327. data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
  328. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  329. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  330. data/lib/ruby_llm/providers/mistral.rb +16 -4
  331. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  332. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  333. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  334. data/lib/ruby_llm/providers/ollama.rb +9 -8
  335. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  336. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  337. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  338. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  339. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  340. data/lib/ruby_llm/providers/openai.rb +92 -11
  341. data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
  342. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  343. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  344. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  345. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  346. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  347. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  348. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  349. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  350. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  351. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  352. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  353. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  354. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  355. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  356. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  357. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  358. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  359. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  360. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  361. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  362. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  363. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  364. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  365. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  366. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  367. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  368. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  369. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  370. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  371. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  372. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  373. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  374. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  375. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  376. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  377. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  378. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  379. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  380. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  381. data/lib/ruby_llm/providers/xai.rb +15 -5
  382. data/lib/ruby_llm/railtie.rb +7 -16
  383. data/lib/ruby_llm/rerank.rb +105 -0
  384. data/lib/ruby_llm/research_job.rb +241 -0
  385. data/lib/ruby_llm/search_results.rb +68 -0
  386. data/lib/ruby_llm/server_tool_call.rb +73 -0
  387. data/lib/ruby_llm/speech.rb +159 -0
  388. data/lib/ruby_llm/speech_chunk.rb +33 -0
  389. data/lib/ruby_llm/support/deprecator.rb +22 -0
  390. data/lib/ruby_llm/support/inspectable.rb +49 -0
  391. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  392. data/lib/ruby_llm/support/utils.rb +147 -0
  393. data/lib/ruby_llm/thinking.rb +127 -20
  394. data/lib/ruby_llm/tokenization.rb +59 -0
  395. data/lib/ruby_llm/tokens.rb +103 -33
  396. data/lib/ruby_llm/tool.rb +266 -91
  397. data/lib/ruby_llm/tool_call.rb +36 -3
  398. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  399. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  400. data/lib/ruby_llm/transcription.rb +138 -13
  401. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  402. data/lib/ruby_llm/transport/connection.rb +193 -0
  403. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  404. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  405. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  406. data/lib/ruby_llm/uploaded_file.rb +144 -0
  407. data/lib/ruby_llm/version.rb +2 -1
  408. data/lib/ruby_llm/video.rb +136 -0
  409. data/lib/ruby_llm/video_job.rb +150 -0
  410. data/lib/ruby_llm/workflow.rb +91 -0
  411. data/lib/ruby_llm.rb +380 -6
  412. data/lib/tasks/ruby_llm.rake +21 -16
  413. data/skills/rubyllm/SKILL.md +81 -0
  414. data/skills/rubyllm/agents/openai.yaml +4 -0
  415. metadata +339 -97
  416. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  417. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  418. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  419. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  420. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  421. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  422. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  423. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  424. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  425. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  426. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  427. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  428. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  429. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  430. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  431. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  432. data/lib/ruby_llm/aliases.rb +0 -41
  433. data/lib/ruby_llm/connection.rb +0 -159
  434. data/lib/ruby_llm/content.rb +0 -91
  435. data/lib/ruby_llm/deprecator.rb +0 -24
  436. data/lib/ruby_llm/error_middleware.rb +0 -81
  437. data/lib/ruby_llm/instrumentation.rb +0 -36
  438. data/lib/ruby_llm/mime_type.rb +0 -96
  439. data/lib/ruby_llm/model/info.rb +0 -164
  440. data/lib/ruby_llm/model_registry.rb +0 -39
  441. data/lib/ruby_llm/models_schema.json +0 -171
  442. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  443. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  444. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  445. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  446. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  447. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  448. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  449. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  450. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  451. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  452. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  453. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  454. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  455. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  456. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  457. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  458. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  459. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  460. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  461. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  462. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  463. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  464. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  465. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  466. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  467. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  468. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  469. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  470. data/lib/ruby_llm/streaming.rb +0 -179
  471. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  472. data/lib/ruby_llm/utils.rb +0 -130
  473. data/lib/tasks/models.rake +0 -593
  474. data/lib/tasks/release.rake +0 -94
  475. data/lib/tasks/vcr.rake +0 -124
@@ -0,0 +1,162 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Gemini Batch API with inlined generateContent requests.
7
+ module Batches
8
+ include RubyLLM::Batch::Helpers
9
+ include Gemini::EmbeddingBatches
10
+
11
+ # The wire enum is BATCH_STATE_*; the SDKs print JOB_STATE_*. Match the
12
+ # suffix so either spelling works.
13
+ TERMINAL = %w[SUCCEEDED FAILED CANCELLED EXPIRED].freeze
14
+ private_constant :TERMINAL
15
+
16
+ def create_batch(requests)
17
+ model = single_batch_model!(requests, 'gemini')
18
+ action = embedding_batch?(requests) ? 'asyncBatchEmbedContent' : 'batchGenerateContent'
19
+ response = @connection.post("models/#{model}:#{action}", {
20
+ batch: {
21
+ displayName: "ruby_llm_#{SecureRandom.hex(8)}",
22
+ inputConfig: {
23
+ requests: {
24
+ requests: requests.flat_map do |request|
25
+ if embedding_batch_payload?(request.fetch(:payload))
26
+ embedding_batch_requests(request, model)
27
+ else
28
+ [gemini_batch_request(request, model)]
29
+ end
30
+ end
31
+ }
32
+ }
33
+ }
34
+ }, idempotent: false)
35
+
36
+ parse_batch_response(response.body)
37
+ end
38
+
39
+ def find_batch(id)
40
+ parse_batch_response @connection.get(batch_name(id)).body
41
+ end
42
+
43
+ def cancel_batch(id)
44
+ @connection.post("#{batch_name(id)}:cancel", {})
45
+ find_batch(id)
46
+ end
47
+
48
+ # Inline answers are correlated by the key we sent (the submission index);
49
+ # Gemini also returns them in order, so we fall back to position.
50
+ def batch_results(id)
51
+ body = @connection.get(batch_name(id)).body
52
+ inlined = inline_batch_responses(body)
53
+ return parse_embedding_batch_results(inlined) if embedding_batch_response?(body)
54
+
55
+ inlined.each_with_index.map { |response, index| parse_inline_response(response, index) }
56
+ end
57
+
58
+ private
59
+
60
+ def inline_batch_responses(body)
61
+ body.dig('response', 'inlinedEmbedContentResponses', 'inlinedResponses') ||
62
+ body.dig('output', 'inlinedEmbedContentResponses', 'inlinedResponses') ||
63
+ body.dig('response', 'inlinedResponses', 'inlinedResponses') ||
64
+ body.dig('output', 'inlinedResponses', 'inlinedResponses') ||
65
+ body.dig('metadata', 'output', 'inlinedResponses', 'inlinedResponses') || []
66
+ end
67
+
68
+ def gemini_batch_request(request, model)
69
+ {
70
+ request: batch_schema_payload(batch_payload(request)).merge(model: "models/#{model}"),
71
+ metadata: {
72
+ custom_id: request[:custom_id]
73
+ }
74
+ }
75
+ end
76
+
77
+ # batchGenerateContent ignores responseJsonSchema and returns JSON of
78
+ # its own shape, while the legacy responseSchema field is honored, so
79
+ # batches carry the schema in Gemini's Schema dialect.
80
+ def batch_schema_payload(payload)
81
+ config = payload[:generationConfig]
82
+ return payload unless config.is_a?(Hash) && config.key?(:responseJsonSchema)
83
+
84
+ schema = response_schema(config[:responseJsonSchema])
85
+ payload.merge(generationConfig: config.except(:responseJsonSchema).merge(responseSchema: schema))
86
+ end
87
+
88
+ JSON_SCHEMA_ONLY_KEYS = %w[$schema $id additionalProperties strict].freeze
89
+ private_constant :JSON_SCHEMA_ONLY_KEYS
90
+
91
+ def response_schema(node)
92
+ case node
93
+ when Hash then response_schema_hash(node)
94
+ when Array then node.map { |value| response_schema(value) }
95
+ else node
96
+ end
97
+ end
98
+
99
+ def response_schema_hash(node)
100
+ schema = node.each_with_object({}) do |(key, value), converted|
101
+ next if JSON_SCHEMA_ONLY_KEYS.include?(key.to_s)
102
+
103
+ converted[key] = if key.to_s == 'properties' && value.is_a?(Hash)
104
+ value.transform_values { |property| response_schema(property) }
105
+ else
106
+ response_schema(value)
107
+ end
108
+ end
109
+ nullable_type(schema)
110
+ end
111
+
112
+ def nullable_type(schema)
113
+ key = schema.key?(:type) ? :type : 'type'
114
+ types = Array(schema[key])
115
+ return schema unless types.length > 1 && types.include?('null')
116
+
117
+ schema.merge(key => (types - ['null']).first, nullable: true)
118
+ end
119
+
120
+ def batch_name(id)
121
+ id.to_s.start_with?('batches/') ? id : "batches/#{id}"
122
+ end
123
+
124
+ # A batch starts as an Operation wrapping the batch in `metadata`; polling
125
+ # returns the batch directly. Read either shape.
126
+ def parse_batch_response(data)
127
+ batch = data['metadata'] || data
128
+ state = batch['state']
129
+ request_counts = batch['batchStats']
130
+
131
+ {
132
+ id: data['name'] || batch['name'],
133
+ raw_status: state,
134
+ completed: TERMINAL.any? { |terminal| state&.end_with?(terminal) },
135
+ request_counts:,
136
+ request_count: (request_counts&.fetch('requestCount', nil)&.to_i unless embedding_batch_response?(data))
137
+ }
138
+ end
139
+
140
+ def parse_batch_status(raw_status, completed:)
141
+ return :pending unless completed
142
+ return :succeeded if raw_status&.end_with?('SUCCEEDED')
143
+ return :cancelled if raw_status&.end_with?('CANCELLED')
144
+
145
+ :failed
146
+ end
147
+
148
+ def parse_inline_response(inline, index)
149
+ key = inline.dig('metadata', 'custom_id') || inline.dig('metadata', 'key')
150
+ index = batch_result_index(key) if key
151
+
152
+ if inline['response']
153
+ body = inline['response']
154
+ [index, parse_completion_body(body, raw: body)]
155
+ else
156
+ [index, nil, batch_failure(key || index, inline.dig('error', 'message'))]
157
+ end
158
+ end
159
+ end
160
+ end
161
+ end
162
+ end
@@ -0,0 +1,59 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Explicit content caching via the Gemini cachedContents API.
7
+ module Caches
8
+ module_function
9
+
10
+ def caches_url
11
+ 'cachedContents'
12
+ end
13
+
14
+ def cache_url(name)
15
+ cache_name(name)
16
+ end
17
+
18
+ def render_cache_payload(content, model:, ttl: nil, instructions: nil, attachments: [])
19
+ payload = {
20
+ model: cache_model_name(model),
21
+ contents: [{ role: 'user', parts: Media.format_content(content, attachments) }]
22
+ }
23
+ payload[:systemInstruction] = { parts: [{ text: instructions }] } if instructions
24
+ payload[:ttl] = format_cache_ttl(ttl) if ttl
25
+ payload
26
+ end
27
+
28
+ def render_cache_update_payload(ttl:)
29
+ { ttl: format_cache_ttl(ttl) }
30
+ end
31
+
32
+ def parse_cache_response(data)
33
+ CachedContent.new(
34
+ name: data['name'],
35
+ model: data['model']&.split('/')&.last,
36
+ provider_instance: @provider,
37
+ created_at: (Time.iso8601(data['createTime']) if data['createTime']),
38
+ expires_at: (Time.iso8601(data['expireTime']) if data['expireTime']),
39
+ tokens: data.dig('usageMetadata', 'totalTokenCount'),
40
+ metadata: data
41
+ )
42
+ end
43
+
44
+ def cache_name(name)
45
+ name = name.name if name.is_a?(CachedContent)
46
+ name.to_s.include?('/') ? name.to_s : "cachedContents/#{name}"
47
+ end
48
+
49
+ def cache_model_name(model_id)
50
+ "models/#{model_id}"
51
+ end
52
+
53
+ def format_cache_ttl(ttl)
54
+ ttl.is_a?(String) ? ttl : "#{ttl.to_i}s"
55
+ end
56
+ end
57
+ end
58
+ end
59
+ end
@@ -0,0 +1,453 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Chat methods for the Gemini API implementation
7
+ module Chat
8
+ FINISH_REASONS = {
9
+ 'STOP' => :stop, 'MAX_TOKENS' => :max_tokens,
10
+ 'SAFETY' => :content_filter, 'RECITATION' => :content_filter, 'BLOCKLIST' => :content_filter,
11
+ 'PROHIBITED_CONTENT' => :content_filter, 'SPII' => :content_filter, 'IMAGE_SAFETY' => :content_filter,
12
+ 'IMAGE_RECITATION' => :content_filter, 'IMAGE_PROHIBITED_CONTENT' => :content_filter,
13
+ 'MODEL_ARMOR' => :content_filter
14
+ }.freeze
15
+
16
+ GEMINI_INLINE_FILE_THRESHOLD = 20 * 1024 * 1024
17
+ VERTEX_INLINE_FILE_THRESHOLD = 7 * 1024 * 1024
18
+ GEMINI_FILE_UPLOAD_LIMIT = 2 * 1024 * 1024 * 1024
19
+
20
+ module_function
21
+
22
+ def finish_reasons = FINISH_REASONS
23
+
24
+ def normalize_finish_reason(reason)
25
+ return nil if reason.nil?
26
+
27
+ finish_reasons.fetch(reason.to_s) { reason.to_s.to_sym }
28
+ end
29
+
30
+ def completion_url
31
+ "models/#{@model.id}:generateContent"
32
+ end
33
+
34
+ # rubocop:disable-next Metrics/PerceivedComplexity,Lint/UnusedMethodArgument
35
+ def render_payload(messages, tools:, temperature:, model:, stream: false, max_output_tokens: nil, schema: nil,
36
+ thinking: nil, citations: false, caching: nil, tool_prefs: nil)
37
+ warn_unsupported_citations(model) if citations && !model.supports?(:citations)
38
+ tool_prefs ||= {}
39
+ payload = {
40
+ contents: format_messages(messages.reject { |msg| msg.role == :system }),
41
+ generationConfig: {}
42
+ }
43
+ system_instruction = format_system_instruction(messages)
44
+ payload[:systemInstruction] = system_instruction if system_instruction
45
+
46
+ payload[:generationConfig][:temperature] = temperature unless temperature.nil?
47
+ payload[:generationConfig][:maxOutputTokens] = max_output_tokens unless max_output_tokens.nil?
48
+
49
+ payload[:generationConfig].merge!(structured_output_config(schema)) if schema
50
+ payload[:generationConfig][:thinkingConfig] = build_thinking_config(model, thinking) if thinking&.enabled?
51
+
52
+ if tools.any?
53
+ payload[:tools] = format_tools(tools)
54
+ # Gemini doesn't support controlling parallel tool calls
55
+ payload[:toolConfig] = build_tool_config(tool_prefs[:choice]) unless tool_prefs[:choice].nil?
56
+ end
57
+
58
+ payload[:cachedContent] = cache_name(caching[:id]) if caching.is_a?(Hash) && caching[:id]
59
+ maybe_log_implicit_caching_note(messages, caching)
60
+
61
+ payload
62
+ end
63
+
64
+ def maybe_log_implicit_caching_note(messages, caching)
65
+ return if caching == false
66
+ return unless (caching && !caching[:id]) || messages.any?(&:cache_until_here?)
67
+
68
+ RubyLLM.logger.debug(
69
+ 'Gemini caches repeated prompt prefixes automatically (implicit caching). ' \
70
+ 'For explicit caching, create a cache with RubyLLM.cache and attach it with ' \
71
+ 'chat.with_caching(id: cache).'
72
+ )
73
+ end
74
+
75
+ def warn_unsupported_citations(model)
76
+ RubyLLM.logger.warn(
77
+ "#{model.id} does not support citations according to the model registry. " \
78
+ 'Gemini citations come from Google Search grounding: ' \
79
+ 'with_provider_options(tools: [{ google_search: {} }]).'
80
+ )
81
+ end
82
+
83
+ def count_tokens_url
84
+ "models/#{@model.id}:countTokens"
85
+ end
86
+
87
+ def render_count_tokens_payload(messages, model:, **options)
88
+ request = count_tokens_request(messages, model: model, **options)
89
+ { generateContentRequest: request.merge(model: "models/#{model.id}") }
90
+ end
91
+
92
+ def count_tokens_request(messages, tools:, model:, tool_prefs: nil, thinking: nil, schema: nil,
93
+ citations: false, caching: nil)
94
+ render_payload(
95
+ messages,
96
+ tools: tools,
97
+ tool_prefs: tool_prefs,
98
+ temperature: nil,
99
+ model: model,
100
+ schema: schema,
101
+ thinking: thinking,
102
+ citations: citations,
103
+ caching: caching
104
+ ).slice(:contents, :systemInstruction, :tools)
105
+ end
106
+
107
+ def parse_count_tokens_response(response)
108
+ response.body['totalTokens']
109
+ end
110
+
111
+ def build_thinking_config(_model, thinking)
112
+ return { includeThoughts: false, thinkingBudget: 0 } if thinking.enabled == false
113
+
114
+ config = { includeThoughts: true }
115
+
116
+ config[:thinkingLevel] = thinking.effort.to_s if thinking.effort
117
+ config[:thinkingBudget] = thinking.budget if thinking.budget.is_a?(Integer)
118
+ config[:thinkingBudget] = -1 if thinking.enabled == true
119
+
120
+ config
121
+ end
122
+
123
+ def supports_provider_file_references?
124
+ true
125
+ end
126
+
127
+ def default_large_file_upload_threshold
128
+ @provider.slug == 'vertexai' ? VERTEX_INLINE_FILE_THRESHOLD : GEMINI_INLINE_FILE_THRESHOLD
129
+ end
130
+
131
+ def provider_file_upload_limit
132
+ GEMINI_FILE_UPLOAD_LIMIT
133
+ end
134
+
135
+ def provider_file_attachable?(attachment)
136
+ attachment.image? || attachment.video? || attachment.audio? || attachment.pdf? || attachment.text?
137
+ end
138
+
139
+ private
140
+
141
+ def format_system_instruction(messages)
142
+ parts = messages.select { |msg| msg.role == :system }.flat_map do |msg|
143
+ text = msg.content.to_s
144
+ Media.format_content(text.empty? ? nil : text, msg.attachments)
145
+ end
146
+
147
+ { parts: parts } if parts.any?
148
+ end
149
+
150
+ def format_messages(messages)
151
+ MessageFormatter.new(self, messages).format
152
+ end
153
+
154
+ def format_role(role)
155
+ case role
156
+ when :assistant then 'model'
157
+ when :system, :tool then 'user'
158
+ else role.to_s
159
+ end
160
+ end
161
+
162
+ def format_parts(msg)
163
+ if msg.role == :assistant && msg.raw_content
164
+ msg.raw_content
165
+ elsif msg.tool_call?
166
+ format_tool_call(msg)
167
+ elsif msg.tool_result?
168
+ format_tool_result(msg)
169
+ else
170
+ format_message_parts(msg)
171
+ end
172
+ end
173
+
174
+ def format_message_parts(msg)
175
+ parts = []
176
+
177
+ parts << build_thought_part(msg.thinking) if msg.role == :assistant && msg.thinking
178
+
179
+ parts.concat(Media.format_content(msg.content, msg.attachments))
180
+ parts
181
+ end
182
+
183
+ def build_thought_part(thinking)
184
+ part = { thought: true }
185
+ part[:text] = thinking.text if thinking.text
186
+ part[:thoughtSignature] = thinking.signature if thinking.signature
187
+ part
188
+ end
189
+
190
+ def parse_completion_body(data, raw:)
191
+ parts = data.dig('candidates', 0, 'content', 'parts') || []
192
+ tool_calls = extract_tool_calls(data)
193
+ content, attachments = parse_content(data)
194
+
195
+ Message.new(
196
+ role: :assistant,
197
+ content: content,
198
+ attachments: attachments,
199
+ citations: extract_citations(data, content),
200
+ thinking: Thinking.build(
201
+ text: extract_thought_parts(parts),
202
+ signature: extract_thought_signature(parts)
203
+ ),
204
+ tool_calls: tool_calls,
205
+ server_tool_calls: extract_server_tool_calls(data, parts),
206
+ raw_content: parts.any? { |part| server_tool_part?(part) } ? parts : nil,
207
+ input_tokens: input_tokens(data),
208
+ output_tokens: calculate_output_tokens(data),
209
+ cache_read_tokens: data.dig('usageMetadata', 'cachedContentTokenCount'),
210
+ thinking_tokens: data.dig('usageMetadata', 'thoughtsTokenCount'),
211
+ finish_reason: normalize_finish_reason(
212
+ data.dig('candidates', 0, 'finishReason') || data.dig('promptFeedback', 'blockReason')
213
+ ),
214
+ model: data['modelVersion'] || @model&.id,
215
+ raw: raw
216
+ )
217
+ end
218
+
219
+ def input_tokens(data)
220
+ prompt_tokens = data.dig('usageMetadata', 'promptTokenCount')
221
+ return unless prompt_tokens
222
+
223
+ [prompt_tokens.to_i - data.dig('usageMetadata', 'cachedContentTokenCount').to_i, 0].max
224
+ end
225
+
226
+ def parse_content(data)
227
+ candidate = data.dig('candidates', 0)
228
+ return ['', []] unless candidate
229
+
230
+ parts = candidate.dig('content', 'parts')
231
+ return ['', []] unless parts&.any?
232
+
233
+ non_thought_parts = parts.reject { |part| part['thought'] }
234
+ return ['', []] unless non_thought_parts.any?
235
+
236
+ build_response_content(non_thought_parts)
237
+ end
238
+
239
+ # Code execution runs come back as parts inside the model turn and
240
+ # must be replayed in history; search and URL fetches come back as
241
+ # response-level metadata.
242
+ def server_tool_part?(part)
243
+ part.key?('executableCode') || part.key?('codeExecutionResult')
244
+ end
245
+
246
+ def extract_server_tool_calls(data, parts)
247
+ calls = parts.select { |part| server_tool_part?(part) }.map do |part|
248
+ ServerToolCall.new(
249
+ type: part.key?('executableCode') ? 'executable_code' : 'code_execution_result',
250
+ input: part['executableCode'],
251
+ result: part['codeExecutionResult'],
252
+ raw: part
253
+ )
254
+ end
255
+ calls.concat(metadata_server_tool_calls(data))
256
+ calls
257
+ end
258
+
259
+ def metadata_server_tool_calls(data)
260
+ candidate = data.dig('candidates', 0) || {}
261
+ calls = []
262
+
263
+ queries = candidate.dig('groundingMetadata', 'webSearchQueries')
264
+ if queries&.any?
265
+ calls << ServerToolCall.new(type: 'google_search', input: { 'queries' => queries },
266
+ raw: { 'webSearchQueries' => queries })
267
+ end
268
+
269
+ url_metadata = candidate['urlContextMetadata']
270
+ calls << ServerToolCall.new(type: 'url_context', result: url_metadata, raw: url_metadata) if url_metadata
271
+ calls
272
+ end
273
+
274
+ # Normalizes grounding metadata (Google Search grounding) into citations.
275
+ def extract_citations(data, content)
276
+ metadata = data.dig('candidates', 0, 'groundingMetadata')
277
+ return [] unless metadata
278
+
279
+ chunks = metadata['groundingChunks'] || []
280
+ supports = metadata['groundingSupports'] || []
281
+ return chunk_citations(chunks) if supports.empty?
282
+
283
+ supports.flat_map { |support| support_citations(support, chunks, content) }
284
+ end
285
+
286
+ def support_citations(support, chunks, content)
287
+ segment = support['segment'] || {}
288
+ end_index = segment['endIndex']
289
+ start_index = segment['startIndex'] || (0 if end_index)
290
+
291
+ Array(support['groundingChunkIndices']).filter_map do |index|
292
+ source = chunk_source(chunks[index])
293
+ next unless source
294
+
295
+ Citation.new(
296
+ url: source['uri'],
297
+ title: source['title'],
298
+ text: segment['text'],
299
+ start_index: byte_to_char_index(content, start_index),
300
+ end_index: byte_to_char_index(content, end_index),
301
+ source_index: index
302
+ )
303
+ end
304
+ end
305
+
306
+ def chunk_citations(chunks)
307
+ chunks.each_with_index.filter_map do |chunk, index|
308
+ source = chunk_source(chunk)
309
+ next unless source
310
+
311
+ Citation.new(url: source['uri'], title: source['title'], source_index: index)
312
+ end
313
+ end
314
+
315
+ def chunk_source(chunk)
316
+ return nil unless chunk.is_a?(Hash)
317
+
318
+ chunk['web'] || chunk['retrievedContext']
319
+ end
320
+
321
+ # Grounding segment indices are byte offsets into the UTF-8 response text.
322
+ def byte_to_char_index(content, byte_index)
323
+ return nil unless content.is_a?(String) && byte_index
324
+
325
+ content.byteslice(0, byte_index)&.length
326
+ end
327
+
328
+ def extract_thought_parts(parts)
329
+ thought_parts = parts.select { |p| p['thought'] }
330
+ thoughts = thought_parts.filter_map { |p| p['text'] }.join
331
+ thoughts.empty? ? nil : thoughts
332
+ end
333
+
334
+ def extract_thought_signature(parts)
335
+ parts.each do |part|
336
+ signature = part['thoughtSignature'] ||
337
+ part['thought_signature'] ||
338
+ part.dig('functionCall', 'thoughtSignature') ||
339
+ part.dig('functionCall', 'thought_signature')
340
+ return signature if signature
341
+ end
342
+
343
+ nil
344
+ end
345
+
346
+ def calculate_output_tokens(data)
347
+ candidates = data.dig('usageMetadata', 'candidatesTokenCount') || 0
348
+ thoughts = data.dig('usageMetadata', 'thoughtsTokenCount') || 0
349
+ candidates + thoughts
350
+ end
351
+
352
+ def build_json_schema(schema)
353
+ normalized = RubyLLM::Support::Utils.deep_dup(schema[:schema])
354
+ normalized.delete(:strict)
355
+ normalized.delete('strict')
356
+ RubyLLM::Support::Utils.deep_stringify_keys(normalized)
357
+ end
358
+
359
+ def structured_output_config(schema)
360
+ {
361
+ responseMimeType: 'application/json',
362
+ responseJsonSchema: build_json_schema(schema)
363
+ }
364
+ end
365
+
366
+ # formats a message
367
+ class MessageFormatter
368
+ def initialize(provider, messages)
369
+ @provider = provider
370
+ @messages = messages
371
+ @index = 0
372
+ @tool_call_names = {}
373
+ end
374
+
375
+ def format
376
+ formatted = []
377
+
378
+ while current_message
379
+ if tool_message?(current_message)
380
+ tool_parts, next_index = collect_tool_parts
381
+ formatted << build_tool_response(tool_parts)
382
+ @index = next_index
383
+ else
384
+ remember_tool_calls if current_message.tool_call?
385
+ formatted << build_standard_message(current_message)
386
+ @index += 1
387
+ end
388
+ end
389
+
390
+ formatted
391
+ end
392
+
393
+ private
394
+
395
+ def current_message
396
+ @messages[@index]
397
+ end
398
+
399
+ def tool_message?(message)
400
+ message&.role == :tool
401
+ end
402
+
403
+ def collect_tool_parts
404
+ results = []
405
+ index = @index
406
+
407
+ while tool_message?(@messages[index])
408
+ results << @messages[index]
409
+ index += 1
410
+ end
411
+
412
+ parts = in_call_order(results).flat_map do |tool_message|
413
+ format_tool_result(tool_message, @tool_call_names.delete(tool_message.tool_call_id))
414
+ end
415
+
416
+ [parts, index]
417
+ end
418
+
419
+ # functionResponse parts carry no call id, so Gemini pairs them with
420
+ # the functionCall parts by position. Results that arrived out of
421
+ # call order have to be put back in it.
422
+ def in_call_order(tool_messages)
423
+ order = @tool_call_names.keys
424
+ tool_messages.sort_by.with_index do |tool_message, index|
425
+ [order.index(tool_message.tool_call_id) || order.size, index]
426
+ end
427
+ end
428
+
429
+ def build_tool_response(parts)
430
+ { role: 'user', parts: parts }
431
+ end
432
+
433
+ def remember_tool_calls
434
+ current_message.tool_calls.each do |tool_call_id, tool_call|
435
+ @tool_call_names[tool_call_id] = tool_call.name
436
+ end
437
+ end
438
+
439
+ def build_standard_message(message)
440
+ {
441
+ role: @provider.send(:format_role, message.role),
442
+ parts: @provider.send(:format_parts, message)
443
+ }
444
+ end
445
+
446
+ def format_tool_result(message, tool_name)
447
+ @provider.send(:format_tool_result, message, tool_name)
448
+ end
449
+ end
450
+ end
451
+ end
452
+ end
453
+ end