ruby_llm 1.16.0 → 2.0.0.rc1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (474) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +85 -32
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -6
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +25 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
  91. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +100 -0
  92. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
  93. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
  94. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  96. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
  97. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
  98. data/lib/ruby_llm/accounting/usage.rb +245 -0
  99. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  100. data/lib/ruby_llm/active_record/attachment_helpers.rb +180 -0
  101. data/lib/ruby_llm/active_record/batch.rb +97 -0
  102. data/lib/ruby_llm/active_record/chat_methods.rb +823 -305
  103. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  104. data/lib/ruby_llm/active_record/model.rb +135 -0
  105. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  106. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  107. data/lib/ruby_llm/active_record/usage.rb +61 -0
  108. data/lib/ruby_llm/agent.rb +1065 -151
  109. data/lib/ruby_llm/aliases.json +268 -100
  110. data/lib/ruby_llm/attachment.rb +187 -48
  111. data/lib/ruby_llm/batch.rb +432 -0
  112. data/lib/ruby_llm/cached_content.rb +112 -0
  113. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  114. data/lib/ruby_llm/chat.rb +1127 -198
  115. data/lib/ruby_llm/chunk.rb +10 -0
  116. data/lib/ruby_llm/citation.rb +105 -0
  117. data/lib/ruby_llm/configuration.rb +261 -24
  118. data/lib/ruby_llm/context.rb +128 -6
  119. data/lib/ruby_llm/cost.rb +217 -80
  120. data/lib/ruby_llm/downloaded_file.rb +33 -0
  121. data/lib/ruby_llm/embedding.rb +121 -17
  122. data/lib/ruby_llm/embedding_request.rb +53 -0
  123. data/lib/ruby_llm/error.rb +159 -23
  124. data/lib/ruby_llm/fallback.rb +133 -0
  125. data/lib/ruby_llm/files/mime_type.rb +97 -0
  126. data/lib/ruby_llm/image.rb +153 -32
  127. data/lib/ruby_llm/message.rb +233 -54
  128. data/lib/ruby_llm/model/modalities.rb +17 -4
  129. data/lib/ruby_llm/model/pricing.rb +24 -5
  130. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  131. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  132. data/lib/ruby_llm/model.rb +244 -2
  133. data/lib/ruby_llm/models/aliases.rb +41 -0
  134. data/lib/ruby_llm/models/registry.rb +165 -0
  135. data/lib/ruby_llm/models/schema.rb +99 -0
  136. data/lib/ruby_llm/models.json +71172 -33253
  137. data/lib/ruby_llm/models.rb +477 -215
  138. data/lib/ruby_llm/moderation.rb +139 -26
  139. data/lib/ruby_llm/ocr.rb +112 -0
  140. data/lib/ruby_llm/prompt.rb +79 -0
  141. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  142. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  143. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  144. data/lib/ruby_llm/protocol.rb +662 -0
  145. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  146. data/lib/ruby_llm/protocols/anthropic/chat.rb +540 -0
  147. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  148. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  149. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  150. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  151. data/lib/ruby_llm/protocols/anthropic/streaming.rb +166 -0
  152. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  153. data/lib/ruby_llm/protocols/anthropic.rb +100 -0
  154. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  155. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
  156. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  157. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  158. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  159. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  160. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  161. data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
  162. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  163. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  164. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  165. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  166. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  167. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  168. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
  169. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  170. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  171. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  172. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  173. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  174. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  175. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  176. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  177. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  178. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  179. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  180. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  181. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  182. data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
  183. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  184. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  185. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  186. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  187. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  188. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  189. data/lib/ruby_llm/protocols/converse/chat.rb +685 -0
  190. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  191. data/lib/ruby_llm/protocols/converse/streaming.rb +424 -0
  192. data/lib/ruby_llm/protocols/converse.rb +54 -0
  193. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  194. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  195. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  196. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  197. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  198. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  199. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  200. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  201. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  202. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  203. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  204. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  208. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  209. data/lib/ruby_llm/protocols/files.rb +119 -0
  210. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  211. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  212. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  213. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
  214. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  215. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  216. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  217. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  218. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  219. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  220. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  221. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  222. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  223. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  224. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  225. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  226. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  227. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  228. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  229. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  230. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  231. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  232. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  233. data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
  234. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  235. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  236. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  237. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  238. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  239. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  240. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  241. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  242. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  243. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  244. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  245. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  246. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  247. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  248. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  249. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  250. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  251. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  252. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  253. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  254. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  255. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  256. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  257. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  258. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  259. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  260. data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
  261. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  262. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  263. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  264. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  265. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  266. data/lib/ruby_llm/protocols/responses.rb +35 -0
  267. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  268. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  269. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  270. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  271. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  272. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  273. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  274. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  275. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  276. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  277. data/lib/ruby_llm/provider.rb +560 -128
  278. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  279. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  280. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  281. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  282. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  283. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  284. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  285. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  286. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  287. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  288. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  289. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  290. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  291. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  292. data/lib/ruby_llm/providers/azure.rb +77 -78
  293. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  294. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  295. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  296. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  297. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  298. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  299. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  300. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  301. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  302. data/lib/ruby_llm/providers/cohere.rb +31 -0
  303. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  304. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  305. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  306. data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
  307. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  308. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  309. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  310. data/lib/ruby_llm/providers/gemini.rb +15 -8
  311. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  312. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  313. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  314. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  315. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  316. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  317. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  318. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  319. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  320. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  321. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  322. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  323. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  324. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  325. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  326. data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
  327. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  328. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  329. data/lib/ruby_llm/providers/mistral.rb +16 -4
  330. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  331. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  332. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  333. data/lib/ruby_llm/providers/ollama.rb +9 -8
  334. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  335. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  336. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  337. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  338. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  339. data/lib/ruby_llm/providers/openai.rb +92 -11
  340. data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
  341. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  342. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  343. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  344. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  345. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  346. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  347. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  348. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  349. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  350. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  351. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  352. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  353. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  354. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  355. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  356. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  357. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  358. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  359. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  360. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  361. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  362. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  363. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  364. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  365. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  366. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  367. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  368. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  369. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  370. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  371. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  372. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  373. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  374. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  375. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  376. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  377. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  378. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  379. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  380. data/lib/ruby_llm/providers/xai.rb +15 -5
  381. data/lib/ruby_llm/railtie.rb +7 -16
  382. data/lib/ruby_llm/rerank.rb +105 -0
  383. data/lib/ruby_llm/research_job.rb +241 -0
  384. data/lib/ruby_llm/search_results.rb +68 -0
  385. data/lib/ruby_llm/server_tool_call.rb +73 -0
  386. data/lib/ruby_llm/speech.rb +159 -0
  387. data/lib/ruby_llm/speech_chunk.rb +33 -0
  388. data/lib/ruby_llm/support/deprecator.rb +22 -0
  389. data/lib/ruby_llm/support/inspectable.rb +49 -0
  390. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  391. data/lib/ruby_llm/support/utils.rb +147 -0
  392. data/lib/ruby_llm/thinking.rb +127 -20
  393. data/lib/ruby_llm/tokenization.rb +59 -0
  394. data/lib/ruby_llm/tokens.rb +103 -33
  395. data/lib/ruby_llm/tool.rb +266 -91
  396. data/lib/ruby_llm/tool_call.rb +36 -3
  397. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  398. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  399. data/lib/ruby_llm/transcription.rb +138 -13
  400. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  401. data/lib/ruby_llm/transport/connection.rb +193 -0
  402. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  403. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  404. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  405. data/lib/ruby_llm/uploaded_file.rb +144 -0
  406. data/lib/ruby_llm/version.rb +2 -1
  407. data/lib/ruby_llm/video.rb +136 -0
  408. data/lib/ruby_llm/video_job.rb +150 -0
  409. data/lib/ruby_llm/workflow.rb +91 -0
  410. data/lib/ruby_llm.rb +380 -6
  411. data/lib/tasks/ruby_llm.rake +21 -16
  412. data/skills/rubyllm/SKILL.md +81 -0
  413. data/skills/rubyllm/agents/openai.yaml +4 -0
  414. metadata +338 -97
  415. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  416. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  417. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  418. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  419. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  420. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  421. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  422. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  423. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  424. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  425. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  426. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  427. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  428. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  429. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  430. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  431. data/lib/ruby_llm/aliases.rb +0 -41
  432. data/lib/ruby_llm/connection.rb +0 -159
  433. data/lib/ruby_llm/content.rb +0 -91
  434. data/lib/ruby_llm/deprecator.rb +0 -24
  435. data/lib/ruby_llm/error_middleware.rb +0 -81
  436. data/lib/ruby_llm/instrumentation.rb +0 -36
  437. data/lib/ruby_llm/mime_type.rb +0 -96
  438. data/lib/ruby_llm/model/info.rb +0 -164
  439. data/lib/ruby_llm/model_registry.rb +0 -39
  440. data/lib/ruby_llm/models_schema.json +0 -171
  441. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  442. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  443. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  444. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  445. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  446. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  447. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  448. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  449. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  450. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  451. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  452. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  453. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  454. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  455. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  456. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  457. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  458. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  459. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  460. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  461. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  462. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  463. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  464. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  465. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  466. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  467. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  468. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  469. data/lib/ruby_llm/streaming.rb +0 -179
  470. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  471. data/lib/ruby_llm/utils.rb +0 -130
  472. data/lib/tasks/models.rake +0 -593
  473. data/lib/tasks/release.rake +0 -94
  474. data/lib/tasks/vcr.rake +0 -124
@@ -0,0 +1,453 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Chat methods for the Gemini API implementation
7
+ module Chat
8
+ FINISH_REASONS = {
9
+ 'STOP' => :stop, 'MAX_TOKENS' => :max_tokens,
10
+ 'SAFETY' => :content_filter, 'RECITATION' => :content_filter, 'BLOCKLIST' => :content_filter,
11
+ 'PROHIBITED_CONTENT' => :content_filter, 'SPII' => :content_filter, 'IMAGE_SAFETY' => :content_filter,
12
+ 'IMAGE_RECITATION' => :content_filter, 'IMAGE_PROHIBITED_CONTENT' => :content_filter,
13
+ 'MODEL_ARMOR' => :content_filter
14
+ }.freeze
15
+
16
+ GEMINI_INLINE_FILE_THRESHOLD = 20 * 1024 * 1024
17
+ VERTEX_INLINE_FILE_THRESHOLD = 7 * 1024 * 1024
18
+ GEMINI_FILE_UPLOAD_LIMIT = 2 * 1024 * 1024 * 1024
19
+
20
+ module_function
21
+
22
+ def finish_reasons = FINISH_REASONS
23
+
24
+ def normalize_finish_reason(reason)
25
+ return nil if reason.nil?
26
+
27
+ finish_reasons.fetch(reason.to_s) { reason.to_s.to_sym }
28
+ end
29
+
30
+ def completion_url
31
+ "models/#{@model.id}:generateContent"
32
+ end
33
+
34
+ # rubocop:disable-next Metrics/PerceivedComplexity,Lint/UnusedMethodArgument
35
+ def render_payload(messages, tools:, temperature:, model:, stream: false, max_output_tokens: nil, schema: nil,
36
+ thinking: nil, citations: false, caching: nil, tool_prefs: nil)
37
+ warn_unsupported_citations(model) if citations && !model.supports?(:citations)
38
+ tool_prefs ||= {}
39
+ payload = {
40
+ contents: format_messages(messages.reject { |msg| msg.role == :system }),
41
+ generationConfig: {}
42
+ }
43
+ system_instruction = format_system_instruction(messages)
44
+ payload[:systemInstruction] = system_instruction if system_instruction
45
+
46
+ payload[:generationConfig][:temperature] = temperature unless temperature.nil?
47
+ payload[:generationConfig][:maxOutputTokens] = max_output_tokens unless max_output_tokens.nil?
48
+
49
+ payload[:generationConfig].merge!(structured_output_config(schema)) if schema
50
+ payload[:generationConfig][:thinkingConfig] = build_thinking_config(model, thinking) if thinking&.enabled?
51
+
52
+ if tools.any?
53
+ payload[:tools] = format_tools(tools)
54
+ # Gemini doesn't support controlling parallel tool calls
55
+ payload[:toolConfig] = build_tool_config(tool_prefs[:choice]) unless tool_prefs[:choice].nil?
56
+ end
57
+
58
+ payload[:cachedContent] = cache_name(caching[:id]) if caching.is_a?(Hash) && caching[:id]
59
+ maybe_log_implicit_caching_note(messages, caching)
60
+
61
+ payload
62
+ end
63
+
64
+ def maybe_log_implicit_caching_note(messages, caching)
65
+ return if caching == false
66
+ return unless (caching && !caching[:id]) || messages.any?(&:cache_until_here?)
67
+
68
+ RubyLLM.logger.debug(
69
+ 'Gemini caches repeated prompt prefixes automatically (implicit caching). ' \
70
+ 'For explicit caching, create a cache with RubyLLM.cache and attach it with ' \
71
+ 'chat.with_caching(id: cache).'
72
+ )
73
+ end
74
+
75
+ def warn_unsupported_citations(model)
76
+ RubyLLM.logger.warn(
77
+ "#{model.id} does not support citations according to the model registry. " \
78
+ 'Gemini citations come from Google Search grounding: ' \
79
+ 'with_provider_options(tools: [{ google_search: {} }]).'
80
+ )
81
+ end
82
+
83
+ def count_tokens_url
84
+ "models/#{@model.id}:countTokens"
85
+ end
86
+
87
+ def render_count_tokens_payload(messages, model:, **options)
88
+ request = count_tokens_request(messages, model: model, **options)
89
+ { generateContentRequest: request.merge(model: "models/#{model.id}") }
90
+ end
91
+
92
+ def count_tokens_request(messages, tools:, model:, tool_prefs: nil, thinking: nil, schema: nil,
93
+ citations: false, caching: nil)
94
+ render_payload(
95
+ messages,
96
+ tools: tools,
97
+ tool_prefs: tool_prefs,
98
+ temperature: nil,
99
+ model: model,
100
+ schema: schema,
101
+ thinking: thinking,
102
+ citations: citations,
103
+ caching: caching
104
+ ).slice(:contents, :systemInstruction, :tools)
105
+ end
106
+
107
+ def parse_count_tokens_response(response)
108
+ response.body['totalTokens']
109
+ end
110
+
111
+ def build_thinking_config(_model, thinking)
112
+ return { includeThoughts: false, thinkingBudget: 0 } if thinking.enabled == false
113
+
114
+ config = { includeThoughts: true }
115
+
116
+ config[:thinkingLevel] = thinking.effort.to_s if thinking.effort
117
+ config[:thinkingBudget] = thinking.budget if thinking.budget.is_a?(Integer)
118
+ config[:thinkingBudget] = -1 if thinking.enabled == true
119
+
120
+ config
121
+ end
122
+
123
+ def supports_provider_file_references?
124
+ true
125
+ end
126
+
127
+ def default_large_file_upload_threshold
128
+ @provider.slug == 'vertexai' ? VERTEX_INLINE_FILE_THRESHOLD : GEMINI_INLINE_FILE_THRESHOLD
129
+ end
130
+
131
+ def provider_file_upload_limit
132
+ GEMINI_FILE_UPLOAD_LIMIT
133
+ end
134
+
135
+ def provider_file_attachable?(attachment)
136
+ attachment.image? || attachment.video? || attachment.audio? || attachment.pdf? || attachment.text?
137
+ end
138
+
139
+ private
140
+
141
+ def format_system_instruction(messages)
142
+ parts = messages.select { |msg| msg.role == :system }.flat_map do |msg|
143
+ text = msg.content.to_s
144
+ Media.format_content(text.empty? ? nil : text, msg.attachments)
145
+ end
146
+
147
+ { parts: parts } if parts.any?
148
+ end
149
+
150
+ def format_messages(messages)
151
+ MessageFormatter.new(self, messages).format
152
+ end
153
+
154
+ def format_role(role)
155
+ case role
156
+ when :assistant then 'model'
157
+ when :system, :tool then 'user'
158
+ else role.to_s
159
+ end
160
+ end
161
+
162
+ def format_parts(msg)
163
+ if msg.role == :assistant && msg.raw_content
164
+ msg.raw_content
165
+ elsif msg.tool_call?
166
+ format_tool_call(msg)
167
+ elsif msg.tool_result?
168
+ format_tool_result(msg)
169
+ else
170
+ format_message_parts(msg)
171
+ end
172
+ end
173
+
174
+ def format_message_parts(msg)
175
+ parts = []
176
+
177
+ parts << build_thought_part(msg.thinking) if msg.role == :assistant && msg.thinking
178
+
179
+ parts.concat(Media.format_content(msg.content, msg.attachments))
180
+ parts
181
+ end
182
+
183
+ def build_thought_part(thinking)
184
+ part = { thought: true }
185
+ part[:text] = thinking.text if thinking.text
186
+ part[:thoughtSignature] = thinking.signature if thinking.signature
187
+ part
188
+ end
189
+
190
+ def parse_completion_body(data, raw:)
191
+ parts = data.dig('candidates', 0, 'content', 'parts') || []
192
+ tool_calls = extract_tool_calls(data)
193
+ content, attachments = parse_content(data)
194
+
195
+ Message.new(
196
+ role: :assistant,
197
+ content: content,
198
+ attachments: attachments,
199
+ citations: extract_citations(data, content),
200
+ thinking: Thinking.build(
201
+ text: extract_thought_parts(parts),
202
+ signature: extract_thought_signature(parts)
203
+ ),
204
+ tool_calls: tool_calls,
205
+ server_tool_calls: extract_server_tool_calls(data, parts),
206
+ raw_content: parts.any? { |part| server_tool_part?(part) } ? parts : nil,
207
+ input_tokens: input_tokens(data),
208
+ output_tokens: calculate_output_tokens(data),
209
+ cache_read_tokens: data.dig('usageMetadata', 'cachedContentTokenCount'),
210
+ thinking_tokens: data.dig('usageMetadata', 'thoughtsTokenCount'),
211
+ finish_reason: normalize_finish_reason(
212
+ data.dig('candidates', 0, 'finishReason') || data.dig('promptFeedback', 'blockReason')
213
+ ),
214
+ model: data['modelVersion'] || @model&.id,
215
+ raw: raw
216
+ )
217
+ end
218
+
219
+ def input_tokens(data)
220
+ prompt_tokens = data.dig('usageMetadata', 'promptTokenCount')
221
+ return unless prompt_tokens
222
+
223
+ [prompt_tokens.to_i - data.dig('usageMetadata', 'cachedContentTokenCount').to_i, 0].max
224
+ end
225
+
226
+ def parse_content(data)
227
+ candidate = data.dig('candidates', 0)
228
+ return ['', []] unless candidate
229
+
230
+ parts = candidate.dig('content', 'parts')
231
+ return ['', []] unless parts&.any?
232
+
233
+ non_thought_parts = parts.reject { |part| part['thought'] }
234
+ return ['', []] unless non_thought_parts.any?
235
+
236
+ build_response_content(non_thought_parts)
237
+ end
238
+
239
+ # Code execution runs come back as parts inside the model turn and
240
+ # must be replayed in history; search and URL fetches come back as
241
+ # response-level metadata.
242
+ def server_tool_part?(part)
243
+ part.key?('executableCode') || part.key?('codeExecutionResult')
244
+ end
245
+
246
+ def extract_server_tool_calls(data, parts)
247
+ calls = parts.select { |part| server_tool_part?(part) }.map do |part|
248
+ ServerToolCall.new(
249
+ type: part.key?('executableCode') ? 'executable_code' : 'code_execution_result',
250
+ input: part['executableCode'],
251
+ result: part['codeExecutionResult'],
252
+ raw: part
253
+ )
254
+ end
255
+ calls.concat(metadata_server_tool_calls(data))
256
+ calls
257
+ end
258
+
259
+ def metadata_server_tool_calls(data)
260
+ candidate = data.dig('candidates', 0) || {}
261
+ calls = []
262
+
263
+ queries = candidate.dig('groundingMetadata', 'webSearchQueries')
264
+ if queries&.any?
265
+ calls << ServerToolCall.new(type: 'google_search', input: { 'queries' => queries },
266
+ raw: { 'webSearchQueries' => queries })
267
+ end
268
+
269
+ url_metadata = candidate['urlContextMetadata']
270
+ calls << ServerToolCall.new(type: 'url_context', result: url_metadata, raw: url_metadata) if url_metadata
271
+ calls
272
+ end
273
+
274
+ # Normalizes grounding metadata (Google Search grounding) into citations.
275
+ def extract_citations(data, content)
276
+ metadata = data.dig('candidates', 0, 'groundingMetadata')
277
+ return [] unless metadata
278
+
279
+ chunks = metadata['groundingChunks'] || []
280
+ supports = metadata['groundingSupports'] || []
281
+ return chunk_citations(chunks) if supports.empty?
282
+
283
+ supports.flat_map { |support| support_citations(support, chunks, content) }
284
+ end
285
+
286
+ def support_citations(support, chunks, content)
287
+ segment = support['segment'] || {}
288
+ end_index = segment['endIndex']
289
+ start_index = segment['startIndex'] || (0 if end_index)
290
+
291
+ Array(support['groundingChunkIndices']).filter_map do |index|
292
+ source = chunk_source(chunks[index])
293
+ next unless source
294
+
295
+ Citation.new(
296
+ url: source['uri'],
297
+ title: source['title'],
298
+ text: segment['text'],
299
+ start_index: byte_to_char_index(content, start_index),
300
+ end_index: byte_to_char_index(content, end_index),
301
+ source_index: index
302
+ )
303
+ end
304
+ end
305
+
306
+ def chunk_citations(chunks)
307
+ chunks.each_with_index.filter_map do |chunk, index|
308
+ source = chunk_source(chunk)
309
+ next unless source
310
+
311
+ Citation.new(url: source['uri'], title: source['title'], source_index: index)
312
+ end
313
+ end
314
+
315
+ def chunk_source(chunk)
316
+ return nil unless chunk.is_a?(Hash)
317
+
318
+ chunk['web'] || chunk['retrievedContext']
319
+ end
320
+
321
+ # Grounding segment indices are byte offsets into the UTF-8 response text.
322
+ def byte_to_char_index(content, byte_index)
323
+ return nil unless content.is_a?(String) && byte_index
324
+
325
+ content.byteslice(0, byte_index)&.length
326
+ end
327
+
328
+ def extract_thought_parts(parts)
329
+ thought_parts = parts.select { |p| p['thought'] }
330
+ thoughts = thought_parts.filter_map { |p| p['text'] }.join
331
+ thoughts.empty? ? nil : thoughts
332
+ end
333
+
334
+ def extract_thought_signature(parts)
335
+ parts.each do |part|
336
+ signature = part['thoughtSignature'] ||
337
+ part['thought_signature'] ||
338
+ part.dig('functionCall', 'thoughtSignature') ||
339
+ part.dig('functionCall', 'thought_signature')
340
+ return signature if signature
341
+ end
342
+
343
+ nil
344
+ end
345
+
346
+ def calculate_output_tokens(data)
347
+ candidates = data.dig('usageMetadata', 'candidatesTokenCount') || 0
348
+ thoughts = data.dig('usageMetadata', 'thoughtsTokenCount') || 0
349
+ candidates + thoughts
350
+ end
351
+
352
+ def build_json_schema(schema)
353
+ normalized = RubyLLM::Support::Utils.deep_dup(schema[:schema])
354
+ normalized.delete(:strict)
355
+ normalized.delete('strict')
356
+ RubyLLM::Support::Utils.deep_stringify_keys(normalized)
357
+ end
358
+
359
+ def structured_output_config(schema)
360
+ {
361
+ responseMimeType: 'application/json',
362
+ responseJsonSchema: build_json_schema(schema)
363
+ }
364
+ end
365
+
366
+ # formats a message
367
+ class MessageFormatter
368
+ def initialize(provider, messages)
369
+ @provider = provider
370
+ @messages = messages
371
+ @index = 0
372
+ @tool_call_names = {}
373
+ end
374
+
375
+ def format
376
+ formatted = []
377
+
378
+ while current_message
379
+ if tool_message?(current_message)
380
+ tool_parts, next_index = collect_tool_parts
381
+ formatted << build_tool_response(tool_parts)
382
+ @index = next_index
383
+ else
384
+ remember_tool_calls if current_message.tool_call?
385
+ formatted << build_standard_message(current_message)
386
+ @index += 1
387
+ end
388
+ end
389
+
390
+ formatted
391
+ end
392
+
393
+ private
394
+
395
+ def current_message
396
+ @messages[@index]
397
+ end
398
+
399
+ def tool_message?(message)
400
+ message&.role == :tool
401
+ end
402
+
403
+ def collect_tool_parts
404
+ results = []
405
+ index = @index
406
+
407
+ while tool_message?(@messages[index])
408
+ results << @messages[index]
409
+ index += 1
410
+ end
411
+
412
+ parts = in_call_order(results).flat_map do |tool_message|
413
+ format_tool_result(tool_message, @tool_call_names.delete(tool_message.tool_call_id))
414
+ end
415
+
416
+ [parts, index]
417
+ end
418
+
419
+ # functionResponse parts carry no call id, so Gemini pairs them with
420
+ # the functionCall parts by position. Results that arrived out of
421
+ # call order have to be put back in it.
422
+ def in_call_order(tool_messages)
423
+ order = @tool_call_names.keys
424
+ tool_messages.sort_by.with_index do |tool_message, index|
425
+ [order.index(tool_message.tool_call_id) || order.size, index]
426
+ end
427
+ end
428
+
429
+ def build_tool_response(parts)
430
+ { role: 'user', parts: parts }
431
+ end
432
+
433
+ def remember_tool_calls
434
+ current_message.tool_calls.each do |tool_call_id, tool_call|
435
+ @tool_call_names[tool_call_id] = tool_call.name
436
+ end
437
+ end
438
+
439
+ def build_standard_message(message)
440
+ {
441
+ role: @provider.send(:format_role, message.role),
442
+ parts: @provider.send(:format_parts, message)
443
+ }
444
+ end
445
+
446
+ def format_tool_result(message, tool_name)
447
+ @provider.send(:format_tool_result, message, tool_name)
448
+ end
449
+ end
450
+ end
451
+ end
452
+ end
453
+ end
@@ -0,0 +1,86 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Inline embedding requests and results for Gemini's asynchronous Batch API.
7
+ module EmbeddingBatches
8
+ private
9
+
10
+ def embedding_batch?(requests)
11
+ kinds = requests.map { |request| embedding_batch_payload?(request.fetch(:payload)) }.uniq
12
+ raise Error, 'Gemini batches take chat or embedding requests, not both' if kinds.size > 1
13
+
14
+ kinds.first
15
+ end
16
+
17
+ def embedding_batch_payload?(payload)
18
+ payload.key?(:content) || payload.key?(:requests)
19
+ end
20
+
21
+ def embedding_batch_response?(data)
22
+ batch = data['metadata'] || data
23
+ batch['@type']&.end_with?('.EmbedContentBatch') ||
24
+ data.dig('response', '@type')&.end_with?('.EmbedContentBatchOutput') ||
25
+ data.dig('response', 'inlinedEmbedContentResponses') ||
26
+ data.dig('output', 'inlinedEmbedContentResponses')
27
+ end
28
+
29
+ def embedding_batch_requests(request, model)
30
+ payload = request.fetch(:payload)
31
+ array_input = payload.key?(:requests)
32
+ inputs = array_input ? payload.fetch(:requests) : [payload]
33
+ raise ArgumentError, 'Gemini embedding batches require at least one text per request' if inputs.empty?
34
+
35
+ inputs.each_with_index.map do |input, index|
36
+ {
37
+ request: input.merge(model: "models/#{model}"),
38
+ metadata: {
39
+ custom_id: request.fetch(:custom_id), model:, array_input:,
40
+ embedding_index: index, embedding_count: inputs.size
41
+ }
42
+ }
43
+ end
44
+ end
45
+
46
+ def parse_embedding_batch_results(responses)
47
+ groups = responses.each_with_index.group_by do |inline, index|
48
+ inline.dig('metadata', 'custom_id') || index.to_s
49
+ end
50
+ groups.filter_map do |key, indexed|
51
+ parse_embedding_batch_group(key, indexed.map(&:first))
52
+ end
53
+ end
54
+
55
+ def parse_embedding_batch_group(key, responses)
56
+ index = batch_result_index(key)
57
+ error = responses.find { |inline| inline['error'] }
58
+ return [index, nil, batch_failure(key, error.dig('error', 'message'))] if error
59
+
60
+ metadata = responses.first.fetch('metadata')
61
+ return if responses.size < metadata.fetch('embedding_count')
62
+
63
+ vectors = embedding_batch_vectors(responses)
64
+ return [index, nil, batch_failure(key, 'Gemini returned no embedding')] unless vectors
65
+
66
+ embedding = Embedding.new(
67
+ vectors: metadata['array_input'] ? vectors : vectors.first,
68
+ model: metadata.fetch('model'), input_tokens: embedding_batch_tokens(responses)
69
+ )
70
+ [index, embedding]
71
+ end
72
+
73
+ def embedding_batch_vectors(responses)
74
+ ordered = responses.sort_by { |inline| inline.dig('metadata', 'embedding_index') }
75
+ vectors = ordered.map { |inline| inline.dig('response', 'embedding', 'values') }
76
+ vectors if vectors.all? { |vector| vector && !vector.empty? }
77
+ end
78
+
79
+ def embedding_batch_tokens(responses)
80
+ counts = responses.filter_map { |inline| inline.dig('response', 'usageMetadata', 'promptTokenCount') }
81
+ counts.sum unless counts.empty?
82
+ end
83
+ end
84
+ end
85
+ end
86
+ end
@@ -0,0 +1,70 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ # Embeddings methods for the Gemini API integration
7
+ module Embeddings
8
+ module_function
9
+
10
+ def embedding_url(model:)
11
+ "models/#{model}:batchEmbedContents"
12
+ end
13
+
14
+ def render_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil, with: [],
15
+ provider_options: {})
16
+ requests = if with.any?
17
+ raise ArgumentError, 'embed one text at a time when embedding attachments' if text.is_a?(Array)
18
+
19
+ [media_embedding_payload(text, with, model:, dimensions:, task_type:, title:)]
20
+ else
21
+ [text].flatten.map do |t|
22
+ single_embedding_payload(t, model:, dimensions:, task_type:, title:)
23
+ end
24
+ end
25
+
26
+ Support::Utils.deep_merge({ requests: requests }, provider_options)
27
+ end
28
+
29
+ def supports_embedding_media?
30
+ true
31
+ end
32
+
33
+ def render_embedding(text, dimensions: nil, **options)
34
+ payload = render_embedding_payload(text, dimensions:, **options)
35
+ return payload if text.is_a?(Array) || !payload.key?(:requests)
36
+
37
+ payload.fetch(:requests).first
38
+ end
39
+ public :render_embedding
40
+
41
+ def parse_embedding_response(response, model:, text:)
42
+ vectors = response.body['embeddings']&.map { |e| e['values'] }
43
+ vectors = vectors.first if vectors&.length == 1 && !text.is_a?(Array)
44
+
45
+ Embedding.new(vectors:, model:)
46
+ end
47
+
48
+ private
49
+
50
+ def single_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil)
51
+ embedding_payload([{ text: text.to_s }], model:, dimensions:, task_type:, title:)
52
+ end
53
+
54
+ def media_embedding_payload(text, attachments, model:, dimensions:, task_type: nil, title: nil)
55
+ embedding_payload(Media.format_content(text, attachments), model:, dimensions:, task_type:, title:)
56
+ end
57
+
58
+ def embedding_payload(parts, model:, dimensions:, task_type:, title:)
59
+ {
60
+ model: "models/#{model}",
61
+ content: { parts: parts },
62
+ outputDimensionality: dimensions,
63
+ taskType: task_type,
64
+ title: title
65
+ }.compact
66
+ end
67
+ end
68
+ end
69
+ end
70
+ end
@@ -0,0 +1,32 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Protocols
5
+ class Gemini
6
+ module FileTranscription # :nodoc: all
7
+ def transcribe(audio_file, model:, language:, format: nil, speaker_names: nil,
8
+ speaker_references: nil, provider_options: {}, prompt: nil, temperature: nil)
9
+ raise_transcription_streaming_unsupported if block_given?
10
+ validate_transcription_request(format:, speaker_references:, temperature:)
11
+ attachments = Attachment.wrap(audio_file, config: @config)
12
+ unless attachments.one? && attachments.first.audio?
13
+ raise ArgumentError, 'Dedicated transcription requires exactly one audio file'
14
+ end
15
+
16
+ track_usage(:transcription) do
17
+ payload = render_transcription_payload(attachments.first, model:, language:, speaker_names:,
18
+ provider_options:, prompt:)
19
+ response = @connection.post(transcription_url(model), payload, usage: @usage_tracker)
20
+ parse_transcription_response(response, model:)
21
+ end
22
+ end
23
+
24
+ def validate_transcription_request(format:, speaker_references:, temperature:)
25
+ return unless format || speaker_references || temperature
26
+
27
+ raise ArgumentError, 'Dedicated transcription does not accept format, speaker references, or temperature'
28
+ end
29
+ end
30
+ end
31
+ end
32
+ end