ruby_llm 1.16.0 → 2.0.0.rc4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (480) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +108 -44
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/legacy_content_sql.rb +34 -0
  91. data/lib/generators/ruby_llm/upgrade/online_copy_migration/data.rb +341 -0
  92. data/lib/generators/ruby_llm/upgrade/online_copy_migration/journal.rb +171 -0
  93. data/lib/generators/ruby_llm/upgrade/online_copy_migration/verification.rb +197 -0
  94. data/lib/generators/ruby_llm/upgrade/online_copy_migration.rb +174 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +468 -0
  96. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
  97. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +247 -0
  98. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +662 -0
  99. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +251 -0
  100. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  101. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +188 -0
  102. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +359 -0
  103. data/lib/ruby_llm/accounting/usage.rb +254 -0
  104. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  105. data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
  106. data/lib/ruby_llm/active_record/batch.rb +97 -0
  107. data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
  108. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  109. data/lib/ruby_llm/active_record/model.rb +135 -0
  110. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  111. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  112. data/lib/ruby_llm/active_record/usage.rb +61 -0
  113. data/lib/ruby_llm/agent.rb +1066 -151
  114. data/lib/ruby_llm/aliases.json +291 -101
  115. data/lib/ruby_llm/attachment.rb +192 -48
  116. data/lib/ruby_llm/batch.rb +432 -0
  117. data/lib/ruby_llm/cached_content.rb +112 -0
  118. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  119. data/lib/ruby_llm/chat.rb +1131 -198
  120. data/lib/ruby_llm/chunk.rb +10 -0
  121. data/lib/ruby_llm/citation.rb +105 -0
  122. data/lib/ruby_llm/configuration.rb +261 -24
  123. data/lib/ruby_llm/context.rb +128 -6
  124. data/lib/ruby_llm/cost.rb +217 -80
  125. data/lib/ruby_llm/downloaded_file.rb +33 -0
  126. data/lib/ruby_llm/embedding.rb +121 -17
  127. data/lib/ruby_llm/embedding_request.rb +53 -0
  128. data/lib/ruby_llm/error.rb +159 -23
  129. data/lib/ruby_llm/fallback.rb +133 -0
  130. data/lib/ruby_llm/files/mime_type.rb +97 -0
  131. data/lib/ruby_llm/image.rb +153 -32
  132. data/lib/ruby_llm/message.rb +242 -54
  133. data/lib/ruby_llm/model/modalities.rb +17 -4
  134. data/lib/ruby_llm/model/pricing.rb +24 -5
  135. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  136. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  137. data/lib/ruby_llm/model.rb +244 -2
  138. data/lib/ruby_llm/models/aliases.rb +41 -0
  139. data/lib/ruby_llm/models/registry.rb +165 -0
  140. data/lib/ruby_llm/models/schema.rb +99 -0
  141. data/lib/ruby_llm/models.json +76038 -34173
  142. data/lib/ruby_llm/models.rb +477 -215
  143. data/lib/ruby_llm/moderation.rb +139 -26
  144. data/lib/ruby_llm/ocr.rb +112 -0
  145. data/lib/ruby_llm/prompt.rb +79 -0
  146. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  147. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  148. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  149. data/lib/ruby_llm/protocol.rb +662 -0
  150. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  151. data/lib/ruby_llm/protocols/anthropic/chat.rb +543 -0
  152. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  153. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  154. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  155. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  156. data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
  157. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  158. data/lib/ruby_llm/protocols/anthropic.rb +101 -0
  159. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  160. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +110 -0
  161. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  162. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  163. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  164. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  165. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  166. data/lib/ruby_llm/protocols/chat_completions/chat.rb +484 -0
  167. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  168. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  169. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  170. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  171. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  172. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  173. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +63 -0
  174. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  175. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  176. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  177. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  178. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  179. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  180. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  181. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  182. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  183. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  184. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  185. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  186. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  187. data/lib/ruby_llm/protocols/cohere/rerank.rb +59 -0
  188. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  189. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  190. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  191. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  192. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  193. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  194. data/lib/ruby_llm/protocols/converse/chat.rb +694 -0
  195. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  196. data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
  197. data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
  198. data/lib/ruby_llm/protocols/converse.rb +54 -0
  199. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  200. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  201. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  202. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  203. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  204. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  208. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  209. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  210. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  211. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  212. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  213. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  214. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  215. data/lib/ruby_llm/protocols/files.rb +119 -0
  216. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  217. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  218. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  219. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +91 -0
  220. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  221. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  222. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  223. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  224. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  225. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  226. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  227. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  228. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  229. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  230. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  231. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  232. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  233. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  234. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  235. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  236. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  237. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  238. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  239. data/lib/ruby_llm/protocols/interactions/tools.rb +53 -0
  240. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  241. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  242. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  243. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  244. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  245. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  246. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  247. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  248. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  249. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  250. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  251. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  252. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  253. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  254. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  255. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  256. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  257. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  258. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  259. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  260. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  261. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  262. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  263. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  264. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  265. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  266. data/lib/ruby_llm/protocols/responses/chat.rb +466 -0
  267. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  268. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  269. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  270. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  271. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  272. data/lib/ruby_llm/protocols/responses.rb +35 -0
  273. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  274. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  275. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  276. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  277. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  278. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  279. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  280. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  281. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  282. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  283. data/lib/ruby_llm/provider.rb +560 -128
  284. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  285. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  286. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  287. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  288. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  289. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  290. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  291. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  292. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  293. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  294. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  295. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  296. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  297. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  298. data/lib/ruby_llm/providers/azure.rb +77 -78
  299. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  300. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  301. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  302. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  303. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  304. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  305. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  306. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  307. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  308. data/lib/ruby_llm/providers/cohere.rb +31 -0
  309. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  310. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  311. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  312. data/lib/ruby_llm/providers/deepseek/responses.rb +67 -0
  313. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  314. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  315. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  316. data/lib/ruby_llm/providers/gemini.rb +15 -8
  317. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  318. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  319. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  320. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  321. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  322. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  323. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  324. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  325. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  326. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  327. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  328. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  329. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  330. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  331. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  332. data/lib/ruby_llm/providers/mistral/ocr.rb +51 -0
  333. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  334. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  335. data/lib/ruby_llm/providers/mistral.rb +16 -4
  336. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  337. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  338. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  339. data/lib/ruby_llm/providers/ollama.rb +9 -8
  340. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  341. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  342. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  343. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  344. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  345. data/lib/ruby_llm/providers/openai.rb +92 -11
  346. data/lib/ruby_llm/providers/openrouter/chat.rb +126 -108
  347. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  348. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  349. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  350. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  351. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  352. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  353. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  354. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  355. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  356. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  357. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  358. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  359. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  360. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  361. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  362. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  363. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  364. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  365. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  366. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  367. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  368. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  369. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  370. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  371. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  372. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  373. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  374. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  375. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  376. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  377. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  378. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  379. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  380. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  381. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  382. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  383. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  384. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  385. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  386. data/lib/ruby_llm/providers/xai.rb +15 -5
  387. data/lib/ruby_llm/railtie.rb +7 -16
  388. data/lib/ruby_llm/rerank.rb +105 -0
  389. data/lib/ruby_llm/research_job.rb +241 -0
  390. data/lib/ruby_llm/search_results.rb +68 -0
  391. data/lib/ruby_llm/server_tool_call.rb +73 -0
  392. data/lib/ruby_llm/speech.rb +159 -0
  393. data/lib/ruby_llm/speech_chunk.rb +33 -0
  394. data/lib/ruby_llm/support/deprecator.rb +22 -0
  395. data/lib/ruby_llm/support/inspectable.rb +49 -0
  396. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  397. data/lib/ruby_llm/support/utils.rb +147 -0
  398. data/lib/ruby_llm/thinking.rb +127 -20
  399. data/lib/ruby_llm/tokenization.rb +59 -0
  400. data/lib/ruby_llm/tokens.rb +103 -33
  401. data/lib/ruby_llm/tool.rb +266 -91
  402. data/lib/ruby_llm/tool_call.rb +36 -3
  403. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  404. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  405. data/lib/ruby_llm/transcription.rb +138 -13
  406. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  407. data/lib/ruby_llm/transport/connection.rb +193 -0
  408. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  409. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  410. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  411. data/lib/ruby_llm/uploaded_file.rb +144 -0
  412. data/lib/ruby_llm/version.rb +2 -1
  413. data/lib/ruby_llm/video.rb +136 -0
  414. data/lib/ruby_llm/video_job.rb +150 -0
  415. data/lib/ruby_llm/workflow.rb +91 -0
  416. data/lib/ruby_llm.rb +380 -6
  417. data/lib/tasks/ruby_llm.rake +21 -16
  418. data/skills/rubyllm/SKILL.md +81 -0
  419. data/skills/rubyllm/agents/openai.yaml +4 -0
  420. metadata +344 -97
  421. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  422. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  423. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  424. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  425. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  426. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  427. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  428. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  429. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  430. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  431. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  432. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  433. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  434. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  435. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  436. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  437. data/lib/ruby_llm/aliases.rb +0 -41
  438. data/lib/ruby_llm/connection.rb +0 -159
  439. data/lib/ruby_llm/content.rb +0 -91
  440. data/lib/ruby_llm/deprecator.rb +0 -24
  441. data/lib/ruby_llm/error_middleware.rb +0 -81
  442. data/lib/ruby_llm/instrumentation.rb +0 -36
  443. data/lib/ruby_llm/mime_type.rb +0 -96
  444. data/lib/ruby_llm/model/info.rb +0 -164
  445. data/lib/ruby_llm/model_registry.rb +0 -39
  446. data/lib/ruby_llm/models_schema.json +0 -171
  447. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  448. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  449. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  450. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  451. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  452. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  453. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  454. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  455. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  456. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  457. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  458. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  459. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  460. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  461. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  462. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  463. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  464. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  465. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  466. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  467. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  468. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  469. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  470. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  471. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  472. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  473. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  474. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  475. data/lib/ruby_llm/streaming.rb +0 -179
  476. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  477. data/lib/ruby_llm/utils.rb +0 -130
  478. data/lib/tasks/models.rake +0 -593
  479. data/lib/tasks/release.rake +0 -94
  480. data/lib/tasks/vcr.rake +0 -124
@@ -0,0 +1,694 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'json'
4
+
5
+ module RubyLLM
6
+ module Protocols
7
+ class Converse
8
+ # Chat methods for Bedrock Converse API.
9
+ module Chat
10
+ FINISH_REASONS = {
11
+ 'end_turn' => :stop, 'stop_sequence' => :stop, 'max_tokens' => :max_tokens,
12
+ 'model_context_window_exceeded' => :max_tokens, 'tool_use' => :tool_calls,
13
+ 'guardrail_intervened' => :content_filter, 'content_filtered' => :content_filter
14
+ }.freeze
15
+
16
+ BEDROCK_INLINE_DOCUMENT_LIMIT = 4_500_000
17
+ PROMPT_CACHE_OPTIONS = %i[ttl].freeze
18
+ COUNT_TOKENS_KEYS = %i[messages system toolConfig additionalModelRequestFields].freeze
19
+ RANGED_EFFORTS = %w[low medium high].freeze
20
+ MINIMUM_BUDGET_TOKENS = 1
21
+
22
+ module_function
23
+
24
+ def finish_reasons = FINISH_REASONS
25
+
26
+ def normalize_finish_reason(reason)
27
+ return nil if reason.nil?
28
+
29
+ finish_reasons.fetch(reason.to_s) { reason.to_s.to_sym }
30
+ end
31
+
32
+ def completion_url
33
+ "/model/#{escape_model_id(@model.id)}/converse"
34
+ end
35
+
36
+ # Application inference profile ARNs contain '/', but Bedrock expects the model id
37
+ # to remain one path segment.
38
+ def escape_model_id(model_id)
39
+ model_id.to_s.gsub('/', '%2F')
40
+ end
41
+
42
+ # rubocop:disable-next Lint/UnusedMethodArgument
43
+ def render_payload(messages, tools:, temperature:, model:, stream: false, max_output_tokens: nil,
44
+ schema: nil, thinking: nil, citations: false, caching: nil, tool_prefs: nil)
45
+ tool_prefs ||= {}
46
+ @used_document_names = {}
47
+ system_messages, chat_messages = messages.partition { |msg| msg.role == :system }
48
+ prompt_cache_options(caching)
49
+ automatic_cache_target = automatic_cache_target(system_messages, chat_messages, caching)
50
+ payload = {
51
+ messages: format_messages(chat_messages, caching:, automatic_cache_target:, citations:)
52
+ }
53
+
54
+ system_blocks = format_system(system_messages, caching:, automatic_cache_target:)
55
+ payload[:system] = system_blocks unless system_blocks.empty?
56
+
57
+ payload[:inferenceConfig] = format_inference_config(model, temperature, max_output_tokens)
58
+
59
+ tool_config = format_tool_config(tools, tool_prefs)
60
+ payload[:toolConfig] = tool_config if tool_config
61
+
62
+ additional_fields = format_additional_model_request_fields(thinking, model, max_output_tokens)
63
+ payload[:additionalModelRequestFields] = additional_fields if additional_fields
64
+
65
+ output_config = build_output_config(schema)
66
+ payload[:outputConfig] = output_config if output_config
67
+
68
+ payload
69
+ end
70
+
71
+ def count_tokens_url
72
+ "/model/#{escape_model_id(@model.id)}/count-tokens"
73
+ end
74
+
75
+ def render_count_tokens_payload(messages, tools:, model:, tool_prefs: nil, thinking: nil, schema: nil,
76
+ citations: false, caching: nil)
77
+ request = render_payload(
78
+ messages,
79
+ tools: tools,
80
+ tool_prefs: tool_prefs,
81
+ temperature: nil,
82
+ model: model,
83
+ schema: schema,
84
+ thinking: thinking,
85
+ citations: citations,
86
+ caching: caching
87
+ ).slice(*COUNT_TOKENS_KEYS)
88
+
89
+ { input: { converse: request } }
90
+ end
91
+
92
+ def parse_count_tokens_response(response)
93
+ response.body['inputTokens']
94
+ end
95
+
96
+ def supports_provider_file_references?
97
+ true
98
+ end
99
+
100
+ def default_large_file_upload_threshold
101
+ BEDROCK_INLINE_DOCUMENT_LIMIT
102
+ end
103
+
104
+ def provider_file_attachable?(attachment)
105
+ (attachment.pdf? || attachment.document? || attachment.text?) &&
106
+ Media.supported_document_format?(attachment)
107
+ end
108
+
109
+ def parse_completion_body(data, raw:)
110
+ content_blocks = data.dig('output', 'message', 'content') || []
111
+ usage = data['usage'] || {}
112
+ thinking_text, thinking_signature = parse_thinking(content_blocks)
113
+ text_content, citations = extract_text_and_citations(content_blocks)
114
+
115
+ Message.new(
116
+ role: :assistant,
117
+ content: text_content,
118
+ citations: citations,
119
+ thinking: Thinking.build(text: thinking_text, signature: thinking_signature),
120
+ raw_reasoning: parse_thinking_blocks(content_blocks),
121
+ tool_calls: parse_tool_calls(content_blocks),
122
+ server_tool_calls: extract_server_tool_calls(content_blocks),
123
+ input_tokens: input_tokens(usage),
124
+ output_tokens: usage['outputTokens'],
125
+ cache_read_tokens: usage['cacheReadInputTokens'],
126
+ cache_write_tokens: usage['cacheWriteInputTokens'],
127
+ thinking_tokens: reasoning_tokens(usage),
128
+ finish_reason: normalize_finish_reason(data['stopReason']),
129
+ model: data['modelId'] || @model&.id,
130
+ raw: raw
131
+ )
132
+ end
133
+
134
+ def input_tokens(usage)
135
+ # AWS Bedrock reports inputTokens as already non-cached; cacheReadInputTokens and
136
+ # cacheWriteInputTokens are separate buckets, not folded into inputTokens. Subtracting
137
+ # them (as inclusive providers require) understates input and floors to zero on cache
138
+ # hits. See https://docs.aws.amazon.com/bedrock/latest/userguide/prompt-caching.html
139
+ usage['inputTokens']
140
+ end
141
+
142
+ def reasoning_tokens(usage)
143
+ usage['reasoningTokens'] || usage.dig('outputTokensDetails', 'reasoningTokens')
144
+ end
145
+
146
+ def format_messages(messages, caching: nil, automatic_cache_target: nil, citations: false)
147
+ rendered = []
148
+ tool_result_blocks = []
149
+
150
+ messages.each do |msg|
151
+ if msg.tool_result?
152
+ tool_result_blocks << format_tool_result_block(msg)
153
+ next
154
+ end
155
+
156
+ unless tool_result_blocks.empty?
157
+ rendered << { role: 'user', content: tool_result_blocks }
158
+ tool_result_blocks = []
159
+ end
160
+
161
+ message = format_non_tool_message(msg, caching:, automatic_cache_target:, citations:)
162
+ rendered << message if message
163
+ end
164
+
165
+ rendered << { role: 'user', content: tool_result_blocks } unless tool_result_blocks.empty?
166
+ rendered
167
+ end
168
+
169
+ def format_non_tool_message(msg, caching: nil, automatic_cache_target: nil, citations: false)
170
+ content = format_message_content(msg, caching:, automatic_cache_target:, citations:)
171
+ return nil if content.empty?
172
+
173
+ {
174
+ role: format_role(msg.role),
175
+ content: content
176
+ }
177
+ end
178
+
179
+ def format_message_content(msg, caching: nil, automatic_cache_target: nil, citations: false)
180
+ blocks = format_structured_message_content(msg, citations:)
181
+
182
+ if msg.tool_call?
183
+ msg.tool_calls.each_value do |tool_call|
184
+ blocks << {
185
+ toolUse: {
186
+ toolUseId: tool_call.id,
187
+ name: tool_call.name,
188
+ input: tool_call.arguments
189
+ }
190
+ }
191
+ end
192
+ end
193
+ blocks << converse_cache_block_for(caching) if cache_boundary?(msg, automatic_cache_target, caching:)
194
+
195
+ blocks
196
+ end
197
+
198
+ def format_structured_message_content(msg, citations: false)
199
+ blocks = msg.role == :assistant ? format_thinking_blocks(msg) : []
200
+
201
+ blocks.concat(
202
+ Media.format_content(msg.content, msg.attachments, used_document_names: @used_document_names, citations:)
203
+ )
204
+
205
+ blocks
206
+ end
207
+
208
+ def format_thinking_blocks(msg)
209
+ blocks = msg.raw_reasoning['converse'] if msg.raw_reasoning.is_a?(Hash)
210
+ return Support::Utils.deep_dup(blocks) if blocks
211
+
212
+ [format_thinking_block(msg.thinking)].compact
213
+ end
214
+
215
+ def format_tool_result_block(msg)
216
+ {
217
+ toolResult: {
218
+ toolUseId: msg.tool_call_id,
219
+ content: format_tool_result_content(msg)
220
+ }
221
+ }
222
+ end
223
+
224
+ def format_tool_result_content(msg)
225
+ search_results = RubyLLM::SearchResults.from_content(msg.content)
226
+ return search_results.results.map { |result| search_result_block(result) } if search_results
227
+
228
+ blocks = Media.format_content(msg.content, msg.attachments, used_document_names: @used_document_names)
229
+ blocks.empty? ? [text_tool_result_block(nil)] : blocks
230
+ end
231
+
232
+ def search_result_block(result)
233
+ {
234
+ searchResult: {
235
+ source: result[:url] || result[:title],
236
+ title: result[:title],
237
+ content: [{ text: result[:text] }],
238
+ citations: { enabled: true }
239
+ }
240
+ }
241
+ end
242
+
243
+ def text_tool_result_block(text)
244
+ text = text.to_s
245
+ text = '(no output)' if text.empty?
246
+ { text: text }
247
+ end
248
+
249
+ def format_role(role)
250
+ case role
251
+ when :assistant then 'assistant'
252
+ else 'user'
253
+ end
254
+ end
255
+
256
+ def format_system(messages, caching: nil, automatic_cache_target: nil)
257
+ messages.flat_map do |msg|
258
+ blocks = Media.format_content(msg.content, msg.attachments, used_document_names: @used_document_names)
259
+ if cache_boundary?(msg, automatic_cache_target, caching:)
260
+ blocks + [converse_cache_block_for(caching)]
261
+ else
262
+ blocks
263
+ end
264
+ end
265
+ end
266
+
267
+ def automatic_cache_target(system_messages, chat_messages, caching)
268
+ return unless caching
269
+
270
+ (chat_messages.reverse + system_messages.reverse).find { |msg| cacheable_message?(msg) }
271
+ end
272
+
273
+ def cacheable_message?(message)
274
+ !message.tool_result?
275
+ end
276
+
277
+ def cache_boundary?(message, automatic_cache_target, caching: nil)
278
+ caching != false && (message.cache_until_here? || message.equal?(automatic_cache_target))
279
+ end
280
+
281
+ def converse_cache_block_for(caching)
282
+ options = prompt_cache_options(caching)
283
+ point = { type: 'default' }
284
+ point[:ttl] = options[:ttl] if options[:ttl]
285
+ { cachePoint: point }
286
+ end
287
+
288
+ def prompt_cache_options(caching)
289
+ return {} unless caching
290
+
291
+ options = caching.to_h.transform_keys(&:to_sym)
292
+ unsupported = options.keys - PROMPT_CACHE_OPTIONS
293
+ return options if unsupported.empty?
294
+
295
+ raise ArgumentError,
296
+ "Bedrock Converse prompt caching accepts :ttl, got #{format_cache_option_keys(unsupported)}"
297
+ end
298
+
299
+ def format_cache_option_keys(keys)
300
+ keys.map { |key| ":#{key}" }.join(', ')
301
+ end
302
+
303
+ def format_inference_config(_model, temperature, max_output_tokens = nil)
304
+ config = {}
305
+ config[:temperature] = temperature unless temperature.nil?
306
+ config[:maxTokens] = max_output_tokens unless max_output_tokens.nil?
307
+ config
308
+ end
309
+
310
+ def format_tool_config(tools, tool_prefs)
311
+ return nil if tools.empty?
312
+
313
+ config = {
314
+ tools: tools.values.map { |tool| format_tool(tool) }
315
+ }
316
+
317
+ return config if tool_prefs.nil? || tool_prefs[:choice].nil?
318
+
319
+ tool_choice = format_tool_choice(tool_prefs[:choice])
320
+ config[:toolChoice] = tool_choice if tool_choice
321
+ config
322
+ end
323
+
324
+ def format_tool_choice(choice)
325
+ case choice
326
+ when :auto
327
+ { auto: {} }
328
+ when :none
329
+ nil
330
+ when :required
331
+ { any: {} }
332
+ else
333
+ { tool: { name: choice.to_s } }
334
+ end
335
+ end
336
+
337
+ def format_tool(tool)
338
+ input_schema = tool.parameters_schema ||
339
+ RubyLLM::Tool::SchemaDefinition.from_parameters(tool.declared_parameters)&.json_schema
340
+
341
+ tool_spec = {
342
+ toolSpec: {
343
+ name: tool.name,
344
+ description: tool.description,
345
+ inputSchema: {
346
+ json: input_schema || default_input_schema
347
+ }
348
+ }
349
+ }
350
+
351
+ return tool_spec if tool.provider_options.empty?
352
+
353
+ RubyLLM::Support::Utils.deep_merge(tool_spec, tool.provider_options)
354
+ end
355
+
356
+ def format_additional_model_request_fields(thinking, model, max_output_tokens = nil)
357
+ fields = {}
358
+
359
+ reasoning_fields = format_reasoning_fields(thinking, model, max_output_tokens)
360
+ fields = RubyLLM::Support::Utils.deep_merge(fields, reasoning_fields) if reasoning_fields
361
+
362
+ fields.empty? ? nil : fields
363
+ end
364
+
365
+ def build_output_config(schema)
366
+ return nil unless schema
367
+
368
+ cleaned = RubyLLM::Support::Utils.deep_dup(schema[:schema])
369
+ cleaned.delete(:strict)
370
+ cleaned.delete('strict')
371
+
372
+ {
373
+ textFormat: {
374
+ type: 'json_schema',
375
+ structure: {
376
+ jsonSchema: {
377
+ schema: JSON.generate(cleaned),
378
+ name: schema[:name]
379
+ }
380
+ }
381
+ }
382
+ }
383
+ end
384
+
385
+ NOVA_DEFAULT_REASONING_EFFORT = 'low'
386
+
387
+ def format_reasoning_fields(thinking, model, max_output_tokens = nil)
388
+ return nil unless thinking&.enabled?
389
+ return format_nova_reasoning_fields(thinking, model) if nova_model?(model)
390
+ return { reasoning_config: { type: 'disabled' } } if thinking.enabled == false
391
+
392
+ effort = thinking.effort.to_s
393
+ budget = reasoning_budget(thinking, effort, model, max_output_tokens)
394
+ return { reasoning_config: { type: 'enabled', budget_tokens: budget } } if budget
395
+ return nil if effort.empty? || effort == 'none'
396
+
397
+ { reasoning_effort: effort }
398
+ end
399
+
400
+ def nova_model?(model)
401
+ foundation_model_id(model&.id).start_with?('amazon.nova')
402
+ end
403
+
404
+ # Nova 2 takes reasoningConfig with an effort level instead of a token budget,
405
+ # and rejects an enabled reasoningConfig that names no effort.
406
+ def format_nova_reasoning_fields(thinking, model)
407
+ if thinking.budget
408
+ raise ArgumentError,
409
+ "#{model&.id} takes a reasoning effort, not a token budget of #{thinking.budget}. " \
410
+ 'Pass with_thinking(effort:) instead.'
411
+ end
412
+
413
+ return { reasoningConfig: { type: 'disabled' } } if thinking.enabled == false
414
+
415
+ effort = thinking.effort.to_s
416
+ return nil if effort == 'none'
417
+
418
+ effort = NOVA_DEFAULT_REASONING_EFFORT if effort.empty?
419
+ { reasoningConfig: { type: 'enabled', maxReasoningEffort: effort } }
420
+ end
421
+
422
+ def reasoning_budget(thinking, effort, model, max_output_tokens)
423
+ return thinking.budget if thinking.budget.is_a?(Integer)
424
+ return nil if effort.empty? || effort == 'none'
425
+
426
+ schema = reasoning_budget_schema(model)
427
+ schema && effort_budget_tokens(effort, schema, model, max_output_tokens)
428
+ end
429
+
430
+ # Bedrock only publishes Converse metadata for some regional entries, so use the
431
+ # schema from another entry for the same foundation model when needed.
432
+ def reasoning_budget_schema(model)
433
+ schema = budget_tokens_schema(model)
434
+ return schema if schema
435
+ return unless model
436
+
437
+ foundation_id = foundation_model_id(model.id)
438
+ RubyLLM.models.all.each do |candidate|
439
+ next unless candidate.provider == 'bedrock' && candidate.id != model.id
440
+ next unless foundation_model_id(candidate.id) == foundation_id
441
+
442
+ return schema if (schema = budget_tokens_schema(candidate))
443
+ end
444
+
445
+ nil
446
+ end
447
+
448
+ def budget_tokens_schema(model)
449
+ metadata = RubyLLM::Support::Utils.deep_symbolize_keys(model&.metadata || {})
450
+ raw_schema = metadata.dig(:converse, :additionalRequestFieldsSchema)
451
+ return unless raw_schema.is_a?(String)
452
+
453
+ schema = JSON.parse(raw_schema, symbolize_names: true)
454
+ budget = schema.is_a?(Hash) ? schema.dig(:reasoningConfig, :budgetTokens) : nil
455
+ budget if budget.is_a?(Hash)
456
+ rescue JSON::ParserError
457
+ nil
458
+ end
459
+
460
+ # Inference profile and foundation model ARNs name the model in their last segment.
461
+ def foundation_model_id(model_id)
462
+ prefixes = Converse::REGION_PREFIXES.join('|')
463
+ model_id.to_s.rpartition('/').last.sub(/\A(?:#{prefixes})\./, '')
464
+ end
465
+
466
+ # Models that take a budget reject reasoning_effort, so effort has to become a budget.
467
+ # Bedrock names the levels of an enumerated budget after the efforts they stand for;
468
+ # otherwise the effort spans the range the schema allows.
469
+ def effort_budget_tokens(effort, schema, model, max_output_tokens)
470
+ budget = enumerated_budget(effort, schema) || ranged_budget(effort, schema)
471
+ return nil unless budget
472
+
473
+ minimum = schema[:minimum].is_a?(Integer) ? schema[:minimum] : MINIMUM_BUDGET_TOKENS
474
+ return [budget, minimum].max unless max_output_tokens
475
+
476
+ budget.clamp(minimum, budget_ceiling(model, minimum, max_output_tokens))
477
+ end
478
+
479
+ # Bedrock rejects a budget that leaves no room for the answer.
480
+ def budget_ceiling(model, minimum, max_output_tokens)
481
+ ceiling = max_output_tokens - 1
482
+ return ceiling if minimum <= ceiling
483
+
484
+ raise ArgumentError, "#{model&.id} reasons on a budget of at least #{minimum} tokens, and " \
485
+ "max_output_tokens: #{max_output_tokens} leaves room for #{ceiling}. " \
486
+ "Raise max_output_tokens above #{minimum} or turn thinking off."
487
+ end
488
+
489
+ def enumerated_budget(effort, schema)
490
+ levels = schema[:enum]
491
+ level = levels.is_a?(Hash) ? levels[effort.to_sym] : nil
492
+ level if level.is_a?(Integer)
493
+ end
494
+
495
+ def ranged_budget(effort, schema)
496
+ minimum = schema[:minimum]
497
+ maximum = schema[:maximum]
498
+ return nil unless minimum.is_a?(Integer) && maximum.is_a?(Integer)
499
+
500
+ case effort
501
+ when 'low' then minimum
502
+ when 'medium' then minimum + ((maximum - minimum) / 2)
503
+ when 'high' then maximum
504
+ else raise ArgumentError, unknown_effort_message(effort, schema)
505
+ end
506
+ end
507
+
508
+ def unknown_effort_message(effort, schema)
509
+ levels = schema[:enum].is_a?(Hash) ? schema[:enum].keys.map(&:to_s) : []
510
+ levels |= RANGED_EFFORTS
511
+ "Bedrock has no reasoning budget for effort #{effort.inspect}. " \
512
+ "Use #{levels.join(', ')}, or pass an explicit budget."
513
+ end
514
+
515
+ def format_thinking_block(thinking)
516
+ return nil unless thinking
517
+
518
+ if thinking.text
519
+ {
520
+ reasoningContent: {
521
+ reasoningText: {
522
+ text: thinking.text,
523
+ signature: thinking.signature
524
+ }.compact
525
+ }
526
+ }
527
+ elsif thinking.signature
528
+ {
529
+ reasoningContent: {
530
+ redactedContent: thinking.signature
531
+ }
532
+ }
533
+ end
534
+ end
535
+
536
+ def extract_text_and_citations(content_blocks)
537
+ text = +''
538
+ citations = []
539
+
540
+ content_blocks.each do |block|
541
+ if block['text'].is_a?(String)
542
+ text << block['text']
543
+ elsif block['citationsContent'].is_a?(Hash)
544
+ append_citations_content(block['citationsContent'], text, citations)
545
+ end
546
+ end
547
+
548
+ [text.empty? ? nil : text, citations]
549
+ end
550
+
551
+ # A citationsContent block replaces the text block for a cited span:
552
+ # the generated text lives in its content member, and each citation
553
+ # points back at the source document or search result.
554
+ def append_citations_content(citations_content, text, citations)
555
+ block_text = joined_text(citations_content['content'])
556
+ span = {}
557
+ unless block_text.empty?
558
+ span = { text: block_text, start_index: text.length, end_index: text.length + block_text.length }
559
+ end
560
+
561
+ Array(citations_content['citations']).each do |citation|
562
+ citations << parse_citation(citation, **span)
563
+ end
564
+ text << block_text
565
+ end
566
+
567
+ def parse_citation(data, text: nil, start_index: nil, end_index: nil)
568
+ location = data['location'] || {}
569
+ page = location['documentPage'] || {}
570
+ end_page = page['end']
571
+ cited_text = joined_text(data['sourceContent'])
572
+
573
+ Citation.new(
574
+ url: citation_url(data, location),
575
+ title: data['title'],
576
+ cited_text: cited_text.empty? ? nil : cited_text,
577
+ text: text,
578
+ start_index: start_index,
579
+ end_index: end_index,
580
+ source_index: citation_source_index(location),
581
+ start_page: page['start'],
582
+ end_page: end_page && (end_page - 1)
583
+ )
584
+ end
585
+
586
+ def joined_text(parts)
587
+ Array(parts).filter_map { |part| part['text'] if part.is_a?(Hash) }.join
588
+ end
589
+
590
+ # Search result citations carry the developer-provided source string.
591
+ def citation_url(data, location)
592
+ url = location.dig('web', 'url') || data['source']
593
+ url if url&.match?(%r{\Ahttps?://}i)
594
+ end
595
+
596
+ def citation_source_index(location)
597
+ %w[documentChar documentPage documentChunk].each do |key|
598
+ index = location.dig(key, 'documentIndex')
599
+ return index if index
600
+ end
601
+
602
+ location.dig('searchResultLocation', 'searchResultIndex')
603
+ end
604
+
605
+ def parse_thinking(content_blocks)
606
+ text = nil
607
+ signature = nil
608
+
609
+ content_blocks.each do |block|
610
+ chunk_text, chunk_signature = parse_reasoning_content_block(block)
611
+ if chunk_text
612
+ text ||= +''
613
+ text << chunk_text
614
+ end
615
+ signature ||= chunk_signature
616
+ end
617
+
618
+ [text, signature]
619
+ end
620
+
621
+ def parse_thinking_blocks(content_blocks)
622
+ blocks = content_blocks.select { |block| block['reasoningContent'].is_a?(Hash) }
623
+ { 'converse' => blocks } unless blocks.empty?
624
+ end
625
+
626
+ def parse_reasoning_content_block(block)
627
+ reasoning_content = block['reasoningContent']
628
+ return [nil, nil] unless reasoning_content.is_a?(Hash)
629
+
630
+ reasoning_text = reasoning_content['reasoningText'] || {}
631
+ text = reasoning_text['text'].is_a?(String) ? reasoning_text['text'] : nil
632
+ signature = reasoning_text['signature'] if reasoning_text['signature'].is_a?(String)
633
+ signature ||= reasoning_content['redactedContent'] if reasoning_content['redactedContent'].is_a?(String)
634
+ [text, signature]
635
+ end
636
+
637
+ def parse_tool_calls(content_blocks)
638
+ tool_calls = {}
639
+
640
+ content_blocks.each do |block|
641
+ tool_use = block['toolUse']
642
+ next unless tool_use
643
+ next if server_tool_use?(tool_use)
644
+
645
+ tool_call_id = tool_use['toolUseId']
646
+ tool_calls[tool_call_id] = ToolCall.new(
647
+ id: tool_call_id,
648
+ name: tool_use['name'],
649
+ arguments: tool_use['input'] || {}
650
+ )
651
+ end
652
+
653
+ tool_calls.empty? ? nil : tool_calls
654
+ end
655
+
656
+ # Provider-executed tool steps come back with a distinguishing type,
657
+ # such as server_tool_use; function calls carry no type or the plain
658
+ # tool_use type.
659
+ def server_tool_use?(tool_use)
660
+ type = tool_use['type']
661
+ !type.nil? && type != 'tool_use'
662
+ end
663
+
664
+ def server_tool_result?(tool_result)
665
+ type = tool_result['type']
666
+ !type.nil? && type != 'tool_result'
667
+ end
668
+
669
+ def extract_server_tool_calls(content_blocks)
670
+ content_blocks.filter_map do |block|
671
+ tool_use = block['toolUse']
672
+ tool_result = block['toolResult']
673
+
674
+ if tool_use && server_tool_use?(tool_use)
675
+ ServerToolCall.new(type: tool_use['type'], name: tool_use['name'], id: tool_use['toolUseId'],
676
+ input: tool_use['input'], raw: block)
677
+ elsif tool_result && server_tool_result?(tool_result)
678
+ ServerToolCall.new(type: tool_result['type'], id: tool_result['toolUseId'],
679
+ result: tool_result['content'], raw: block)
680
+ end
681
+ end
682
+ end
683
+
684
+ def default_input_schema
685
+ {
686
+ 'type' => 'object',
687
+ 'properties' => {},
688
+ 'required' => []
689
+ }
690
+ end
691
+ end
692
+ end
693
+ end
694
+ end