ruby_llm 1.16.0 → 2.0.0.rc2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (475) hide show
  1. checksums.yaml +4 -4
  2. data/.rdoc_options +25 -0
  3. data/README.md +85 -32
  4. data/exe/ruby_llm +8 -0
  5. data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
  6. data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
  7. data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
  8. data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
  9. data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
  10. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
  11. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
  12. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
  13. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
  14. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
  15. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
  16. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
  17. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
  18. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
  19. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
  20. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
  21. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
  22. data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
  23. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
  24. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
  25. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
  26. data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
  27. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
  28. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
  29. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
  30. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
  31. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
  32. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
  33. data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
  34. data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
  35. data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
  36. data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
  37. data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
  38. data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
  39. data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
  40. data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
  41. data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
  42. data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
  43. data/lib/generators/ruby_llm/provider/cli.rb +175 -0
  44. data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
  45. data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
  46. data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
  47. data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
  48. data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
  49. data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
  50. data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
  51. data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
  52. data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
  53. data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
  54. data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
  55. data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
  56. data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
  57. data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
  58. data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
  59. data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
  60. data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
  61. data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
  62. data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
  63. data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
  64. data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
  65. data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
  66. data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
  67. data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
  68. data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
  69. data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
  70. data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
  71. data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
  72. data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
  73. data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
  74. data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
  75. data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
  76. data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
  77. data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
  78. data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
  79. data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
  80. data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
  81. data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
  82. data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
  83. data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
  84. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
  85. data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
  86. data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
  87. data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
  88. data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
  89. data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
  90. data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
  91. data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
  92. data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
  93. data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
  94. data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
  95. data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
  96. data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
  97. data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
  98. data/lib/ruby_llm/accounting/usage.rb +245 -0
  99. data/lib/ruby_llm/active_record/acts_as.rb +94 -111
  100. data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
  101. data/lib/ruby_llm/active_record/batch.rb +97 -0
  102. data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
  103. data/lib/ruby_llm/active_record/message_methods.rb +113 -136
  104. data/lib/ruby_llm/active_record/model.rb +135 -0
  105. data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
  106. data/lib/ruby_llm/active_record/tool_call.rb +33 -0
  107. data/lib/ruby_llm/active_record/usage.rb +61 -0
  108. data/lib/ruby_llm/agent.rb +1065 -151
  109. data/lib/ruby_llm/aliases.json +268 -100
  110. data/lib/ruby_llm/attachment.rb +187 -48
  111. data/lib/ruby_llm/batch.rb +432 -0
  112. data/lib/ruby_llm/cached_content.rb +112 -0
  113. data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
  114. data/lib/ruby_llm/chat.rb +1127 -198
  115. data/lib/ruby_llm/chunk.rb +10 -0
  116. data/lib/ruby_llm/citation.rb +105 -0
  117. data/lib/ruby_llm/configuration.rb +261 -24
  118. data/lib/ruby_llm/context.rb +128 -6
  119. data/lib/ruby_llm/cost.rb +217 -80
  120. data/lib/ruby_llm/downloaded_file.rb +33 -0
  121. data/lib/ruby_llm/embedding.rb +121 -17
  122. data/lib/ruby_llm/embedding_request.rb +53 -0
  123. data/lib/ruby_llm/error.rb +159 -23
  124. data/lib/ruby_llm/fallback.rb +133 -0
  125. data/lib/ruby_llm/files/mime_type.rb +97 -0
  126. data/lib/ruby_llm/image.rb +153 -32
  127. data/lib/ruby_llm/message.rb +233 -54
  128. data/lib/ruby_llm/model/modalities.rb +17 -4
  129. data/lib/ruby_llm/model/pricing.rb +24 -5
  130. data/lib/ruby_llm/model/pricing_category.rb +103 -14
  131. data/lib/ruby_llm/model/pricing_tier.rb +55 -15
  132. data/lib/ruby_llm/model.rb +244 -2
  133. data/lib/ruby_llm/models/aliases.rb +41 -0
  134. data/lib/ruby_llm/models/registry.rb +165 -0
  135. data/lib/ruby_llm/models/schema.rb +99 -0
  136. data/lib/ruby_llm/models.json +73261 -33733
  137. data/lib/ruby_llm/models.rb +477 -215
  138. data/lib/ruby_llm/moderation.rb +139 -26
  139. data/lib/ruby_llm/ocr.rb +112 -0
  140. data/lib/ruby_llm/prompt.rb +79 -0
  141. data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
  142. data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
  143. data/lib/ruby_llm/protocol/streaming.rb +230 -0
  144. data/lib/ruby_llm/protocol.rb +662 -0
  145. data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
  146. data/lib/ruby_llm/protocols/anthropic/chat.rb +548 -0
  147. data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
  148. data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
  149. data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
  150. data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
  151. data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
  152. data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
  153. data/lib/ruby_llm/protocols/anthropic.rb +101 -0
  154. data/lib/ruby_llm/protocols/azure/files.rb +16 -0
  155. data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
  156. data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
  157. data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
  158. data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
  159. data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
  160. data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
  161. data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
  162. data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
  163. data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
  164. data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
  165. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
  166. data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
  167. data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
  168. data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
  169. data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
  170. data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
  171. data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
  172. data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
  173. data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
  174. data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
  175. data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
  176. data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
  177. data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
  178. data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
  179. data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
  180. data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
  181. data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
  182. data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
  183. data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
  184. data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
  185. data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
  186. data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
  187. data/lib/ruby_llm/protocols/cohere.rb +21 -0
  188. data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
  189. data/lib/ruby_llm/protocols/converse/chat.rb +695 -0
  190. data/lib/ruby_llm/protocols/converse/media.rb +178 -0
  191. data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
  192. data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
  193. data/lib/ruby_llm/protocols/converse.rb +54 -0
  194. data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
  195. data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
  196. data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
  197. data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
  198. data/lib/ruby_llm/protocols/deepgram.rb +19 -0
  199. data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
  200. data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
  201. data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
  202. data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
  203. data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
  204. data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
  205. data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
  206. data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
  207. data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
  208. data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
  209. data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
  210. data/lib/ruby_llm/protocols/files.rb +119 -0
  211. data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
  212. data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
  213. data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
  214. data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
  215. data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
  216. data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
  217. data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
  218. data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
  219. data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
  220. data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
  221. data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
  222. data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
  223. data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
  224. data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
  225. data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
  226. data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
  227. data/lib/ruby_llm/protocols/gemini.rb +35 -0
  228. data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
  229. data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
  230. data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
  231. data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
  232. data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
  233. data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
  234. data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
  235. data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
  236. data/lib/ruby_llm/protocols/interactions.rb +29 -0
  237. data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
  238. data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
  239. data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
  240. data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
  241. data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
  242. data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
  243. data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
  244. data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
  245. data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
  246. data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
  247. data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
  248. data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
  249. data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
  250. data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
  251. data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
  252. data/lib/ruby_llm/protocols/openai/files.rb +42 -0
  253. data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
  254. data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
  255. data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
  256. data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
  257. data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
  258. data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
  259. data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
  260. data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
  261. data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
  262. data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
  263. data/lib/ruby_llm/protocols/responses/media.rb +61 -0
  264. data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
  265. data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
  266. data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
  267. data/lib/ruby_llm/protocols/responses.rb +35 -0
  268. data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
  269. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
  270. data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
  271. data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
  272. data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
  273. data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
  274. data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
  275. data/lib/ruby_llm/protocols/xai/files.rb +30 -0
  276. data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
  277. data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
  278. data/lib/ruby_llm/provider.rb +560 -128
  279. data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
  280. data/lib/ruby_llm/providers/anthropic.rb +4 -6
  281. data/lib/ruby_llm/providers/azure/audio.rb +18 -0
  282. data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
  283. data/lib/ruby_llm/providers/azure/chat.rb +2 -9
  284. data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
  285. data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
  286. data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
  287. data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
  288. data/lib/ruby_llm/providers/azure/images.rb +22 -0
  289. data/lib/ruby_llm/providers/azure/media.rb +5 -14
  290. data/lib/ruby_llm/providers/azure/models.rb +35 -0
  291. data/lib/ruby_llm/providers/azure/responses.rb +26 -0
  292. data/lib/ruby_llm/providers/azure/videos.rb +64 -0
  293. data/lib/ruby_llm/providers/azure.rb +77 -78
  294. data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
  295. data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
  296. data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
  297. data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
  298. data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
  299. data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
  300. data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
  301. data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
  302. data/lib/ruby_llm/providers/bedrock.rb +216 -45
  303. data/lib/ruby_llm/providers/cohere.rb +31 -0
  304. data/lib/ruby_llm/providers/deepgram.rb +37 -0
  305. data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
  306. data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
  307. data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
  308. data/lib/ruby_llm/providers/deepseek.rb +9 -2
  309. data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
  310. data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
  311. data/lib/ruby_llm/providers/gemini.rb +15 -8
  312. data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
  313. data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
  314. data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
  315. data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
  316. data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
  317. data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
  318. data/lib/ruby_llm/providers/gpustack.rb +35 -11
  319. data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
  320. data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
  321. data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
  322. data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
  323. data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
  324. data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
  325. data/lib/ruby_llm/providers/mistral/media.rb +6 -18
  326. data/lib/ruby_llm/providers/mistral/models.rb +57 -23
  327. data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
  328. data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
  329. data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
  330. data/lib/ruby_llm/providers/mistral.rb +16 -4
  331. data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
  332. data/lib/ruby_llm/providers/ollama/media.rb +6 -15
  333. data/lib/ruby_llm/providers/ollama/models.rb +50 -9
  334. data/lib/ruby_llm/providers/ollama.rb +9 -8
  335. data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
  336. data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
  337. data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
  338. data/lib/ruby_llm/providers/openai/models.rb +23 -23
  339. data/lib/ruby_llm/providers/openai/responses.rb +13 -0
  340. data/lib/ruby_llm/providers/openai.rb +92 -11
  341. data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
  342. data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
  343. data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
  344. data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
  345. data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
  346. data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
  347. data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
  348. data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
  349. data/lib/ruby_llm/providers/openrouter.rb +78 -20
  350. data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
  351. data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
  352. data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
  353. data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
  354. data/lib/ruby_llm/providers/perplexity.rb +28 -20
  355. data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
  356. data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
  357. data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
  358. data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
  359. data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
  360. data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
  361. data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
  362. data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
  363. data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
  364. data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
  365. data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
  366. data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
  367. data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
  368. data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
  369. data/lib/ruby_llm/providers/vertexai.rb +160 -17
  370. data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
  371. data/lib/ruby_llm/providers/xai/chat.rb +3 -2
  372. data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
  373. data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
  374. data/lib/ruby_llm/providers/xai/images.rb +91 -0
  375. data/lib/ruby_llm/providers/xai/models.rb +32 -36
  376. data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
  377. data/lib/ruby_llm/providers/xai/responses.rb +52 -0
  378. data/lib/ruby_llm/providers/xai/speech.rb +45 -0
  379. data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
  380. data/lib/ruby_llm/providers/xai/videos.rb +87 -0
  381. data/lib/ruby_llm/providers/xai.rb +15 -5
  382. data/lib/ruby_llm/railtie.rb +7 -16
  383. data/lib/ruby_llm/rerank.rb +105 -0
  384. data/lib/ruby_llm/research_job.rb +241 -0
  385. data/lib/ruby_llm/search_results.rb +68 -0
  386. data/lib/ruby_llm/server_tool_call.rb +73 -0
  387. data/lib/ruby_llm/speech.rb +159 -0
  388. data/lib/ruby_llm/speech_chunk.rb +33 -0
  389. data/lib/ruby_llm/support/deprecator.rb +22 -0
  390. data/lib/ruby_llm/support/inspectable.rb +49 -0
  391. data/lib/ruby_llm/support/instrumentation.rb +41 -0
  392. data/lib/ruby_llm/support/utils.rb +147 -0
  393. data/lib/ruby_llm/thinking.rb +127 -20
  394. data/lib/ruby_llm/tokenization.rb +59 -0
  395. data/lib/ruby_llm/tokens.rb +103 -33
  396. data/lib/ruby_llm/tool.rb +266 -91
  397. data/lib/ruby_llm/tool_call.rb +36 -3
  398. data/lib/ruby_llm/tools/server_tools.rb +109 -0
  399. data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
  400. data/lib/ruby_llm/transcription.rb +138 -13
  401. data/lib/ruby_llm/transcription_chunk.rb +68 -0
  402. data/lib/ruby_llm/transport/connection.rb +193 -0
  403. data/lib/ruby_llm/transport/error_middleware.rb +131 -0
  404. data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
  405. data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
  406. data/lib/ruby_llm/uploaded_file.rb +144 -0
  407. data/lib/ruby_llm/version.rb +2 -1
  408. data/lib/ruby_llm/video.rb +136 -0
  409. data/lib/ruby_llm/video_job.rb +150 -0
  410. data/lib/ruby_llm/workflow.rb +91 -0
  411. data/lib/ruby_llm.rb +380 -6
  412. data/lib/tasks/ruby_llm.rake +21 -16
  413. data/skills/rubyllm/SKILL.md +81 -0
  414. data/skills/rubyllm/agents/openai.yaml +4 -0
  415. metadata +339 -97
  416. data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
  417. data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
  418. data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
  419. data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
  420. data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
  421. data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
  422. data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
  423. data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
  424. data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
  425. data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
  426. data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
  427. data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
  428. data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
  429. data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
  430. data/lib/ruby_llm/active_record/model_methods.rb +0 -82
  431. data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
  432. data/lib/ruby_llm/aliases.rb +0 -41
  433. data/lib/ruby_llm/connection.rb +0 -159
  434. data/lib/ruby_llm/content.rb +0 -91
  435. data/lib/ruby_llm/deprecator.rb +0 -24
  436. data/lib/ruby_llm/error_middleware.rb +0 -81
  437. data/lib/ruby_llm/instrumentation.rb +0 -36
  438. data/lib/ruby_llm/mime_type.rb +0 -96
  439. data/lib/ruby_llm/model/info.rb +0 -164
  440. data/lib/ruby_llm/model_registry.rb +0 -39
  441. data/lib/ruby_llm/models_schema.json +0 -171
  442. data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
  443. data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
  444. data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
  445. data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
  446. data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
  447. data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
  448. data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
  449. data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
  450. data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
  451. data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
  452. data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
  453. data/lib/ruby_llm/providers/gemini/images.rb +0 -47
  454. data/lib/ruby_llm/providers/gemini/models.rb +0 -38
  455. data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
  456. data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
  457. data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
  458. data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
  459. data/lib/ruby_llm/providers/openai/chat.rb +0 -236
  460. data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
  461. data/lib/ruby_llm/providers/openai/images.rb +0 -90
  462. data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
  463. data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
  464. data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
  465. data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
  466. data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
  467. data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
  468. data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
  469. data/lib/ruby_llm/stream_accumulator.rb +0 -218
  470. data/lib/ruby_llm/streaming.rb +0 -179
  471. data/lib/ruby_llm/tool_concurrency.rb +0 -105
  472. data/lib/ruby_llm/utils.rb +0 -130
  473. data/lib/tasks/models.rake +0 -593
  474. data/lib/tasks/release.rake +0 -94
  475. data/lib/tasks/vcr.rake +0 -124
@@ -0,0 +1,62 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ class Transcription
5
+ class WavAudio # :nodoc: all
6
+ attr_reader :data, :sample_rate, :channels, :bits_per_sample, :encoding
7
+
8
+ def initialize(content)
9
+ content = wav_content(content)
10
+ parse_chunks(content)
11
+ raise ArgumentError, 'WAV file must contain audio data and format information' unless @data && @sample_rate
12
+ end
13
+
14
+ def duration
15
+ data.bytesize.fdiv(sample_rate * channels * bits_per_sample / 8)
16
+ end
17
+
18
+ private
19
+
20
+ def wav_content(content)
21
+ unless content.start_with?('RIFF') && content.byteslice(8, 4) == 'WAVE'
22
+ raise ArgumentError, 'This streaming transcription endpoint requires a WAV file'
23
+ end
24
+
25
+ declared_size = content.byteslice(4, 4).unpack1('V')
26
+ return content if declared_size == 0xFFFFFFFF
27
+ raise ArgumentError, 'WAV file is truncated' if content.bytesize < declared_size + 8
28
+
29
+ content.byteslice(0, declared_size + 8)
30
+ end
31
+
32
+ def parse_chunks(content)
33
+ offset = 12
34
+ while offset + 8 <= content.bytesize
35
+ name = content.byteslice(offset, 4)
36
+ length = content.byteslice(offset + 4, 4).unpack1('V')
37
+ length = content.bytesize - offset - 8 if name == 'data' && length == 0xFFFFFFFF
38
+ body = content.byteslice(offset + 8, length)
39
+ validate_chunk_length(body, length)
40
+
41
+ parse_format(body) if name == 'fmt '
42
+ @data = body.b if name == 'data'
43
+ offset += 8 + length + (length % 2)
44
+ end
45
+ raise ArgumentError, 'WAV file contains a truncated chunk or padding' unless offset == content.bytesize
46
+ end
47
+
48
+ def validate_chunk_length(body, length)
49
+ raise ArgumentError, 'WAV file contains a truncated chunk' unless body && body.bytesize == length
50
+ end
51
+
52
+ def parse_format(data)
53
+ raise ArgumentError, 'WAV file contains an invalid audio format' if data.bytesize < 16
54
+
55
+ @encoding, @channels, @sample_rate, _, _, @bits_per_sample = data.unpack('vvVVvv')
56
+ return if @channels.positive? && @sample_rate.positive? && @bits_per_sample.positive?
57
+
58
+ raise ArgumentError, 'WAV file contains an invalid audio format'
59
+ end
60
+ end
61
+ end
62
+ end
@@ -1,11 +1,41 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module RubyLLM
4
- # Represents a transcription of audio content.
4
+ # A Transcription is text produced from spoken audio. RubyLLM.transcribe
5
+ # returns one. It holds the transcript along with any metadata the provider
6
+ # reports, such as language, duration, and timed segments.
7
+ #
8
+ # transcription = RubyLLM.transcribe("meeting.wav")
9
+ # transcription.text # => "Welcome to today's meeting..."
10
+ # transcription.model # => "gpt-transcribe"
11
+ #
5
12
  class Transcription
6
- attr_reader :text, :model, :language, :duration, :segments, :words, :input_tokens, :output_tokens
13
+ include Support::Inspectable
14
+ include Accounting::Usage::Result
7
15
 
8
- def initialize(text:, model:, **attributes)
16
+ # The transcribed text.
17
+ attr_reader :text
18
+
19
+ # The id of the model that produced the transcription.
20
+ attr_reader :model
21
+
22
+ # The language of the audio, or +nil+ when the provider does not report it.
23
+ attr_reader :language
24
+
25
+ # The audio duration in seconds, or +nil+ when the provider does not
26
+ # report it.
27
+ attr_reader :duration
28
+
29
+ # The timed segments of the transcript as an array of hashes, or +nil+
30
+ # when the provider does not return segments. Diarization models add a
31
+ # speaker label to each segment.
32
+ attr_reader :segments
33
+
34
+ # Word timing and speaker labels as an array of hashes, or +nil+ when
35
+ # the provider does not return them. Request timing with +timestamps:+.
36
+ attr_reader :words
37
+
38
+ def initialize(text:, model:, **attributes) # :nodoc:
9
39
  @text = text
10
40
  @model = model
11
41
  @language = attributes[:language]
@@ -14,23 +44,118 @@ module RubyLLM
14
44
  @words = attributes[:words]
15
45
  @input_tokens = attributes[:input_tokens]
16
46
  @output_tokens = attributes[:output_tokens]
47
+ @reported_cost = attributes[:reported_cost]
17
48
  end
18
49
 
19
- def self.transcribe(audio_file, **kwargs)
20
- model = kwargs.delete(:model)
21
- language = kwargs.delete(:language)
22
- provider = kwargs.delete(:provider)
23
- assume_model_exists = kwargs.delete(:assume_model_exists) { false }
24
- context = kwargs.delete(:context)
25
- options = kwargs
50
+ # Returns usage aggregated across every provider attempt.
51
+ def tokens
52
+ return ruby_llm_usage_tokens unless ruby_llm_usage_entries.empty?
53
+
54
+ Tokens.new(input: @input_tokens, output: @output_tokens, reported_cost: @reported_cost)
55
+ end
26
56
 
57
+ # Returns the transcription cost across every provider attempt.
58
+ def cost
59
+ return ruby_llm_usage_cost unless ruby_llm_usage_entries.empty?
60
+
61
+ Cost.new(tokens:, model: model_info, category: :audio_tokens)
62
+ end
63
+
64
+ def model_info # :nodoc:
65
+ @model_info ||= RubyLLM.models.find(model)
66
+ rescue ModelNotFoundError
67
+ nil
68
+ end
69
+
70
+ # Transcribes +audio_file+ and returns a Transcription. The file may be
71
+ # a path, URL, or IO object. Uses
72
+ # <tt>config.default_transcription_model</tt> unless +model:+ is given.
73
+ # Pass +provider:+ and <tt>assume_model_exists: true</tt> to use a model
74
+ # that is not in the registry.
75
+ #
76
+ # RubyLLM.transcribe("meeting.wav")
77
+ # RubyLLM.transcribe("entrevista.mp3", language: "es")
78
+ # RubyLLM.transcribe(
79
+ # "team-meeting.wav",
80
+ # model: "gpt-4o-transcribe-diarize",
81
+ # speaker_names: ["Alice", "Bob"],
82
+ # speaker_references: ["alice-voice.wav", "bob-voice.wav"]
83
+ # )
84
+ #
85
+ # +language:+ hints at the spoken language using the provider's accepted
86
+ # ISO 639-1 or BCP-47 language code.
87
+ # +prompt:+ gives the model vocabulary or formatting guidance, and
88
+ # +temperature:+ adjusts sampling. +format:+ selects the transcript
89
+ # format in the provider's own vocabulary: OpenAI takes values such as
90
+ # <tt>"text"</tt>, <tt>"verbose_json"</tt>, or <tt>"diarized_json"</tt>,
91
+ # while Gemini takes a MIME type such as <tt>"text/plain"</tt>.
92
+ # +speaker_names:+ and +speaker_references:+ label the speakers on
93
+ # models that support diarization; references may be paths, URLs, or IO
94
+ # objects. Option support depends on the selected model and provider.
95
+ # +timestamps:+ requests +:word+ timestamps. Some providers also accept
96
+ # +:segment+ or +:character+; unsupported granularities raise ArgumentError.
97
+ # +provider_options:+ takes options in the provider's request vocabulary
98
+ # and merges them into the rendered request as-is.
99
+ #
100
+ # Given a block, the transcript streams: each TranscriptionChunk is
101
+ # yielded as it arrives and the completed Transcription is still
102
+ # returned. Partial chunks replace earlier tentative text; only
103
+ # +delta+ fields append to the committed transcript. WebSocket-based
104
+ # providers require the optional +websocket-driver+ gem.
105
+ #
106
+ # transcription = RubyLLM.transcribe("meeting.wav", model: "gpt-4o-transcribe") do |chunk|
107
+ # print chunk.delta
108
+ # end
109
+ #
110
+ # Raises RubyLLM::ModelNotFoundError if +model:+ is not in the registry,
111
+ # and RubyLLM::Error when a block is given to a provider that does not
112
+ # stream transcriptions.
113
+ def self.transcribe(audio_file,
114
+ model: nil,
115
+ language: nil,
116
+ provider: nil,
117
+ assume_model_exists: false,
118
+ context: nil,
119
+ prompt: nil,
120
+ temperature: nil,
121
+ format: nil,
122
+ timestamps: nil,
123
+ speaker_names: nil,
124
+ speaker_references: nil,
125
+ provider_options: {},
126
+ metadata: nil,
127
+ &block)
27
128
  config = context&.config || RubyLLM.config
28
129
  model ||= config.default_transcription_model
29
- model, provider_instance = Models.resolve(model, provider: provider, assume_exists: assume_model_exists,
130
+ model, provider_instance = Models.resolve(model, provider: provider, assume_model_exists: assume_model_exists,
30
131
  config: config)
31
- model_id = model.id
132
+ empty_tokens = Tokens.new
133
+ payload = {
134
+ provider: provider_instance.slug,
135
+ provider_class: provider_instance.class.display_name,
136
+ model: model.id,
137
+ model_info: model,
138
+ language: language,
139
+ provider_options: provider_options,
140
+ metadata: metadata,
141
+ tokens: empty_tokens,
142
+ cost: Cost.new(tokens: empty_tokens, model:, category: :audio_tokens)
143
+ }
144
+
145
+ RubyLLM.instrument('transcription.ruby_llm', payload, config: config) do |event|
146
+ result = provider_instance.transcribe(audio_file, model:, language:, format:, timestamps:, speaker_names:,
147
+ speaker_references:, provider_options:, prompt:,
148
+ temperature:, &block)
149
+ event[:result] = result
150
+ event[:response_model] = result.model
151
+ event[:tokens] = result.tokens
152
+ event[:cost] = result.cost
153
+ result
154
+ end
155
+ end
32
156
 
33
- provider_instance.transcribe(audio_file, model: model_id, language:, **options)
157
+ def inspect_attributes # :nodoc:
158
+ { text: text, model: model, language: language, duration: duration }
34
159
  end
35
160
  end
36
161
  end
@@ -0,0 +1,68 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ # A TranscriptionChunk is one event from a streaming transcription.
5
+ # RubyLLM.transcribe yields these to its block as the provider transcribes
6
+ # the audio, then returns the final Transcription.
7
+ #
8
+ # RubyLLM.transcribe("meeting.wav", model: "gpt-4o-transcribe") do |chunk|
9
+ # print chunk.delta if chunk.delta?
10
+ # end
11
+ #
12
+ class TranscriptionChunk
13
+ include Support::Inspectable
14
+
15
+ # Text deltas, arriving as the model transcribes.
16
+ DELTA = 'transcript.text.delta'
17
+
18
+ # A tentative transcript that may change before its segment completes.
19
+ PARTIAL = 'transcript.text.partial'
20
+
21
+ # A completed segment, on models that return timed or diarized segments.
22
+ SEGMENT = 'transcript.text.segment'
23
+
24
+ # The final event, carrying the complete transcript.
25
+ DONE = 'transcript.text.done'
26
+
27
+ # The normalized event type, such as
28
+ # <tt>"transcript.text.delta"</tt>.
29
+ attr_reader :type
30
+
31
+ # The text added by this event, or +nil+ for events that add no text.
32
+ attr_reader :delta
33
+
34
+ # The transcript for a partial event or the complete final transcript.
35
+ attr_reader :text
36
+
37
+ # The segment this event completed as a Hash, or +nil+. Diarization
38
+ # models label each segment with a speaker.
39
+ attr_reader :segment
40
+
41
+ # The parsed provider event, for fields RubyLLM does not normalize.
42
+ attr_reader :raw
43
+
44
+ def initialize(type:, delta: nil, text: nil, segment: nil, raw: nil) # :nodoc:
45
+ @type = type
46
+ @delta = delta
47
+ @text = text
48
+ @segment = segment
49
+ @raw = raw
50
+ end
51
+
52
+ # Whether this event carries a text delta.
53
+ def delta? = !delta.nil?
54
+
55
+ # Whether this is a tentative transcript, replacing the previous partial.
56
+ def partial? = type == PARTIAL
57
+
58
+ # Whether this event completed a segment.
59
+ def segment? = !segment.nil?
60
+
61
+ # Whether this is the final event of the transcription.
62
+ def done? = type == DONE
63
+
64
+ def inspect_attributes # :nodoc:
65
+ { type: type, delta: delta, text: text, segment: segment }
66
+ end
67
+ end
68
+ end
@@ -0,0 +1,193 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'faraday'
4
+ require 'faraday/multipart'
5
+ require 'faraday/retry'
6
+ require 'ruby_llm/transport/error_middleware'
7
+ require 'ruby_llm/transport/usage_middleware'
8
+ require 'timeout'
9
+
10
+ module RubyLLM
11
+ module Transport # :nodoc:
12
+ class Connection # :nodoc:
13
+ include Support::Inspectable
14
+
15
+ IDEMPOTENT_KEY = :ruby_llm_idempotent
16
+ STREAM_PROGRESS_KEY = :ruby_llm_stream_progress
17
+
18
+ attr_reader :provider, :connection, :config
19
+
20
+ def self.basic(config = RubyLLM.config, &)
21
+ Faraday.new do |f|
22
+ f.options.timeout = config.request_timeout
23
+ f.proxy = config.http_proxy if config.http_proxy
24
+ f.response :logger,
25
+ RubyLLM.logger,
26
+ bodies: false,
27
+ errors: true,
28
+ headers: false,
29
+ log_level: :debug
30
+ f.response :raise_error
31
+ yield f if block_given?
32
+ end
33
+ end
34
+
35
+ def initialize(provider, config, api_base: nil)
36
+ @provider = provider
37
+ @config = config
38
+
39
+ @connection = Faraday.new(api_base || provider.api_base) do |faraday|
40
+ setup_timeout(faraday)
41
+ setup_logging(faraday)
42
+ setup_retry(faraday)
43
+ setup_middleware(faraday)
44
+ setup_http_proxy(faraday)
45
+ end
46
+ end
47
+
48
+ def post(url, payload, usage: nil, idempotent: true, &)
49
+ instrument_request(:post, url) do
50
+ @connection.post url, payload do |req|
51
+ req.headers.merge! @provider.headers
52
+ set_usage_tracker(req, usage) if usage
53
+ mark_non_idempotent(req) unless idempotent
54
+ yield req if block_given?
55
+ end
56
+ end
57
+ end
58
+
59
+ def get(url, &)
60
+ instrument_request(:get, url) do
61
+ @connection.get url do |req|
62
+ req.headers.merge! @provider.headers
63
+ yield req if block_given?
64
+ end
65
+ end
66
+ end
67
+
68
+ def patch(url, payload, &)
69
+ instrument_request(:patch, url) do
70
+ @connection.patch url, payload do |req|
71
+ req.headers.merge! @provider.headers
72
+ yield req if block_given?
73
+ end
74
+ end
75
+ end
76
+
77
+ def delete(url, &)
78
+ instrument_request(:delete, url) do
79
+ @connection.delete url do |req|
80
+ req.headers.merge! @provider.headers
81
+ yield req if block_given?
82
+ end
83
+ end
84
+ end
85
+
86
+ private
87
+
88
+ def instrument_request(method, url)
89
+ payload = {
90
+ provider: @provider.slug,
91
+ method: method,
92
+ url: url
93
+ }
94
+
95
+ RubyLLM.instrument('request.ruby_llm', payload, config: @config) do |event|
96
+ response = yield
97
+ event[:status] = response.status if response.respond_to?(:status)
98
+ response
99
+ end
100
+ end
101
+
102
+ def setup_timeout(faraday)
103
+ faraday.options.timeout = @config.request_timeout
104
+ end
105
+
106
+ def setup_logging(faraday)
107
+ faraday.response :logger,
108
+ RubyLLM.logger,
109
+ bodies: RubyLLM.logger.debug?,
110
+ errors: true,
111
+ headers: false,
112
+ log_level: :debug do |logger|
113
+ logger.filter(logging_regexp('[A-Za-z0-9+/=]{100,}'), '[BASE64 DATA]')
114
+ logger.filter(logging_regexp('[-\\d.e,\\s]{100,}'), '[EMBEDDINGS ARRAY]')
115
+ end
116
+ end
117
+
118
+ def logging_regexp(pattern)
119
+ return Regexp.new(pattern) if @config.log_regexp_timeout.nil? || !Regexp.respond_to?(:timeout)
120
+
121
+ Regexp.new(pattern, timeout: @config.log_regexp_timeout)
122
+ end
123
+
124
+ def setup_retry(faraday)
125
+ faraday.request :retry, {
126
+ max: @config.max_retries,
127
+ interval: @config.retry_interval,
128
+ max_interval: @config.retry_max_interval,
129
+ interval_randomness: @config.retry_interval_randomness,
130
+ backoff_factor: @config.retry_backoff_factor,
131
+ methods: Faraday::Retry::Middleware::IDEMPOTENT_METHODS,
132
+ retry_if: lambda { |env, _exception|
133
+ env[:method] == :post && idempotent?(env) && !stream_delivered?(env)
134
+ },
135
+ exceptions: retry_exceptions
136
+ }
137
+ faraday.use :llm_usage
138
+ end
139
+
140
+ def stream_delivered?(env)
141
+ env[:request]&.context&.dig(STREAM_PROGRESS_KEY, :started)
142
+ end
143
+
144
+ def idempotent?(env)
145
+ env[:request]&.context&.dig(IDEMPOTENT_KEY) != false
146
+ end
147
+
148
+ def setup_middleware(faraday)
149
+ faraday.request :multipart
150
+ faraday.request :json
151
+ faraday.response :json
152
+ faraday.adapter(@config.faraday_adapter)
153
+ faraday.use :llm_errors, provider: @provider
154
+ end
155
+
156
+ def setup_http_proxy(faraday)
157
+ return unless @config.http_proxy
158
+
159
+ faraday.proxy = @config.http_proxy
160
+ end
161
+
162
+ def retry_exceptions
163
+ [
164
+ Errno::ETIMEDOUT,
165
+ Timeout::Error,
166
+ Faraday::TimeoutError,
167
+ Faraday::ConnectionFailed,
168
+ Faraday::RetriableResponse,
169
+ RubyLLM::RateLimitError,
170
+ RubyLLM::ServerError,
171
+ RubyLLM::ServiceUnavailableError,
172
+ RubyLLM::OverloadedError
173
+ ]
174
+ end
175
+
176
+ def set_usage_tracker(request, tracker)
177
+ context = request.options.context ||= {}
178
+ context[UsageMiddleware::CONTEXT_KEY] = tracker
179
+ end
180
+
181
+ # A request that creates server-side state cannot be replayed: a retry
182
+ # after a lost response submits the job a second time.
183
+ def mark_non_idempotent(request)
184
+ context = request.options.context ||= {}
185
+ context[IDEMPOTENT_KEY] = false
186
+ end
187
+
188
+ def inspect_attributes # :nodoc:
189
+ { provider: @provider.slug }
190
+ end
191
+ end
192
+ end
193
+ end
@@ -0,0 +1,131 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'faraday'
4
+ require 'ruby_llm/error'
5
+
6
+ module RubyLLM
7
+ module Transport # :nodoc:
8
+ class ErrorMiddleware < Faraday::Middleware # :nodoc: all
9
+ def initialize(app, options = {})
10
+ super(app)
11
+ @provider = options[:provider]
12
+ end
13
+
14
+ # Sits directly above the adapter, inside the retry middleware, so this
15
+ # runs once per attempt: streaming state stored on the env by a previous
16
+ # attempt must not leak into the next one.
17
+ def call(env)
18
+ env[:streaming_error_response] = nil
19
+ env[:streaming_state] = nil
20
+ @app.call(env).on_complete do |response|
21
+ apply_retry_delay(response)
22
+ self.class.parse_error(provider: @provider, response: streaming_error_response(response))
23
+ end
24
+ end
25
+
26
+ private
27
+
28
+ # The retry middleware only reads the standard Retry-After header, so
29
+ # provider-specific rate-limit headers are normalized into it here,
30
+ # where the provider is known.
31
+ def apply_retry_delay(response)
32
+ status = response.respond_to?(:status) ? response.status : response[:status]
33
+ return unless status == 429
34
+
35
+ headers = response[:response_headers]
36
+ if @provider && !headers['Retry-After'] && (delay = @provider.retry_delay(response))
37
+ headers['Retry-After'] = delay.to_s
38
+ end
39
+ end
40
+
41
+ def streaming_error_response(response)
42
+ stored_response = if response.respond_to?(:env) && response.env.respond_to?(:[])
43
+ response.env[:streaming_error_response]
44
+ elsif response.respond_to?(:[])
45
+ response[:streaming_error_response]
46
+ end
47
+
48
+ stored_response || response
49
+ rescue NameError
50
+ response
51
+ end
52
+
53
+ class << self
54
+ CONTEXT_LENGTH_PATTERNS = [
55
+ /context length/i,
56
+ /context window/i,
57
+ /exceeds?.*context size/i,
58
+ /maximum context/i,
59
+ /request too large/i,
60
+ /too many tokens/i,
61
+ /token count exceeds/i,
62
+ /input[_\s-]?token/i,
63
+ /input or output tokens? must be reduced/i,
64
+ /reduce the length of messages/i,
65
+ /prompt is too long/i,
66
+ /context limit/i
67
+ ].freeze
68
+
69
+ RATE_LIMIT_PATTERNS = [
70
+ /rate limit/i,
71
+ /per minute/i,
72
+ /per hour/i,
73
+ /per day/i
74
+ ].freeze
75
+
76
+ def parse_error(provider:, response:)
77
+ message = provider&.parse_error(response)
78
+
79
+ case response.status
80
+ when 200..399
81
+ message
82
+ when 400
83
+ raise ContextLengthExceededError.new(message, response:) if context_length_exceeded?(message)
84
+
85
+ raise BadRequestError.new(message, response:)
86
+ when 401
87
+ raise UnauthorizedError.new(message, response:)
88
+ when 402
89
+ raise PaymentRequiredError.new(message, response:)
90
+ when 403
91
+ raise ForbiddenError.new(message, response:)
92
+ when 429
93
+ raise RateLimitError.new(message, response:) if rate_limited?(message)
94
+ raise ContextLengthExceededError.new(message, response:) if context_length_exceeded?(message)
95
+
96
+ raise RateLimitError.new(message, response:)
97
+ when 500
98
+ raise ServerError.new(message, response:)
99
+ when 502..504
100
+ raise ServiceUnavailableError.new(message, response:)
101
+ when 529
102
+ raise OverloadedError.new(message, response:)
103
+ else
104
+ raise Error.new(message, response:)
105
+ end
106
+ end
107
+
108
+ private
109
+
110
+ # Providers hand back whatever their error body holds, which is not
111
+ # always a String: bedrock-mantle nests code, message, and type in a
112
+ # Hash. Match on the rendered text so any shape classifies.
113
+ def context_length_exceeded?(message)
114
+ text = message.to_s
115
+ return false if text.empty?
116
+
117
+ CONTEXT_LENGTH_PATTERNS.any? { |pattern| text.match?(pattern) }
118
+ end
119
+
120
+ def rate_limited?(message)
121
+ text = message.to_s
122
+ return false if text.empty?
123
+
124
+ RATE_LIMIT_PATTERNS.any? { |pattern| text.match?(pattern) }
125
+ end
126
+ end
127
+ end
128
+ end
129
+ end
130
+
131
+ Faraday::Middleware.register_middleware(llm_errors: RubyLLM::Transport::ErrorMiddleware)
@@ -0,0 +1,28 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'faraday'
4
+
5
+ module RubyLLM
6
+ module Transport # :nodoc:
7
+ # Sits inside Faraday retry middleware so every transport attempt produces
8
+ # one usage observation.
9
+ class UsageMiddleware < Faraday::Middleware # :nodoc: all
10
+ CONTEXT_KEY = :ruby_llm_usage_tracker
11
+
12
+ def call(env)
13
+ tracker = env.request.context&.[](CONTEXT_KEY)
14
+ return @app.call(env) unless tracker
15
+
16
+ entry = tracker.start
17
+ begin
18
+ @app.call(env)
19
+ rescue StandardError => e
20
+ tracker.fail_attempt(entry, e)
21
+ raise
22
+ end
23
+ end
24
+ end
25
+ end
26
+ end
27
+
28
+ Faraday::Middleware.register_middleware(llm_usage: RubyLLM::Transport::UsageMiddleware)