ruby_llm 1.16.0 → 2.0.0.rc2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.rdoc_options +25 -0
- data/README.md +85 -32
- data/exe/ruby_llm +8 -0
- data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
- data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
- data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
- data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
- data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
- data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
- data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
- data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
- data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
- data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
- data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
- data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
- data/lib/generators/ruby_llm/provider/cli.rb +175 -0
- data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
- data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
- data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
- data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
- data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
- data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
- data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
- data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
- data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
- data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
- data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
- data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
- data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
- data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
- data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
- data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
- data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
- data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
- data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
- data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
- data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
- data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
- data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
- data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
- data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
- data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
- data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
- data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
- data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
- data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
- data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
- data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
- data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
- data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
- data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
- data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
- data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
- data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
- data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
- data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
- data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
- data/lib/ruby_llm/accounting/usage.rb +245 -0
- data/lib/ruby_llm/active_record/acts_as.rb +94 -111
- data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
- data/lib/ruby_llm/active_record/batch.rb +97 -0
- data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
- data/lib/ruby_llm/active_record/message_methods.rb +113 -136
- data/lib/ruby_llm/active_record/model.rb +135 -0
- data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
- data/lib/ruby_llm/active_record/tool_call.rb +33 -0
- data/lib/ruby_llm/active_record/usage.rb +61 -0
- data/lib/ruby_llm/agent.rb +1065 -151
- data/lib/ruby_llm/aliases.json +268 -100
- data/lib/ruby_llm/attachment.rb +187 -48
- data/lib/ruby_llm/batch.rb +432 -0
- data/lib/ruby_llm/cached_content.rb +112 -0
- data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
- data/lib/ruby_llm/chat.rb +1127 -198
- data/lib/ruby_llm/chunk.rb +10 -0
- data/lib/ruby_llm/citation.rb +105 -0
- data/lib/ruby_llm/configuration.rb +261 -24
- data/lib/ruby_llm/context.rb +128 -6
- data/lib/ruby_llm/cost.rb +217 -80
- data/lib/ruby_llm/downloaded_file.rb +33 -0
- data/lib/ruby_llm/embedding.rb +121 -17
- data/lib/ruby_llm/embedding_request.rb +53 -0
- data/lib/ruby_llm/error.rb +159 -23
- data/lib/ruby_llm/fallback.rb +133 -0
- data/lib/ruby_llm/files/mime_type.rb +97 -0
- data/lib/ruby_llm/image.rb +153 -32
- data/lib/ruby_llm/message.rb +233 -54
- data/lib/ruby_llm/model/modalities.rb +17 -4
- data/lib/ruby_llm/model/pricing.rb +24 -5
- data/lib/ruby_llm/model/pricing_category.rb +103 -14
- data/lib/ruby_llm/model/pricing_tier.rb +55 -15
- data/lib/ruby_llm/model.rb +244 -2
- data/lib/ruby_llm/models/aliases.rb +41 -0
- data/lib/ruby_llm/models/registry.rb +165 -0
- data/lib/ruby_llm/models/schema.rb +99 -0
- data/lib/ruby_llm/models.json +73261 -33733
- data/lib/ruby_llm/models.rb +477 -215
- data/lib/ruby_llm/moderation.rb +139 -26
- data/lib/ruby_llm/ocr.rb +112 -0
- data/lib/ruby_llm/prompt.rb +79 -0
- data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
- data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
- data/lib/ruby_llm/protocol/streaming.rb +230 -0
- data/lib/ruby_llm/protocol.rb +662 -0
- data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
- data/lib/ruby_llm/protocols/anthropic/chat.rb +548 -0
- data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
- data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
- data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
- data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
- data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
- data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
- data/lib/ruby_llm/protocols/anthropic.rb +101 -0
- data/lib/ruby_llm/protocols/azure/files.rb +16 -0
- data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
- data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
- data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
- data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
- data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
- data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
- data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
- data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
- data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
- data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
- data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
- data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
- data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
- data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
- data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
- data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
- data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
- data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
- data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
- data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
- data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
- data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
- data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
- data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
- data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
- data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
- data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
- data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
- data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
- data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
- data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
- data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
- data/lib/ruby_llm/protocols/cohere.rb +21 -0
- data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
- data/lib/ruby_llm/protocols/converse/chat.rb +695 -0
- data/lib/ruby_llm/protocols/converse/media.rb +178 -0
- data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
- data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
- data/lib/ruby_llm/protocols/converse.rb +54 -0
- data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
- data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
- data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
- data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
- data/lib/ruby_llm/protocols/deepgram.rb +19 -0
- data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
- data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
- data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
- data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
- data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
- data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
- data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
- data/lib/ruby_llm/protocols/files.rb +119 -0
- data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
- data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
- data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
- data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
- data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
- data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
- data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
- data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
- data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
- data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
- data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
- data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
- data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
- data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
- data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
- data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
- data/lib/ruby_llm/protocols/gemini.rb +35 -0
- data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
- data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
- data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
- data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
- data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
- data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
- data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
- data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
- data/lib/ruby_llm/protocols/interactions.rb +29 -0
- data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
- data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
- data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
- data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
- data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
- data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
- data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
- data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
- data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
- data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
- data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
- data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
- data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
- data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
- data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
- data/lib/ruby_llm/protocols/openai/files.rb +42 -0
- data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
- data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
- data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
- data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
- data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
- data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
- data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
- data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
- data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
- data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
- data/lib/ruby_llm/protocols/responses/media.rb +61 -0
- data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
- data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
- data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
- data/lib/ruby_llm/protocols/responses.rb +35 -0
- data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
- data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
- data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
- data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
- data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
- data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
- data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
- data/lib/ruby_llm/protocols/xai/files.rb +30 -0
- data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
- data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
- data/lib/ruby_llm/provider.rb +560 -128
- data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
- data/lib/ruby_llm/providers/anthropic.rb +4 -6
- data/lib/ruby_llm/providers/azure/audio.rb +18 -0
- data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
- data/lib/ruby_llm/providers/azure/chat.rb +2 -9
- data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
- data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
- data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
- data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
- data/lib/ruby_llm/providers/azure/images.rb +22 -0
- data/lib/ruby_llm/providers/azure/media.rb +5 -14
- data/lib/ruby_llm/providers/azure/models.rb +35 -0
- data/lib/ruby_llm/providers/azure/responses.rb +26 -0
- data/lib/ruby_llm/providers/azure/videos.rb +64 -0
- data/lib/ruby_llm/providers/azure.rb +77 -78
- data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
- data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
- data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
- data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
- data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
- data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
- data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
- data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
- data/lib/ruby_llm/providers/bedrock.rb +216 -45
- data/lib/ruby_llm/providers/cohere.rb +31 -0
- data/lib/ruby_llm/providers/deepgram.rb +37 -0
- data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
- data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
- data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
- data/lib/ruby_llm/providers/deepseek.rb +9 -2
- data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
- data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
- data/lib/ruby_llm/providers/gemini.rb +15 -8
- data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
- data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
- data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
- data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
- data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
- data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
- data/lib/ruby_llm/providers/gpustack.rb +35 -11
- data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
- data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
- data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
- data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
- data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
- data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
- data/lib/ruby_llm/providers/mistral/media.rb +6 -18
- data/lib/ruby_llm/providers/mistral/models.rb +57 -23
- data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
- data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
- data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
- data/lib/ruby_llm/providers/mistral.rb +16 -4
- data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
- data/lib/ruby_llm/providers/ollama/media.rb +6 -15
- data/lib/ruby_llm/providers/ollama/models.rb +50 -9
- data/lib/ruby_llm/providers/ollama.rb +9 -8
- data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
- data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
- data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
- data/lib/ruby_llm/providers/openai/models.rb +23 -23
- data/lib/ruby_llm/providers/openai/responses.rb +13 -0
- data/lib/ruby_llm/providers/openai.rb +92 -11
- data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
- data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
- data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
- data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
- data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
- data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
- data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
- data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
- data/lib/ruby_llm/providers/openrouter.rb +78 -20
- data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
- data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
- data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
- data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
- data/lib/ruby_llm/providers/perplexity.rb +28 -20
- data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
- data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
- data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
- data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
- data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
- data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
- data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
- data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
- data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
- data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
- data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
- data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
- data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
- data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
- data/lib/ruby_llm/providers/vertexai.rb +160 -17
- data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
- data/lib/ruby_llm/providers/xai/chat.rb +3 -2
- data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
- data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
- data/lib/ruby_llm/providers/xai/images.rb +91 -0
- data/lib/ruby_llm/providers/xai/models.rb +32 -36
- data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
- data/lib/ruby_llm/providers/xai/responses.rb +52 -0
- data/lib/ruby_llm/providers/xai/speech.rb +45 -0
- data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
- data/lib/ruby_llm/providers/xai/videos.rb +87 -0
- data/lib/ruby_llm/providers/xai.rb +15 -5
- data/lib/ruby_llm/railtie.rb +7 -16
- data/lib/ruby_llm/rerank.rb +105 -0
- data/lib/ruby_llm/research_job.rb +241 -0
- data/lib/ruby_llm/search_results.rb +68 -0
- data/lib/ruby_llm/server_tool_call.rb +73 -0
- data/lib/ruby_llm/speech.rb +159 -0
- data/lib/ruby_llm/speech_chunk.rb +33 -0
- data/lib/ruby_llm/support/deprecator.rb +22 -0
- data/lib/ruby_llm/support/inspectable.rb +49 -0
- data/lib/ruby_llm/support/instrumentation.rb +41 -0
- data/lib/ruby_llm/support/utils.rb +147 -0
- data/lib/ruby_llm/thinking.rb +127 -20
- data/lib/ruby_llm/tokenization.rb +59 -0
- data/lib/ruby_llm/tokens.rb +103 -33
- data/lib/ruby_llm/tool.rb +266 -91
- data/lib/ruby_llm/tool_call.rb +36 -3
- data/lib/ruby_llm/tools/server_tools.rb +109 -0
- data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
- data/lib/ruby_llm/transcription.rb +138 -13
- data/lib/ruby_llm/transcription_chunk.rb +68 -0
- data/lib/ruby_llm/transport/connection.rb +193 -0
- data/lib/ruby_llm/transport/error_middleware.rb +131 -0
- data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
- data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
- data/lib/ruby_llm/uploaded_file.rb +144 -0
- data/lib/ruby_llm/version.rb +2 -1
- data/lib/ruby_llm/video.rb +136 -0
- data/lib/ruby_llm/video_job.rb +150 -0
- data/lib/ruby_llm/workflow.rb +91 -0
- data/lib/ruby_llm.rb +380 -6
- data/lib/tasks/ruby_llm.rake +21 -16
- data/skills/rubyllm/SKILL.md +81 -0
- data/skills/rubyllm/agents/openai.yaml +4 -0
- metadata +339 -97
- data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
- data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
- data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
- data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
- data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
- data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
- data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
- data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
- data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
- data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
- data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
- data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
- data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
- data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
- data/lib/ruby_llm/active_record/model_methods.rb +0 -82
- data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
- data/lib/ruby_llm/aliases.rb +0 -41
- data/lib/ruby_llm/connection.rb +0 -159
- data/lib/ruby_llm/content.rb +0 -91
- data/lib/ruby_llm/deprecator.rb +0 -24
- data/lib/ruby_llm/error_middleware.rb +0 -81
- data/lib/ruby_llm/instrumentation.rb +0 -36
- data/lib/ruby_llm/mime_type.rb +0 -96
- data/lib/ruby_llm/model/info.rb +0 -164
- data/lib/ruby_llm/model_registry.rb +0 -39
- data/lib/ruby_llm/models_schema.json +0 -171
- data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
- data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
- data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
- data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
- data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
- data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
- data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
- data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
- data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
- data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
- data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
- data/lib/ruby_llm/providers/gemini/images.rb +0 -47
- data/lib/ruby_llm/providers/gemini/models.rb +0 -38
- data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
- data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
- data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
- data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
- data/lib/ruby_llm/providers/openai/chat.rb +0 -236
- data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
- data/lib/ruby_llm/providers/openai/images.rb +0 -90
- data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
- data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
- data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
- data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
- data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
- data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
- data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
- data/lib/ruby_llm/stream_accumulator.rb +0 -218
- data/lib/ruby_llm/streaming.rb +0 -179
- data/lib/ruby_llm/tool_concurrency.rb +0 -105
- data/lib/ruby_llm/utils.rb +0 -130
- data/lib/tasks/models.rake +0 -593
- data/lib/tasks/release.rake +0 -94
- data/lib/tasks/vcr.rake +0 -124
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
class Transcription
|
|
5
|
+
class WavAudio # :nodoc: all
|
|
6
|
+
attr_reader :data, :sample_rate, :channels, :bits_per_sample, :encoding
|
|
7
|
+
|
|
8
|
+
def initialize(content)
|
|
9
|
+
content = wav_content(content)
|
|
10
|
+
parse_chunks(content)
|
|
11
|
+
raise ArgumentError, 'WAV file must contain audio data and format information' unless @data && @sample_rate
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
def duration
|
|
15
|
+
data.bytesize.fdiv(sample_rate * channels * bits_per_sample / 8)
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
private
|
|
19
|
+
|
|
20
|
+
def wav_content(content)
|
|
21
|
+
unless content.start_with?('RIFF') && content.byteslice(8, 4) == 'WAVE'
|
|
22
|
+
raise ArgumentError, 'This streaming transcription endpoint requires a WAV file'
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
declared_size = content.byteslice(4, 4).unpack1('V')
|
|
26
|
+
return content if declared_size == 0xFFFFFFFF
|
|
27
|
+
raise ArgumentError, 'WAV file is truncated' if content.bytesize < declared_size + 8
|
|
28
|
+
|
|
29
|
+
content.byteslice(0, declared_size + 8)
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
def parse_chunks(content)
|
|
33
|
+
offset = 12
|
|
34
|
+
while offset + 8 <= content.bytesize
|
|
35
|
+
name = content.byteslice(offset, 4)
|
|
36
|
+
length = content.byteslice(offset + 4, 4).unpack1('V')
|
|
37
|
+
length = content.bytesize - offset - 8 if name == 'data' && length == 0xFFFFFFFF
|
|
38
|
+
body = content.byteslice(offset + 8, length)
|
|
39
|
+
validate_chunk_length(body, length)
|
|
40
|
+
|
|
41
|
+
parse_format(body) if name == 'fmt '
|
|
42
|
+
@data = body.b if name == 'data'
|
|
43
|
+
offset += 8 + length + (length % 2)
|
|
44
|
+
end
|
|
45
|
+
raise ArgumentError, 'WAV file contains a truncated chunk or padding' unless offset == content.bytesize
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def validate_chunk_length(body, length)
|
|
49
|
+
raise ArgumentError, 'WAV file contains a truncated chunk' unless body && body.bytesize == length
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def parse_format(data)
|
|
53
|
+
raise ArgumentError, 'WAV file contains an invalid audio format' if data.bytesize < 16
|
|
54
|
+
|
|
55
|
+
@encoding, @channels, @sample_rate, _, _, @bits_per_sample = data.unpack('vvVVvv')
|
|
56
|
+
return if @channels.positive? && @sample_rate.positive? && @bits_per_sample.positive?
|
|
57
|
+
|
|
58
|
+
raise ArgumentError, 'WAV file contains an invalid audio format'
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
end
|
|
@@ -1,11 +1,41 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
module RubyLLM
|
|
4
|
-
#
|
|
4
|
+
# A Transcription is text produced from spoken audio. RubyLLM.transcribe
|
|
5
|
+
# returns one. It holds the transcript along with any metadata the provider
|
|
6
|
+
# reports, such as language, duration, and timed segments.
|
|
7
|
+
#
|
|
8
|
+
# transcription = RubyLLM.transcribe("meeting.wav")
|
|
9
|
+
# transcription.text # => "Welcome to today's meeting..."
|
|
10
|
+
# transcription.model # => "gpt-transcribe"
|
|
11
|
+
#
|
|
5
12
|
class Transcription
|
|
6
|
-
|
|
13
|
+
include Support::Inspectable
|
|
14
|
+
include Accounting::Usage::Result
|
|
7
15
|
|
|
8
|
-
|
|
16
|
+
# The transcribed text.
|
|
17
|
+
attr_reader :text
|
|
18
|
+
|
|
19
|
+
# The id of the model that produced the transcription.
|
|
20
|
+
attr_reader :model
|
|
21
|
+
|
|
22
|
+
# The language of the audio, or +nil+ when the provider does not report it.
|
|
23
|
+
attr_reader :language
|
|
24
|
+
|
|
25
|
+
# The audio duration in seconds, or +nil+ when the provider does not
|
|
26
|
+
# report it.
|
|
27
|
+
attr_reader :duration
|
|
28
|
+
|
|
29
|
+
# The timed segments of the transcript as an array of hashes, or +nil+
|
|
30
|
+
# when the provider does not return segments. Diarization models add a
|
|
31
|
+
# speaker label to each segment.
|
|
32
|
+
attr_reader :segments
|
|
33
|
+
|
|
34
|
+
# Word timing and speaker labels as an array of hashes, or +nil+ when
|
|
35
|
+
# the provider does not return them. Request timing with +timestamps:+.
|
|
36
|
+
attr_reader :words
|
|
37
|
+
|
|
38
|
+
def initialize(text:, model:, **attributes) # :nodoc:
|
|
9
39
|
@text = text
|
|
10
40
|
@model = model
|
|
11
41
|
@language = attributes[:language]
|
|
@@ -14,23 +44,118 @@ module RubyLLM
|
|
|
14
44
|
@words = attributes[:words]
|
|
15
45
|
@input_tokens = attributes[:input_tokens]
|
|
16
46
|
@output_tokens = attributes[:output_tokens]
|
|
47
|
+
@reported_cost = attributes[:reported_cost]
|
|
17
48
|
end
|
|
18
49
|
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
options = kwargs
|
|
50
|
+
# Returns usage aggregated across every provider attempt.
|
|
51
|
+
def tokens
|
|
52
|
+
return ruby_llm_usage_tokens unless ruby_llm_usage_entries.empty?
|
|
53
|
+
|
|
54
|
+
Tokens.new(input: @input_tokens, output: @output_tokens, reported_cost: @reported_cost)
|
|
55
|
+
end
|
|
26
56
|
|
|
57
|
+
# Returns the transcription cost across every provider attempt.
|
|
58
|
+
def cost
|
|
59
|
+
return ruby_llm_usage_cost unless ruby_llm_usage_entries.empty?
|
|
60
|
+
|
|
61
|
+
Cost.new(tokens:, model: model_info, category: :audio_tokens)
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
def model_info # :nodoc:
|
|
65
|
+
@model_info ||= RubyLLM.models.find(model)
|
|
66
|
+
rescue ModelNotFoundError
|
|
67
|
+
nil
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# Transcribes +audio_file+ and returns a Transcription. The file may be
|
|
71
|
+
# a path, URL, or IO object. Uses
|
|
72
|
+
# <tt>config.default_transcription_model</tt> unless +model:+ is given.
|
|
73
|
+
# Pass +provider:+ and <tt>assume_model_exists: true</tt> to use a model
|
|
74
|
+
# that is not in the registry.
|
|
75
|
+
#
|
|
76
|
+
# RubyLLM.transcribe("meeting.wav")
|
|
77
|
+
# RubyLLM.transcribe("entrevista.mp3", language: "es")
|
|
78
|
+
# RubyLLM.transcribe(
|
|
79
|
+
# "team-meeting.wav",
|
|
80
|
+
# model: "gpt-4o-transcribe-diarize",
|
|
81
|
+
# speaker_names: ["Alice", "Bob"],
|
|
82
|
+
# speaker_references: ["alice-voice.wav", "bob-voice.wav"]
|
|
83
|
+
# )
|
|
84
|
+
#
|
|
85
|
+
# +language:+ hints at the spoken language using the provider's accepted
|
|
86
|
+
# ISO 639-1 or BCP-47 language code.
|
|
87
|
+
# +prompt:+ gives the model vocabulary or formatting guidance, and
|
|
88
|
+
# +temperature:+ adjusts sampling. +format:+ selects the transcript
|
|
89
|
+
# format in the provider's own vocabulary: OpenAI takes values such as
|
|
90
|
+
# <tt>"text"</tt>, <tt>"verbose_json"</tt>, or <tt>"diarized_json"</tt>,
|
|
91
|
+
# while Gemini takes a MIME type such as <tt>"text/plain"</tt>.
|
|
92
|
+
# +speaker_names:+ and +speaker_references:+ label the speakers on
|
|
93
|
+
# models that support diarization; references may be paths, URLs, or IO
|
|
94
|
+
# objects. Option support depends on the selected model and provider.
|
|
95
|
+
# +timestamps:+ requests +:word+ timestamps. Some providers also accept
|
|
96
|
+
# +:segment+ or +:character+; unsupported granularities raise ArgumentError.
|
|
97
|
+
# +provider_options:+ takes options in the provider's request vocabulary
|
|
98
|
+
# and merges them into the rendered request as-is.
|
|
99
|
+
#
|
|
100
|
+
# Given a block, the transcript streams: each TranscriptionChunk is
|
|
101
|
+
# yielded as it arrives and the completed Transcription is still
|
|
102
|
+
# returned. Partial chunks replace earlier tentative text; only
|
|
103
|
+
# +delta+ fields append to the committed transcript. WebSocket-based
|
|
104
|
+
# providers require the optional +websocket-driver+ gem.
|
|
105
|
+
#
|
|
106
|
+
# transcription = RubyLLM.transcribe("meeting.wav", model: "gpt-4o-transcribe") do |chunk|
|
|
107
|
+
# print chunk.delta
|
|
108
|
+
# end
|
|
109
|
+
#
|
|
110
|
+
# Raises RubyLLM::ModelNotFoundError if +model:+ is not in the registry,
|
|
111
|
+
# and RubyLLM::Error when a block is given to a provider that does not
|
|
112
|
+
# stream transcriptions.
|
|
113
|
+
def self.transcribe(audio_file,
|
|
114
|
+
model: nil,
|
|
115
|
+
language: nil,
|
|
116
|
+
provider: nil,
|
|
117
|
+
assume_model_exists: false,
|
|
118
|
+
context: nil,
|
|
119
|
+
prompt: nil,
|
|
120
|
+
temperature: nil,
|
|
121
|
+
format: nil,
|
|
122
|
+
timestamps: nil,
|
|
123
|
+
speaker_names: nil,
|
|
124
|
+
speaker_references: nil,
|
|
125
|
+
provider_options: {},
|
|
126
|
+
metadata: nil,
|
|
127
|
+
&block)
|
|
27
128
|
config = context&.config || RubyLLM.config
|
|
28
129
|
model ||= config.default_transcription_model
|
|
29
|
-
model, provider_instance = Models.resolve(model, provider: provider,
|
|
130
|
+
model, provider_instance = Models.resolve(model, provider: provider, assume_model_exists: assume_model_exists,
|
|
30
131
|
config: config)
|
|
31
|
-
|
|
132
|
+
empty_tokens = Tokens.new
|
|
133
|
+
payload = {
|
|
134
|
+
provider: provider_instance.slug,
|
|
135
|
+
provider_class: provider_instance.class.display_name,
|
|
136
|
+
model: model.id,
|
|
137
|
+
model_info: model,
|
|
138
|
+
language: language,
|
|
139
|
+
provider_options: provider_options,
|
|
140
|
+
metadata: metadata,
|
|
141
|
+
tokens: empty_tokens,
|
|
142
|
+
cost: Cost.new(tokens: empty_tokens, model:, category: :audio_tokens)
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
RubyLLM.instrument('transcription.ruby_llm', payload, config: config) do |event|
|
|
146
|
+
result = provider_instance.transcribe(audio_file, model:, language:, format:, timestamps:, speaker_names:,
|
|
147
|
+
speaker_references:, provider_options:, prompt:,
|
|
148
|
+
temperature:, &block)
|
|
149
|
+
event[:result] = result
|
|
150
|
+
event[:response_model] = result.model
|
|
151
|
+
event[:tokens] = result.tokens
|
|
152
|
+
event[:cost] = result.cost
|
|
153
|
+
result
|
|
154
|
+
end
|
|
155
|
+
end
|
|
32
156
|
|
|
33
|
-
|
|
157
|
+
def inspect_attributes # :nodoc:
|
|
158
|
+
{ text: text, model: model, language: language, duration: duration }
|
|
34
159
|
end
|
|
35
160
|
end
|
|
36
161
|
end
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
# A TranscriptionChunk is one event from a streaming transcription.
|
|
5
|
+
# RubyLLM.transcribe yields these to its block as the provider transcribes
|
|
6
|
+
# the audio, then returns the final Transcription.
|
|
7
|
+
#
|
|
8
|
+
# RubyLLM.transcribe("meeting.wav", model: "gpt-4o-transcribe") do |chunk|
|
|
9
|
+
# print chunk.delta if chunk.delta?
|
|
10
|
+
# end
|
|
11
|
+
#
|
|
12
|
+
class TranscriptionChunk
|
|
13
|
+
include Support::Inspectable
|
|
14
|
+
|
|
15
|
+
# Text deltas, arriving as the model transcribes.
|
|
16
|
+
DELTA = 'transcript.text.delta'
|
|
17
|
+
|
|
18
|
+
# A tentative transcript that may change before its segment completes.
|
|
19
|
+
PARTIAL = 'transcript.text.partial'
|
|
20
|
+
|
|
21
|
+
# A completed segment, on models that return timed or diarized segments.
|
|
22
|
+
SEGMENT = 'transcript.text.segment'
|
|
23
|
+
|
|
24
|
+
# The final event, carrying the complete transcript.
|
|
25
|
+
DONE = 'transcript.text.done'
|
|
26
|
+
|
|
27
|
+
# The normalized event type, such as
|
|
28
|
+
# <tt>"transcript.text.delta"</tt>.
|
|
29
|
+
attr_reader :type
|
|
30
|
+
|
|
31
|
+
# The text added by this event, or +nil+ for events that add no text.
|
|
32
|
+
attr_reader :delta
|
|
33
|
+
|
|
34
|
+
# The transcript for a partial event or the complete final transcript.
|
|
35
|
+
attr_reader :text
|
|
36
|
+
|
|
37
|
+
# The segment this event completed as a Hash, or +nil+. Diarization
|
|
38
|
+
# models label each segment with a speaker.
|
|
39
|
+
attr_reader :segment
|
|
40
|
+
|
|
41
|
+
# The parsed provider event, for fields RubyLLM does not normalize.
|
|
42
|
+
attr_reader :raw
|
|
43
|
+
|
|
44
|
+
def initialize(type:, delta: nil, text: nil, segment: nil, raw: nil) # :nodoc:
|
|
45
|
+
@type = type
|
|
46
|
+
@delta = delta
|
|
47
|
+
@text = text
|
|
48
|
+
@segment = segment
|
|
49
|
+
@raw = raw
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
# Whether this event carries a text delta.
|
|
53
|
+
def delta? = !delta.nil?
|
|
54
|
+
|
|
55
|
+
# Whether this is a tentative transcript, replacing the previous partial.
|
|
56
|
+
def partial? = type == PARTIAL
|
|
57
|
+
|
|
58
|
+
# Whether this event completed a segment.
|
|
59
|
+
def segment? = !segment.nil?
|
|
60
|
+
|
|
61
|
+
# Whether this is the final event of the transcription.
|
|
62
|
+
def done? = type == DONE
|
|
63
|
+
|
|
64
|
+
def inspect_attributes # :nodoc:
|
|
65
|
+
{ type: type, delta: delta, text: text, segment: segment }
|
|
66
|
+
end
|
|
67
|
+
end
|
|
68
|
+
end
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'faraday'
|
|
4
|
+
require 'faraday/multipart'
|
|
5
|
+
require 'faraday/retry'
|
|
6
|
+
require 'ruby_llm/transport/error_middleware'
|
|
7
|
+
require 'ruby_llm/transport/usage_middleware'
|
|
8
|
+
require 'timeout'
|
|
9
|
+
|
|
10
|
+
module RubyLLM
|
|
11
|
+
module Transport # :nodoc:
|
|
12
|
+
class Connection # :nodoc:
|
|
13
|
+
include Support::Inspectable
|
|
14
|
+
|
|
15
|
+
IDEMPOTENT_KEY = :ruby_llm_idempotent
|
|
16
|
+
STREAM_PROGRESS_KEY = :ruby_llm_stream_progress
|
|
17
|
+
|
|
18
|
+
attr_reader :provider, :connection, :config
|
|
19
|
+
|
|
20
|
+
def self.basic(config = RubyLLM.config, &)
|
|
21
|
+
Faraday.new do |f|
|
|
22
|
+
f.options.timeout = config.request_timeout
|
|
23
|
+
f.proxy = config.http_proxy if config.http_proxy
|
|
24
|
+
f.response :logger,
|
|
25
|
+
RubyLLM.logger,
|
|
26
|
+
bodies: false,
|
|
27
|
+
errors: true,
|
|
28
|
+
headers: false,
|
|
29
|
+
log_level: :debug
|
|
30
|
+
f.response :raise_error
|
|
31
|
+
yield f if block_given?
|
|
32
|
+
end
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
def initialize(provider, config, api_base: nil)
|
|
36
|
+
@provider = provider
|
|
37
|
+
@config = config
|
|
38
|
+
|
|
39
|
+
@connection = Faraday.new(api_base || provider.api_base) do |faraday|
|
|
40
|
+
setup_timeout(faraday)
|
|
41
|
+
setup_logging(faraday)
|
|
42
|
+
setup_retry(faraday)
|
|
43
|
+
setup_middleware(faraday)
|
|
44
|
+
setup_http_proxy(faraday)
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def post(url, payload, usage: nil, idempotent: true, &)
|
|
49
|
+
instrument_request(:post, url) do
|
|
50
|
+
@connection.post url, payload do |req|
|
|
51
|
+
req.headers.merge! @provider.headers
|
|
52
|
+
set_usage_tracker(req, usage) if usage
|
|
53
|
+
mark_non_idempotent(req) unless idempotent
|
|
54
|
+
yield req if block_given?
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def get(url, &)
|
|
60
|
+
instrument_request(:get, url) do
|
|
61
|
+
@connection.get url do |req|
|
|
62
|
+
req.headers.merge! @provider.headers
|
|
63
|
+
yield req if block_given?
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
def patch(url, payload, &)
|
|
69
|
+
instrument_request(:patch, url) do
|
|
70
|
+
@connection.patch url, payload do |req|
|
|
71
|
+
req.headers.merge! @provider.headers
|
|
72
|
+
yield req if block_given?
|
|
73
|
+
end
|
|
74
|
+
end
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
def delete(url, &)
|
|
78
|
+
instrument_request(:delete, url) do
|
|
79
|
+
@connection.delete url do |req|
|
|
80
|
+
req.headers.merge! @provider.headers
|
|
81
|
+
yield req if block_given?
|
|
82
|
+
end
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
private
|
|
87
|
+
|
|
88
|
+
def instrument_request(method, url)
|
|
89
|
+
payload = {
|
|
90
|
+
provider: @provider.slug,
|
|
91
|
+
method: method,
|
|
92
|
+
url: url
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
RubyLLM.instrument('request.ruby_llm', payload, config: @config) do |event|
|
|
96
|
+
response = yield
|
|
97
|
+
event[:status] = response.status if response.respond_to?(:status)
|
|
98
|
+
response
|
|
99
|
+
end
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
def setup_timeout(faraday)
|
|
103
|
+
faraday.options.timeout = @config.request_timeout
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
def setup_logging(faraday)
|
|
107
|
+
faraday.response :logger,
|
|
108
|
+
RubyLLM.logger,
|
|
109
|
+
bodies: RubyLLM.logger.debug?,
|
|
110
|
+
errors: true,
|
|
111
|
+
headers: false,
|
|
112
|
+
log_level: :debug do |logger|
|
|
113
|
+
logger.filter(logging_regexp('[A-Za-z0-9+/=]{100,}'), '[BASE64 DATA]')
|
|
114
|
+
logger.filter(logging_regexp('[-\\d.e,\\s]{100,}'), '[EMBEDDINGS ARRAY]')
|
|
115
|
+
end
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
def logging_regexp(pattern)
|
|
119
|
+
return Regexp.new(pattern) if @config.log_regexp_timeout.nil? || !Regexp.respond_to?(:timeout)
|
|
120
|
+
|
|
121
|
+
Regexp.new(pattern, timeout: @config.log_regexp_timeout)
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
def setup_retry(faraday)
|
|
125
|
+
faraday.request :retry, {
|
|
126
|
+
max: @config.max_retries,
|
|
127
|
+
interval: @config.retry_interval,
|
|
128
|
+
max_interval: @config.retry_max_interval,
|
|
129
|
+
interval_randomness: @config.retry_interval_randomness,
|
|
130
|
+
backoff_factor: @config.retry_backoff_factor,
|
|
131
|
+
methods: Faraday::Retry::Middleware::IDEMPOTENT_METHODS,
|
|
132
|
+
retry_if: lambda { |env, _exception|
|
|
133
|
+
env[:method] == :post && idempotent?(env) && !stream_delivered?(env)
|
|
134
|
+
},
|
|
135
|
+
exceptions: retry_exceptions
|
|
136
|
+
}
|
|
137
|
+
faraday.use :llm_usage
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
def stream_delivered?(env)
|
|
141
|
+
env[:request]&.context&.dig(STREAM_PROGRESS_KEY, :started)
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
def idempotent?(env)
|
|
145
|
+
env[:request]&.context&.dig(IDEMPOTENT_KEY) != false
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
def setup_middleware(faraday)
|
|
149
|
+
faraday.request :multipart
|
|
150
|
+
faraday.request :json
|
|
151
|
+
faraday.response :json
|
|
152
|
+
faraday.adapter(@config.faraday_adapter)
|
|
153
|
+
faraday.use :llm_errors, provider: @provider
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
def setup_http_proxy(faraday)
|
|
157
|
+
return unless @config.http_proxy
|
|
158
|
+
|
|
159
|
+
faraday.proxy = @config.http_proxy
|
|
160
|
+
end
|
|
161
|
+
|
|
162
|
+
def retry_exceptions
|
|
163
|
+
[
|
|
164
|
+
Errno::ETIMEDOUT,
|
|
165
|
+
Timeout::Error,
|
|
166
|
+
Faraday::TimeoutError,
|
|
167
|
+
Faraday::ConnectionFailed,
|
|
168
|
+
Faraday::RetriableResponse,
|
|
169
|
+
RubyLLM::RateLimitError,
|
|
170
|
+
RubyLLM::ServerError,
|
|
171
|
+
RubyLLM::ServiceUnavailableError,
|
|
172
|
+
RubyLLM::OverloadedError
|
|
173
|
+
]
|
|
174
|
+
end
|
|
175
|
+
|
|
176
|
+
def set_usage_tracker(request, tracker)
|
|
177
|
+
context = request.options.context ||= {}
|
|
178
|
+
context[UsageMiddleware::CONTEXT_KEY] = tracker
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
# A request that creates server-side state cannot be replayed: a retry
|
|
182
|
+
# after a lost response submits the job a second time.
|
|
183
|
+
def mark_non_idempotent(request)
|
|
184
|
+
context = request.options.context ||= {}
|
|
185
|
+
context[IDEMPOTENT_KEY] = false
|
|
186
|
+
end
|
|
187
|
+
|
|
188
|
+
def inspect_attributes # :nodoc:
|
|
189
|
+
{ provider: @provider.slug }
|
|
190
|
+
end
|
|
191
|
+
end
|
|
192
|
+
end
|
|
193
|
+
end
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'faraday'
|
|
4
|
+
require 'ruby_llm/error'
|
|
5
|
+
|
|
6
|
+
module RubyLLM
|
|
7
|
+
module Transport # :nodoc:
|
|
8
|
+
class ErrorMiddleware < Faraday::Middleware # :nodoc: all
|
|
9
|
+
def initialize(app, options = {})
|
|
10
|
+
super(app)
|
|
11
|
+
@provider = options[:provider]
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
# Sits directly above the adapter, inside the retry middleware, so this
|
|
15
|
+
# runs once per attempt: streaming state stored on the env by a previous
|
|
16
|
+
# attempt must not leak into the next one.
|
|
17
|
+
def call(env)
|
|
18
|
+
env[:streaming_error_response] = nil
|
|
19
|
+
env[:streaming_state] = nil
|
|
20
|
+
@app.call(env).on_complete do |response|
|
|
21
|
+
apply_retry_delay(response)
|
|
22
|
+
self.class.parse_error(provider: @provider, response: streaming_error_response(response))
|
|
23
|
+
end
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
private
|
|
27
|
+
|
|
28
|
+
# The retry middleware only reads the standard Retry-After header, so
|
|
29
|
+
# provider-specific rate-limit headers are normalized into it here,
|
|
30
|
+
# where the provider is known.
|
|
31
|
+
def apply_retry_delay(response)
|
|
32
|
+
status = response.respond_to?(:status) ? response.status : response[:status]
|
|
33
|
+
return unless status == 429
|
|
34
|
+
|
|
35
|
+
headers = response[:response_headers]
|
|
36
|
+
if @provider && !headers['Retry-After'] && (delay = @provider.retry_delay(response))
|
|
37
|
+
headers['Retry-After'] = delay.to_s
|
|
38
|
+
end
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
def streaming_error_response(response)
|
|
42
|
+
stored_response = if response.respond_to?(:env) && response.env.respond_to?(:[])
|
|
43
|
+
response.env[:streaming_error_response]
|
|
44
|
+
elsif response.respond_to?(:[])
|
|
45
|
+
response[:streaming_error_response]
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
stored_response || response
|
|
49
|
+
rescue NameError
|
|
50
|
+
response
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
class << self
|
|
54
|
+
CONTEXT_LENGTH_PATTERNS = [
|
|
55
|
+
/context length/i,
|
|
56
|
+
/context window/i,
|
|
57
|
+
/exceeds?.*context size/i,
|
|
58
|
+
/maximum context/i,
|
|
59
|
+
/request too large/i,
|
|
60
|
+
/too many tokens/i,
|
|
61
|
+
/token count exceeds/i,
|
|
62
|
+
/input[_\s-]?token/i,
|
|
63
|
+
/input or output tokens? must be reduced/i,
|
|
64
|
+
/reduce the length of messages/i,
|
|
65
|
+
/prompt is too long/i,
|
|
66
|
+
/context limit/i
|
|
67
|
+
].freeze
|
|
68
|
+
|
|
69
|
+
RATE_LIMIT_PATTERNS = [
|
|
70
|
+
/rate limit/i,
|
|
71
|
+
/per minute/i,
|
|
72
|
+
/per hour/i,
|
|
73
|
+
/per day/i
|
|
74
|
+
].freeze
|
|
75
|
+
|
|
76
|
+
def parse_error(provider:, response:)
|
|
77
|
+
message = provider&.parse_error(response)
|
|
78
|
+
|
|
79
|
+
case response.status
|
|
80
|
+
when 200..399
|
|
81
|
+
message
|
|
82
|
+
when 400
|
|
83
|
+
raise ContextLengthExceededError.new(message, response:) if context_length_exceeded?(message)
|
|
84
|
+
|
|
85
|
+
raise BadRequestError.new(message, response:)
|
|
86
|
+
when 401
|
|
87
|
+
raise UnauthorizedError.new(message, response:)
|
|
88
|
+
when 402
|
|
89
|
+
raise PaymentRequiredError.new(message, response:)
|
|
90
|
+
when 403
|
|
91
|
+
raise ForbiddenError.new(message, response:)
|
|
92
|
+
when 429
|
|
93
|
+
raise RateLimitError.new(message, response:) if rate_limited?(message)
|
|
94
|
+
raise ContextLengthExceededError.new(message, response:) if context_length_exceeded?(message)
|
|
95
|
+
|
|
96
|
+
raise RateLimitError.new(message, response:)
|
|
97
|
+
when 500
|
|
98
|
+
raise ServerError.new(message, response:)
|
|
99
|
+
when 502..504
|
|
100
|
+
raise ServiceUnavailableError.new(message, response:)
|
|
101
|
+
when 529
|
|
102
|
+
raise OverloadedError.new(message, response:)
|
|
103
|
+
else
|
|
104
|
+
raise Error.new(message, response:)
|
|
105
|
+
end
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
private
|
|
109
|
+
|
|
110
|
+
# Providers hand back whatever their error body holds, which is not
|
|
111
|
+
# always a String: bedrock-mantle nests code, message, and type in a
|
|
112
|
+
# Hash. Match on the rendered text so any shape classifies.
|
|
113
|
+
def context_length_exceeded?(message)
|
|
114
|
+
text = message.to_s
|
|
115
|
+
return false if text.empty?
|
|
116
|
+
|
|
117
|
+
CONTEXT_LENGTH_PATTERNS.any? { |pattern| text.match?(pattern) }
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
def rate_limited?(message)
|
|
121
|
+
text = message.to_s
|
|
122
|
+
return false if text.empty?
|
|
123
|
+
|
|
124
|
+
RATE_LIMIT_PATTERNS.any? { |pattern| text.match?(pattern) }
|
|
125
|
+
end
|
|
126
|
+
end
|
|
127
|
+
end
|
|
128
|
+
end
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
Faraday::Middleware.register_middleware(llm_errors: RubyLLM::Transport::ErrorMiddleware)
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'faraday'
|
|
4
|
+
|
|
5
|
+
module RubyLLM
|
|
6
|
+
module Transport # :nodoc:
|
|
7
|
+
# Sits inside Faraday retry middleware so every transport attempt produces
|
|
8
|
+
# one usage observation.
|
|
9
|
+
class UsageMiddleware < Faraday::Middleware # :nodoc: all
|
|
10
|
+
CONTEXT_KEY = :ruby_llm_usage_tracker
|
|
11
|
+
|
|
12
|
+
def call(env)
|
|
13
|
+
tracker = env.request.context&.[](CONTEXT_KEY)
|
|
14
|
+
return @app.call(env) unless tracker
|
|
15
|
+
|
|
16
|
+
entry = tracker.start
|
|
17
|
+
begin
|
|
18
|
+
@app.call(env)
|
|
19
|
+
rescue StandardError => e
|
|
20
|
+
tracker.fail_attempt(entry, e)
|
|
21
|
+
raise
|
|
22
|
+
end
|
|
23
|
+
end
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
Faraday::Middleware.register_middleware(llm_usage: RubyLLM::Transport::UsageMiddleware)
|