ruby_llm 1.16.0 → 2.0.0.rc2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.rdoc_options +25 -0
- data/README.md +85 -32
- data/exe/ruby_llm +8 -0
- data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
- data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
- data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
- data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
- data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
- data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
- data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
- data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
- data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
- data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
- data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
- data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
- data/lib/generators/ruby_llm/provider/cli.rb +175 -0
- data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
- data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
- data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
- data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
- data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
- data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
- data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
- data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
- data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
- data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
- data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
- data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
- data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
- data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
- data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
- data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
- data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
- data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
- data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
- data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
- data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
- data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
- data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
- data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
- data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
- data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
- data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
- data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
- data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
- data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
- data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
- data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
- data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
- data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
- data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
- data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
- data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
- data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
- data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
- data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
- data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
- data/lib/ruby_llm/accounting/usage.rb +245 -0
- data/lib/ruby_llm/active_record/acts_as.rb +94 -111
- data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
- data/lib/ruby_llm/active_record/batch.rb +97 -0
- data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
- data/lib/ruby_llm/active_record/message_methods.rb +113 -136
- data/lib/ruby_llm/active_record/model.rb +135 -0
- data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
- data/lib/ruby_llm/active_record/tool_call.rb +33 -0
- data/lib/ruby_llm/active_record/usage.rb +61 -0
- data/lib/ruby_llm/agent.rb +1065 -151
- data/lib/ruby_llm/aliases.json +268 -100
- data/lib/ruby_llm/attachment.rb +187 -48
- data/lib/ruby_llm/batch.rb +432 -0
- data/lib/ruby_llm/cached_content.rb +112 -0
- data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
- data/lib/ruby_llm/chat.rb +1127 -198
- data/lib/ruby_llm/chunk.rb +10 -0
- data/lib/ruby_llm/citation.rb +105 -0
- data/lib/ruby_llm/configuration.rb +261 -24
- data/lib/ruby_llm/context.rb +128 -6
- data/lib/ruby_llm/cost.rb +217 -80
- data/lib/ruby_llm/downloaded_file.rb +33 -0
- data/lib/ruby_llm/embedding.rb +121 -17
- data/lib/ruby_llm/embedding_request.rb +53 -0
- data/lib/ruby_llm/error.rb +159 -23
- data/lib/ruby_llm/fallback.rb +133 -0
- data/lib/ruby_llm/files/mime_type.rb +97 -0
- data/lib/ruby_llm/image.rb +153 -32
- data/lib/ruby_llm/message.rb +233 -54
- data/lib/ruby_llm/model/modalities.rb +17 -4
- data/lib/ruby_llm/model/pricing.rb +24 -5
- data/lib/ruby_llm/model/pricing_category.rb +103 -14
- data/lib/ruby_llm/model/pricing_tier.rb +55 -15
- data/lib/ruby_llm/model.rb +244 -2
- data/lib/ruby_llm/models/aliases.rb +41 -0
- data/lib/ruby_llm/models/registry.rb +165 -0
- data/lib/ruby_llm/models/schema.rb +99 -0
- data/lib/ruby_llm/models.json +73261 -33733
- data/lib/ruby_llm/models.rb +477 -215
- data/lib/ruby_llm/moderation.rb +139 -26
- data/lib/ruby_llm/ocr.rb +112 -0
- data/lib/ruby_llm/prompt.rb +79 -0
- data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
- data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
- data/lib/ruby_llm/protocol/streaming.rb +230 -0
- data/lib/ruby_llm/protocol.rb +662 -0
- data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
- data/lib/ruby_llm/protocols/anthropic/chat.rb +548 -0
- data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
- data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
- data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
- data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
- data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
- data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
- data/lib/ruby_llm/protocols/anthropic.rb +101 -0
- data/lib/ruby_llm/protocols/azure/files.rb +16 -0
- data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
- data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
- data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
- data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
- data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
- data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
- data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
- data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
- data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
- data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
- data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
- data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
- data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
- data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
- data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
- data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
- data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
- data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
- data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
- data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
- data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
- data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
- data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
- data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
- data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
- data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
- data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
- data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
- data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
- data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
- data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
- data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
- data/lib/ruby_llm/protocols/cohere.rb +21 -0
- data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
- data/lib/ruby_llm/protocols/converse/chat.rb +695 -0
- data/lib/ruby_llm/protocols/converse/media.rb +178 -0
- data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
- data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
- data/lib/ruby_llm/protocols/converse.rb +54 -0
- data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
- data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
- data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
- data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
- data/lib/ruby_llm/protocols/deepgram.rb +19 -0
- data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
- data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
- data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
- data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
- data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
- data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
- data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
- data/lib/ruby_llm/protocols/files.rb +119 -0
- data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
- data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
- data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
- data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
- data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
- data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
- data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
- data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
- data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
- data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
- data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
- data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
- data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
- data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
- data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
- data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
- data/lib/ruby_llm/protocols/gemini.rb +35 -0
- data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
- data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
- data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
- data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
- data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
- data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
- data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
- data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
- data/lib/ruby_llm/protocols/interactions.rb +29 -0
- data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
- data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
- data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
- data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
- data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
- data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
- data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
- data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
- data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
- data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
- data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
- data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
- data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
- data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
- data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
- data/lib/ruby_llm/protocols/openai/files.rb +42 -0
- data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
- data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
- data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
- data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
- data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
- data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
- data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
- data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
- data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
- data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
- data/lib/ruby_llm/protocols/responses/media.rb +61 -0
- data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
- data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
- data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
- data/lib/ruby_llm/protocols/responses.rb +35 -0
- data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
- data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
- data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
- data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
- data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
- data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
- data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
- data/lib/ruby_llm/protocols/xai/files.rb +30 -0
- data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
- data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
- data/lib/ruby_llm/provider.rb +560 -128
- data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
- data/lib/ruby_llm/providers/anthropic.rb +4 -6
- data/lib/ruby_llm/providers/azure/audio.rb +18 -0
- data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
- data/lib/ruby_llm/providers/azure/chat.rb +2 -9
- data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
- data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
- data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
- data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
- data/lib/ruby_llm/providers/azure/images.rb +22 -0
- data/lib/ruby_llm/providers/azure/media.rb +5 -14
- data/lib/ruby_llm/providers/azure/models.rb +35 -0
- data/lib/ruby_llm/providers/azure/responses.rb +26 -0
- data/lib/ruby_llm/providers/azure/videos.rb +64 -0
- data/lib/ruby_llm/providers/azure.rb +77 -78
- data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
- data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
- data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
- data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
- data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
- data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
- data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
- data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
- data/lib/ruby_llm/providers/bedrock.rb +216 -45
- data/lib/ruby_llm/providers/cohere.rb +31 -0
- data/lib/ruby_llm/providers/deepgram.rb +37 -0
- data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
- data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
- data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
- data/lib/ruby_llm/providers/deepseek.rb +9 -2
- data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
- data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
- data/lib/ruby_llm/providers/gemini.rb +15 -8
- data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
- data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
- data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
- data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
- data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
- data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
- data/lib/ruby_llm/providers/gpustack.rb +35 -11
- data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
- data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
- data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
- data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
- data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
- data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
- data/lib/ruby_llm/providers/mistral/media.rb +6 -18
- data/lib/ruby_llm/providers/mistral/models.rb +57 -23
- data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
- data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
- data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
- data/lib/ruby_llm/providers/mistral.rb +16 -4
- data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
- data/lib/ruby_llm/providers/ollama/media.rb +6 -15
- data/lib/ruby_llm/providers/ollama/models.rb +50 -9
- data/lib/ruby_llm/providers/ollama.rb +9 -8
- data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
- data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
- data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
- data/lib/ruby_llm/providers/openai/models.rb +23 -23
- data/lib/ruby_llm/providers/openai/responses.rb +13 -0
- data/lib/ruby_llm/providers/openai.rb +92 -11
- data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
- data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
- data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
- data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
- data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
- data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
- data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
- data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
- data/lib/ruby_llm/providers/openrouter.rb +78 -20
- data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
- data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
- data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
- data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
- data/lib/ruby_llm/providers/perplexity.rb +28 -20
- data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
- data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
- data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
- data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
- data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
- data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
- data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
- data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
- data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
- data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
- data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
- data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
- data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
- data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
- data/lib/ruby_llm/providers/vertexai.rb +160 -17
- data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
- data/lib/ruby_llm/providers/xai/chat.rb +3 -2
- data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
- data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
- data/lib/ruby_llm/providers/xai/images.rb +91 -0
- data/lib/ruby_llm/providers/xai/models.rb +32 -36
- data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
- data/lib/ruby_llm/providers/xai/responses.rb +52 -0
- data/lib/ruby_llm/providers/xai/speech.rb +45 -0
- data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
- data/lib/ruby_llm/providers/xai/videos.rb +87 -0
- data/lib/ruby_llm/providers/xai.rb +15 -5
- data/lib/ruby_llm/railtie.rb +7 -16
- data/lib/ruby_llm/rerank.rb +105 -0
- data/lib/ruby_llm/research_job.rb +241 -0
- data/lib/ruby_llm/search_results.rb +68 -0
- data/lib/ruby_llm/server_tool_call.rb +73 -0
- data/lib/ruby_llm/speech.rb +159 -0
- data/lib/ruby_llm/speech_chunk.rb +33 -0
- data/lib/ruby_llm/support/deprecator.rb +22 -0
- data/lib/ruby_llm/support/inspectable.rb +49 -0
- data/lib/ruby_llm/support/instrumentation.rb +41 -0
- data/lib/ruby_llm/support/utils.rb +147 -0
- data/lib/ruby_llm/thinking.rb +127 -20
- data/lib/ruby_llm/tokenization.rb +59 -0
- data/lib/ruby_llm/tokens.rb +103 -33
- data/lib/ruby_llm/tool.rb +266 -91
- data/lib/ruby_llm/tool_call.rb +36 -3
- data/lib/ruby_llm/tools/server_tools.rb +109 -0
- data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
- data/lib/ruby_llm/transcription.rb +138 -13
- data/lib/ruby_llm/transcription_chunk.rb +68 -0
- data/lib/ruby_llm/transport/connection.rb +193 -0
- data/lib/ruby_llm/transport/error_middleware.rb +131 -0
- data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
- data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
- data/lib/ruby_llm/uploaded_file.rb +144 -0
- data/lib/ruby_llm/version.rb +2 -1
- data/lib/ruby_llm/video.rb +136 -0
- data/lib/ruby_llm/video_job.rb +150 -0
- data/lib/ruby_llm/workflow.rb +91 -0
- data/lib/ruby_llm.rb +380 -6
- data/lib/tasks/ruby_llm.rake +21 -16
- data/skills/rubyllm/SKILL.md +81 -0
- data/skills/rubyllm/agents/openai.yaml +4 -0
- metadata +339 -97
- data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
- data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
- data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
- data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
- data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
- data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
- data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
- data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
- data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
- data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
- data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
- data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
- data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
- data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
- data/lib/ruby_llm/active_record/model_methods.rb +0 -82
- data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
- data/lib/ruby_llm/aliases.rb +0 -41
- data/lib/ruby_llm/connection.rb +0 -159
- data/lib/ruby_llm/content.rb +0 -91
- data/lib/ruby_llm/deprecator.rb +0 -24
- data/lib/ruby_llm/error_middleware.rb +0 -81
- data/lib/ruby_llm/instrumentation.rb +0 -36
- data/lib/ruby_llm/mime_type.rb +0 -96
- data/lib/ruby_llm/model/info.rb +0 -164
- data/lib/ruby_llm/model_registry.rb +0 -39
- data/lib/ruby_llm/models_schema.json +0 -171
- data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
- data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
- data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
- data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
- data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
- data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
- data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
- data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
- data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
- data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
- data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
- data/lib/ruby_llm/providers/gemini/images.rb +0 -47
- data/lib/ruby_llm/providers/gemini/models.rb +0 -38
- data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
- data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
- data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
- data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
- data/lib/ruby_llm/providers/openai/chat.rb +0 -236
- data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
- data/lib/ruby_llm/providers/openai/images.rb +0 -90
- data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
- data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
- data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
- data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
- data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
- data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
- data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
- data/lib/ruby_llm/stream_accumulator.rb +0 -218
- data/lib/ruby_llm/streaming.rb +0 -179
- data/lib/ruby_llm/tool_concurrency.rb +0 -105
- data/lib/ruby_llm/utils.rb +0 -130
- data/lib/tasks/models.rake +0 -593
- data/lib/tasks/release.rake +0 -94
- data/lib/tasks/vcr.rake +0 -124
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Providers
|
|
5
|
+
class VertexAI < Provider
|
|
6
|
+
# The Anthropic protocol over Vertex AI rawPredict endpoints.
|
|
7
|
+
class Anthropic < Protocols::Anthropic
|
|
8
|
+
ANTHROPIC_VERSION = 'vertex-2023-10-16'
|
|
9
|
+
|
|
10
|
+
def completion_url
|
|
11
|
+
"#{@provider.model_path(@model.id, publisher: 'anthropic')}:rawPredict"
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
def stream_url
|
|
15
|
+
"#{@provider.model_path(@model.id, publisher: 'anthropic')}:streamRawPredict"
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
def render_payload(messages, **)
|
|
19
|
+
payload = super
|
|
20
|
+
payload.delete(:model)
|
|
21
|
+
payload.merge(anthropic_version: ANTHROPIC_VERSION)
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
def count_tokens(*, **)
|
|
25
|
+
raise Error, "#{@provider.name} doesn't support token counting for Claude models"
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def supports_provider_file_references?
|
|
29
|
+
false
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
end
|
|
34
|
+
end
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Providers
|
|
5
|
+
class VertexAI
|
|
6
|
+
# Feature capability gaps not represented in upstream model catalogs.
|
|
7
|
+
module Capabilities
|
|
8
|
+
def self.augment(capabilities, model_id:, modalities:)
|
|
9
|
+
return capabilities if model_id.include?('embedding')
|
|
10
|
+
|
|
11
|
+
additions = []
|
|
12
|
+
additions << 'tool_choice' if model_id == 'gemini-2.5-flash'
|
|
13
|
+
additions << 'transcription' if modalities[:input].include?('audio') && modalities[:output].include?('text')
|
|
14
|
+
capabilities | additions
|
|
15
|
+
end
|
|
16
|
+
end
|
|
17
|
+
end
|
|
18
|
+
end
|
|
19
|
+
end
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Providers
|
|
5
|
+
class VertexAI < Provider
|
|
6
|
+
class ChatCompletions
|
|
7
|
+
# Vertex AI MaaS batch prediction rows using OpenAI JSONL shape.
|
|
8
|
+
module Batches
|
|
9
|
+
include Protocols::VertexAI::BatchPrediction
|
|
10
|
+
|
|
11
|
+
private
|
|
12
|
+
|
|
13
|
+
def vertex_batch_request(request)
|
|
14
|
+
{
|
|
15
|
+
custom_id: request[:custom_id],
|
|
16
|
+
method: 'POST',
|
|
17
|
+
url: '/v1/chat/completions',
|
|
18
|
+
body: batch_payload(request)
|
|
19
|
+
}
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def validate_batch_requests!(requests)
|
|
23
|
+
return if requests.all? { |request| chat_completion_payload?(request.fetch(:payload)) }
|
|
24
|
+
|
|
25
|
+
raise Error, 'vertexai MaaS batch requests require chat completion payloads'
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def chat_completion_payload?(payload)
|
|
29
|
+
payload.key?(:messages) || payload.key?('messages')
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
def vertex_batch_model_path(model)
|
|
33
|
+
publisher, name = model.split('/', 2)
|
|
34
|
+
raise Error, 'vertexai MaaS batch requests require publisher/model ids' unless publisher && name
|
|
35
|
+
|
|
36
|
+
@provider.model_path(name, publisher:)
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def parse_vertex_batch_result(line, fallback_index)
|
|
40
|
+
index = vertex_batch_result_index(line, fallback_index)
|
|
41
|
+
response = line['response']
|
|
42
|
+
body = response.is_a?(Hash) ? response['body'] || response : response
|
|
43
|
+
|
|
44
|
+
if body
|
|
45
|
+
[index, parse_completion_body(body, raw: body)]
|
|
46
|
+
else
|
|
47
|
+
[index, nil, batch_failure(index, line.dig('status', 'message') || batch_error_message(line))]
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
end
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Providers
|
|
5
|
+
class VertexAI < Provider
|
|
6
|
+
# MaaS models (publisher-prefixed ids like meta/llama-3.3-70b-instruct-maas)
|
|
7
|
+
# speak Chat Completions through Vertex AI's OpenAI-compatible endpoint.
|
|
8
|
+
class ChatCompletions < Protocols::ChatCompletions
|
|
9
|
+
def completion_url
|
|
10
|
+
"#{@provider.location_path}/endpoints/openapi/chat/completions"
|
|
11
|
+
end
|
|
12
|
+
end
|
|
13
|
+
end
|
|
14
|
+
end
|
|
15
|
+
end
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Providers
|
|
5
|
+
class VertexAI
|
|
6
|
+
# Gemini Embedding 2 over Vertex AI's single-content embedding endpoint.
|
|
7
|
+
class EmbedContent < Protocols::Gemini
|
|
8
|
+
def embedding_url(model:)
|
|
9
|
+
"#{@provider.model_path(model)}:embedContent"
|
|
10
|
+
end
|
|
11
|
+
|
|
12
|
+
def render_embedding(text, **options)
|
|
13
|
+
return super unless text.is_a?(Array)
|
|
14
|
+
|
|
15
|
+
{ requests: text.map { |value| render_embedding_payload(value, **options) } }
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
def render_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil, with: [],
|
|
19
|
+
provider_options: {})
|
|
20
|
+
if text.is_a?(Array) && text.size != 1
|
|
21
|
+
raise ArgumentError, 'Vertex AI embedContent accepts one text at a time'
|
|
22
|
+
end
|
|
23
|
+
raise ArgumentError, "#{model} takes task instructions and titles in the text" if task_type || title
|
|
24
|
+
|
|
25
|
+
payload = {
|
|
26
|
+
content: { parts: Protocols::Gemini::Media.format_content(text.is_a?(Array) ? text.first : text, with) },
|
|
27
|
+
outputDimensionality: dimensions
|
|
28
|
+
}.compact
|
|
29
|
+
Support::Utils.deep_merge(payload, provider_options)
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
def parse_embedding_response(response, model:, text:)
|
|
33
|
+
vectors = response.body.dig('embedding', 'values')
|
|
34
|
+
raise Error.new('Vertex AI returned no embedding', response:) if vectors.nil? || vectors.empty?
|
|
35
|
+
|
|
36
|
+
vectors = [vectors] if text.is_a?(Array)
|
|
37
|
+
Embedding.new(vectors:, model:, input_tokens: response.body.dig('usageMetadata', 'promptTokenCount'))
|
|
38
|
+
end
|
|
39
|
+
end
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
end
|
|
@@ -8,23 +8,38 @@ module RubyLLM
|
|
|
8
8
|
module_function
|
|
9
9
|
|
|
10
10
|
def embedding_url(model:)
|
|
11
|
-
"
|
|
11
|
+
"#{@provider.model_path(model)}:predict"
|
|
12
12
|
end
|
|
13
13
|
|
|
14
|
-
def
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
14
|
+
def supports_embedding_media?
|
|
15
|
+
false
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
# rubocop:disable-next Lint/UnusedMethodArgument
|
|
19
|
+
def render_embedding_payload(text, model:, dimensions:, task_type: nil, title: nil, provider_options: {})
|
|
20
|
+
instances = [text].flatten.map do |t|
|
|
21
|
+
{ content: t.to_s, task_type: task_type, title: title }.compact
|
|
19
22
|
end
|
|
23
|
+
payload = { instances: instances }
|
|
24
|
+
payload[:parameters] = { outputDimensionality: dimensions } if dimensions
|
|
25
|
+
|
|
26
|
+
Support::Utils.deep_merge(payload, provider_options)
|
|
20
27
|
end
|
|
21
28
|
|
|
22
29
|
def parse_embedding_response(response, model:, text:)
|
|
23
30
|
predictions = response.body['predictions']
|
|
24
31
|
vectors = predictions&.map { |p| p.dig('embeddings', 'values') }
|
|
32
|
+
input_tokens = embedding_input_tokens(predictions)
|
|
25
33
|
vectors = vectors.first if vectors&.length == 1 && !text.is_a?(Array)
|
|
26
34
|
|
|
27
|
-
Embedding.new(vectors:, model:, input_tokens:
|
|
35
|
+
Embedding.new(vectors:, model:, input_tokens:)
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
def embedding_input_tokens(predictions)
|
|
39
|
+
counts = Array(predictions).filter_map do |prediction|
|
|
40
|
+
prediction.dig('embeddings', 'statistics', 'token_count')
|
|
41
|
+
end
|
|
42
|
+
counts.sum unless counts.empty?
|
|
28
43
|
end
|
|
29
44
|
end
|
|
30
45
|
end
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Providers
|
|
5
|
+
class VertexAI < Provider
|
|
6
|
+
class Gemini
|
|
7
|
+
# Vertex AI Gemini batch prediction rows.
|
|
8
|
+
module Batches
|
|
9
|
+
include Protocols::VertexAI::BatchPrediction
|
|
10
|
+
|
|
11
|
+
private
|
|
12
|
+
|
|
13
|
+
def vertex_batch_request(request)
|
|
14
|
+
payload = RubyLLM::Support::Utils.deep_stringify_keys(batch_payload(request))
|
|
15
|
+
labels = payload.fetch('labels', {}).merge('ruby_llm_batch_id' => request[:custom_id])
|
|
16
|
+
{ request: payload.merge('labels' => labels) }
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def parse_vertex_batch_result(line, fallback_index)
|
|
20
|
+
index = vertex_batch_result_index(line, fallback_index)
|
|
21
|
+
|
|
22
|
+
if line['response']
|
|
23
|
+
body = line['response']
|
|
24
|
+
[index, parse_completion_body(body, raw: body)]
|
|
25
|
+
else
|
|
26
|
+
[index, nil, batch_failure(index, vertex_batch_status_message(line))]
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
# Gemini prediction rows carry status as a plain string, empty on
|
|
31
|
+
# success and the error text on failure.
|
|
32
|
+
def vertex_batch_status_message(line)
|
|
33
|
+
status = line['status']
|
|
34
|
+
message = status.is_a?(Hash) ? status['message'] : status
|
|
35
|
+
|
|
36
|
+
message.to_s.empty? ? batch_error_message(line) : message
|
|
37
|
+
end
|
|
38
|
+
end
|
|
39
|
+
end
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
end
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Providers
|
|
5
|
+
class VertexAI < Provider
|
|
6
|
+
# The Gemini protocol over Vertex AI endpoints.
|
|
7
|
+
class Gemini < Protocols::Gemini
|
|
8
|
+
include VertexAI::Embeddings
|
|
9
|
+
include VertexAI::Models
|
|
10
|
+
include VertexAI::Videos
|
|
11
|
+
|
|
12
|
+
SERVER_TOOL_ALIASES = Protocols::Gemini::SERVER_TOOL_ALIASES.merge(
|
|
13
|
+
file_search: lambda { |options|
|
|
14
|
+
{ tool: { retrieval: { vertexAiSearch: Support::Utils.deep_symbolize_keys(options) } } }
|
|
15
|
+
}
|
|
16
|
+
).freeze
|
|
17
|
+
|
|
18
|
+
def server_tool_aliases
|
|
19
|
+
SERVER_TOOL_ALIASES
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def completion_url
|
|
23
|
+
"#{@provider.model_path(@model.id)}:generateContent"
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def stream_url
|
|
27
|
+
"#{@provider.model_path(@model.id)}:streamGenerateContent?alt=sse"
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
def count_tokens_url
|
|
31
|
+
"#{@provider.model_path(@model.id)}:countTokens"
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def render_count_tokens_payload(messages, **options)
|
|
35
|
+
count_tokens_request(messages, **options)
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
def caches_url
|
|
39
|
+
"#{@provider.location_path}/cachedContents"
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
def cache_name(name)
|
|
43
|
+
name = name.name if name.is_a?(CachedContent)
|
|
44
|
+
name.to_s.include?('/') ? name.to_s : "#{caches_url}/#{name}"
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
def cache_model_name(model_id)
|
|
48
|
+
@provider.model_path(model_id)
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def images_url(with: nil, mask: nil) # rubocop:disable Lint/UnusedMethodArgument
|
|
52
|
+
id = model_id(@model)
|
|
53
|
+
|
|
54
|
+
"#{@provider.model_path(id)}:#{image_endpoint_action(id)}"
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
def speech_url(model:)
|
|
58
|
+
"#{@provider.model_path(model)}:generateContent"
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
private
|
|
62
|
+
|
|
63
|
+
def transcription_url(model)
|
|
64
|
+
"#{@provider.model_path(model)}:generateContent"
|
|
65
|
+
end
|
|
66
|
+
end
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
end
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Providers
|
|
5
|
+
class VertexAI
|
|
6
|
+
class LiveTranscription < Protocols::Gemini::LiveTranscription # :nodoc: all
|
|
7
|
+
def validate_transcription_request(...)
|
|
8
|
+
super
|
|
9
|
+
return if @config.vertexai_location == 'global'
|
|
10
|
+
|
|
11
|
+
raise ArgumentError, 'Vertex AI Live transcription requires vertexai_location = "global"'
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
def transcription_model_name(model)
|
|
15
|
+
@provider.model_path(model)
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
def websocket_service
|
|
19
|
+
'google.cloud.aiplatform.v1beta1.LlmBidiService/BidiGenerateContent'
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
end
|
|
23
|
+
end
|
|
24
|
+
end
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Providers
|
|
5
|
+
class VertexAI < Provider
|
|
6
|
+
# Mistral's models speak their own dialect of Chat Completions over Vertex AI
|
|
7
|
+
# rawPredict endpoints. We reuse Mistral's dialect wholesale and only swap the URLs.
|
|
8
|
+
class Mistral < Protocols::ChatCompletions
|
|
9
|
+
include Providers::Mistral::Chat
|
|
10
|
+
include Providers::Mistral::Embeddings
|
|
11
|
+
include Providers::Mistral::Media
|
|
12
|
+
include Providers::Mistral::Models
|
|
13
|
+
|
|
14
|
+
# The Mistral model families Vertex AI serves directly. Shared by the
|
|
15
|
+
# registry (which models to list) and protocol_for (where to route them).
|
|
16
|
+
MODELS = /\A(mistral|ministral|codestral|magistral|mathstral|pixtral|devstral|voxtral)/
|
|
17
|
+
|
|
18
|
+
def completion_url
|
|
19
|
+
"#{@provider.model_path(@model.id, publisher: 'mistralai')}:rawPredict"
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def stream_url
|
|
23
|
+
"#{@provider.model_path(@model.id, publisher: 'mistralai')}:streamRawPredict"
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
end
|
|
27
|
+
end
|
|
28
|
+
end
|
|
@@ -5,75 +5,171 @@ module RubyLLM
|
|
|
5
5
|
class VertexAI
|
|
6
6
|
# Models methods for the Vertex AI integration
|
|
7
7
|
module Models
|
|
8
|
-
|
|
8
|
+
def self.models_dev_alias(model_id, models_dev_by_key, _provider_model = nil)
|
|
9
|
+
source = models_dev_by_key["gemini:#{model_id}"]
|
|
10
|
+
Model.new(source.to_h.merge(provider: 'vertexai')) if source
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
# Google models the publisher catalog omits in some regions while still
|
|
14
|
+
# serving them there. Every id must be callable in at least one region;
|
|
15
|
+
# ids that answer nowhere do not belong here.
|
|
9
16
|
KNOWN_GOOGLE_MODELS = %w[
|
|
10
17
|
gemini-2.5-flash-lite
|
|
11
18
|
gemini-2.5-pro
|
|
12
19
|
gemini-2.5-flash
|
|
13
20
|
gemini-2.0-flash-lite-001
|
|
14
21
|
gemini-2.0-flash-001
|
|
15
|
-
gemini-2.0-flash
|
|
16
|
-
gemini-2.0-flash-exp
|
|
17
22
|
gemini-1.5-pro-002
|
|
18
23
|
gemini-1.5-pro
|
|
19
|
-
gemini-1.5-flash-002
|
|
20
|
-
gemini-1.5-flash
|
|
21
|
-
gemini-1.5-flash-8b
|
|
22
24
|
gemini-pro
|
|
23
25
|
gemini-pro-vision
|
|
24
|
-
gemini-exp-1206
|
|
25
|
-
gemini-exp-1121
|
|
26
26
|
gemini-embedding-001
|
|
27
27
|
text-embedding-005
|
|
28
28
|
text-embedding-004
|
|
29
29
|
text-multilingual-embedding-002
|
|
30
30
|
].freeze
|
|
31
31
|
|
|
32
|
+
# Every publisher with models Vertex AI serves as a service. The rest
|
|
33
|
+
# of the Model Garden is deploy-it-yourself and not callable directly.
|
|
34
|
+
PUBLISHERS = %w[google anthropic mistralai meta deepseek-ai qwen openai moonshotai zai-org].freeze
|
|
35
|
+
|
|
36
|
+
# Vertex AI serves a different slice of the catalog in each location and
|
|
37
|
+
# neither of these is a superset of the other, so a listing unions them.
|
|
38
|
+
CATALOG_LOCATIONS = %w[global us-central1].freeze
|
|
39
|
+
|
|
40
|
+
# The configured location has to answer: without it we would report a
|
|
41
|
+
# catalog the caller cannot reach. A supplementary location is a bonus,
|
|
42
|
+
# so a failure there is a warning and the rest of the union stands.
|
|
32
43
|
def list_models
|
|
33
|
-
|
|
34
|
-
|
|
44
|
+
fetched = []
|
|
45
|
+
counts = {}
|
|
46
|
+
|
|
47
|
+
catalog_connections.each do |location, connection|
|
|
48
|
+
models = location_models(connection)
|
|
49
|
+
counts[location] = models.size
|
|
50
|
+
fetched.concat(models)
|
|
51
|
+
rescue StandardError => e
|
|
52
|
+
raise if location == configured_location
|
|
53
|
+
|
|
54
|
+
RubyLLM.logger.warn "Skipping the Vertex AI catalog at #{location}: #{e.class}: #{e.message}"
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
models = fetched.uniq(&:id)
|
|
58
|
+
log_catalog(counts, models)
|
|
59
|
+
|
|
60
|
+
models + build_known_models(models.map(&:id))
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
private
|
|
64
|
+
|
|
65
|
+
def configured_location
|
|
66
|
+
@config.vertexai_location.to_s
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
# The location lives in the host, so locations sharing one api base,
|
|
70
|
+
# as they do behind a custom vertexai_api_base, are one catalog.
|
|
71
|
+
def catalog_connections
|
|
72
|
+
[configured_location, *CATALOG_LOCATIONS]
|
|
73
|
+
.uniq { |location| @provider.api_base_for(location) }
|
|
74
|
+
.to_h { |location| [location, connection_for(location)] }
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
def connection_for(location)
|
|
78
|
+
return @connection if location == configured_location
|
|
79
|
+
|
|
80
|
+
Transport::Connection.new(@provider, @config, api_base: @provider.api_base_for(location))
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
def log_catalog(counts, models)
|
|
84
|
+
per_location = counts.map { |location, count| "#{location} (#{count})" }.join(', ')
|
|
85
|
+
RubyLLM.logger.info "Fetched the Vertex AI catalog from #{per_location}: #{models.size} models"
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
# A publisher with nothing to offer in a region answers 200 with an
|
|
89
|
+
# empty list, so any error here is infrastructure, not an empty
|
|
90
|
+
# catalog. Reporting a partial catalog as a success would drop the
|
|
91
|
+
# missing publishers from the registry.
|
|
92
|
+
def location_models(connection)
|
|
93
|
+
failures = []
|
|
94
|
+
models = PUBLISHERS.flat_map do |publisher|
|
|
95
|
+
publisher_models(publisher, connection)
|
|
96
|
+
rescue StandardError => e
|
|
97
|
+
failures << "#{publisher} (#{e.class}: #{e.message})"
|
|
98
|
+
[]
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
raise Error, "Could not fetch the Vertex AI catalog for #{failures.join(', ')}" if failures.any?
|
|
102
|
+
|
|
103
|
+
models
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
def publisher_models(publisher, connection)
|
|
107
|
+
catalog(publisher, connection).filter_map { |model_data| build_publisher_model(publisher, model_data) }
|
|
108
|
+
end
|
|
35
109
|
|
|
36
|
-
|
|
110
|
+
# MaaS models are called as publisher/name through the OpenAI-compatible
|
|
111
|
+
# endpoint; directly served models by their bare catalog name.
|
|
112
|
+
def build_publisher_model(publisher, model_data)
|
|
113
|
+
name = model_data['name'].split('/').last
|
|
114
|
+
return if deployable?(model_data)
|
|
115
|
+
|
|
116
|
+
if name.end_with?('-maas')
|
|
117
|
+
build_model_from_api_data(model_data, "#{publisher}/#{name}")
|
|
118
|
+
elsif served_directly?(publisher, name)
|
|
119
|
+
build_model_from_api_data(model_data, name)
|
|
120
|
+
end
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
# Deploy-it-yourself Model Garden cards expose deploy actions; managed
|
|
124
|
+
# services Vertex AI serves on our behalf never do.
|
|
125
|
+
def deployable?(model_data)
|
|
126
|
+
actions = model_data['supportedActions'] || {}
|
|
127
|
+
actions.key?('deploy') || actions.key?('multiDeployVertex') || actions.key?('deployGke')
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
# Among the managed models, which publishers we route by bare name, and
|
|
131
|
+
# for Google (whose catalog is a grab-bag of vision, media, and AutoML
|
|
132
|
+
# products) which of those names are chat or embedding models.
|
|
133
|
+
def served_directly?(publisher, name)
|
|
134
|
+
case publisher
|
|
135
|
+
when 'google' then name.match?(/\Agemini|embedding/)
|
|
136
|
+
when 'anthropic' then true
|
|
137
|
+
when 'mistralai' then VertexAI::Mistral::MODELS.match?(name)
|
|
138
|
+
else false
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
def catalog(publisher, connection)
|
|
143
|
+
models = []
|
|
144
|
+
page_token = nil
|
|
37
145
|
|
|
38
146
|
loop do
|
|
39
|
-
response =
|
|
147
|
+
response = connection.get("publishers/#{publisher}/models") do |req|
|
|
40
148
|
req.headers['x-goog-user-project'] = @config.vertexai_project_id
|
|
41
149
|
req.params = { pageSize: 100 }
|
|
42
150
|
req.params[:pageToken] = page_token if page_token
|
|
43
151
|
end
|
|
44
152
|
|
|
45
|
-
|
|
46
|
-
publisher_models.each do |model_data|
|
|
47
|
-
next if model_data['launchStage'] == 'DEPRECATED'
|
|
48
|
-
|
|
49
|
-
model_id = extract_model_id_from_path(model_data['name'])
|
|
50
|
-
all_models << build_model_from_api_data(model_data, model_id)
|
|
51
|
-
end
|
|
52
|
-
|
|
153
|
+
models.concat(response.body['publisherModels'] || [])
|
|
53
154
|
page_token = response.body['nextPageToken']
|
|
54
155
|
break unless page_token
|
|
55
156
|
end
|
|
56
157
|
|
|
57
|
-
|
|
58
|
-
rescue StandardError => e
|
|
59
|
-
RubyLLM.logger.debug { "Error fetching Vertex AI models: #{e.message}" }
|
|
60
|
-
build_known_models
|
|
158
|
+
models.reject { |model_data| model_data['launchStage'] == 'DEPRECATED' }
|
|
61
159
|
end
|
|
62
160
|
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
KNOWN_GOOGLE_MODELS.map do |model_id|
|
|
67
|
-
Model::Info.new(
|
|
161
|
+
def build_known_models(fetched_ids)
|
|
162
|
+
(KNOWN_GOOGLE_MODELS - fetched_ids).map do |model_id|
|
|
163
|
+
Model.new(
|
|
68
164
|
id: model_id,
|
|
69
165
|
name: model_id,
|
|
70
|
-
provider: slug,
|
|
166
|
+
provider: @provider.slug,
|
|
71
167
|
family: determine_model_family(model_id),
|
|
72
168
|
created_at: nil,
|
|
73
169
|
context_window: nil,
|
|
74
170
|
max_output_tokens: nil,
|
|
75
171
|
modalities: nil,
|
|
76
|
-
capabilities:
|
|
172
|
+
capabilities: extract_capabilities(model_id),
|
|
77
173
|
pricing: nil,
|
|
78
174
|
metadata: {
|
|
79
175
|
source: 'known_models'
|
|
@@ -83,16 +179,16 @@ module RubyLLM
|
|
|
83
179
|
end
|
|
84
180
|
|
|
85
181
|
def build_model_from_api_data(model_data, model_id)
|
|
86
|
-
Model
|
|
182
|
+
Model.new(
|
|
87
183
|
id: model_id,
|
|
88
184
|
name: model_id,
|
|
89
|
-
provider: slug,
|
|
185
|
+
provider: @provider.slug,
|
|
90
186
|
family: determine_model_family(model_id),
|
|
91
187
|
created_at: nil,
|
|
92
188
|
context_window: nil,
|
|
93
189
|
max_output_tokens: nil,
|
|
94
190
|
modalities: nil,
|
|
95
|
-
capabilities: extract_capabilities(model_data),
|
|
191
|
+
capabilities: extract_capabilities(model_data['name']),
|
|
96
192
|
pricing: nil,
|
|
97
193
|
metadata: {
|
|
98
194
|
version_id: model_data['versionId'],
|
|
@@ -104,12 +200,21 @@ module RubyLLM
|
|
|
104
200
|
)
|
|
105
201
|
end
|
|
106
202
|
|
|
107
|
-
def extract_model_id_from_path(path)
|
|
108
|
-
path.split('/').last
|
|
109
|
-
end
|
|
110
|
-
|
|
111
203
|
def determine_model_family(model_id)
|
|
112
204
|
case model_id
|
|
205
|
+
when /^claude.*haiku/ then 'claude-haiku'
|
|
206
|
+
when /^claude.*sonnet/ then 'claude-sonnet'
|
|
207
|
+
when /^claude.*opus/ then 'claude-opus'
|
|
208
|
+
when /^claude/ then 'claude'
|
|
209
|
+
when %r{^meta/} then 'llama'
|
|
210
|
+
when %r{^deepseek-ai/} then 'deepseek'
|
|
211
|
+
when %r{^qwen/} then 'qwen'
|
|
212
|
+
when %r{^moonshotai/} then 'kimi'
|
|
213
|
+
when %r{^zai-org/} then 'glm'
|
|
214
|
+
when %r{^openai/} then 'gpt-oss'
|
|
215
|
+
when %r{^google/} then 'gemma'
|
|
216
|
+
when /^codestral/ then 'codestral'
|
|
217
|
+
when /^mi(ni)?stral/ then 'mistral'
|
|
113
218
|
when /^gemini-2\.\d+/ then 'gemini-2'
|
|
114
219
|
when /^gemini-1\.\d+/ then 'gemini-1.5'
|
|
115
220
|
when /^text-embedding/ then 'text-embedding'
|
|
@@ -118,11 +223,8 @@ module RubyLLM
|
|
|
118
223
|
end
|
|
119
224
|
end
|
|
120
225
|
|
|
121
|
-
def extract_capabilities(
|
|
122
|
-
|
|
123
|
-
model_name = model_data['name']
|
|
124
|
-
capabilities << 'function_calling' if model_name.include?('gemini')
|
|
125
|
-
capabilities.uniq
|
|
226
|
+
def extract_capabilities(name)
|
|
227
|
+
name.match?(/ocr|embedding/) ? %w[streaming] : %w[streaming function_calling]
|
|
126
228
|
end
|
|
127
229
|
end
|
|
128
230
|
end
|