ruby_llm 1.16.0 → 2.0.0.rc2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.rdoc_options +25 -0
- data/README.md +85 -32
- data/exe/ruby_llm +8 -0
- data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
- data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
- data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
- data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
- data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
- data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
- data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
- data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
- data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
- data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
- data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
- data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
- data/lib/generators/ruby_llm/provider/cli.rb +175 -0
- data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
- data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
- data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
- data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
- data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
- data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
- data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
- data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
- data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
- data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
- data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
- data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
- data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
- data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
- data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
- data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
- data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
- data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
- data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
- data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
- data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
- data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
- data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
- data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
- data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
- data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
- data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
- data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
- data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
- data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
- data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
- data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
- data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
- data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
- data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
- data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
- data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
- data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
- data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
- data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
- data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
- data/lib/ruby_llm/accounting/usage.rb +245 -0
- data/lib/ruby_llm/active_record/acts_as.rb +94 -111
- data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
- data/lib/ruby_llm/active_record/batch.rb +97 -0
- data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
- data/lib/ruby_llm/active_record/message_methods.rb +113 -136
- data/lib/ruby_llm/active_record/model.rb +135 -0
- data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
- data/lib/ruby_llm/active_record/tool_call.rb +33 -0
- data/lib/ruby_llm/active_record/usage.rb +61 -0
- data/lib/ruby_llm/agent.rb +1065 -151
- data/lib/ruby_llm/aliases.json +268 -100
- data/lib/ruby_llm/attachment.rb +187 -48
- data/lib/ruby_llm/batch.rb +432 -0
- data/lib/ruby_llm/cached_content.rb +112 -0
- data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
- data/lib/ruby_llm/chat.rb +1127 -198
- data/lib/ruby_llm/chunk.rb +10 -0
- data/lib/ruby_llm/citation.rb +105 -0
- data/lib/ruby_llm/configuration.rb +261 -24
- data/lib/ruby_llm/context.rb +128 -6
- data/lib/ruby_llm/cost.rb +217 -80
- data/lib/ruby_llm/downloaded_file.rb +33 -0
- data/lib/ruby_llm/embedding.rb +121 -17
- data/lib/ruby_llm/embedding_request.rb +53 -0
- data/lib/ruby_llm/error.rb +159 -23
- data/lib/ruby_llm/fallback.rb +133 -0
- data/lib/ruby_llm/files/mime_type.rb +97 -0
- data/lib/ruby_llm/image.rb +153 -32
- data/lib/ruby_llm/message.rb +233 -54
- data/lib/ruby_llm/model/modalities.rb +17 -4
- data/lib/ruby_llm/model/pricing.rb +24 -5
- data/lib/ruby_llm/model/pricing_category.rb +103 -14
- data/lib/ruby_llm/model/pricing_tier.rb +55 -15
- data/lib/ruby_llm/model.rb +244 -2
- data/lib/ruby_llm/models/aliases.rb +41 -0
- data/lib/ruby_llm/models/registry.rb +165 -0
- data/lib/ruby_llm/models/schema.rb +99 -0
- data/lib/ruby_llm/models.json +73261 -33733
- data/lib/ruby_llm/models.rb +477 -215
- data/lib/ruby_llm/moderation.rb +139 -26
- data/lib/ruby_llm/ocr.rb +112 -0
- data/lib/ruby_llm/prompt.rb +79 -0
- data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
- data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
- data/lib/ruby_llm/protocol/streaming.rb +230 -0
- data/lib/ruby_llm/protocol.rb +662 -0
- data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
- data/lib/ruby_llm/protocols/anthropic/chat.rb +548 -0
- data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
- data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
- data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
- data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
- data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
- data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
- data/lib/ruby_llm/protocols/anthropic.rb +101 -0
- data/lib/ruby_llm/protocols/azure/files.rb +16 -0
- data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
- data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
- data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
- data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
- data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
- data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
- data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
- data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
- data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
- data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
- data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
- data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
- data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
- data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
- data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
- data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
- data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
- data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
- data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
- data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
- data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
- data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
- data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
- data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
- data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
- data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
- data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
- data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
- data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
- data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
- data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
- data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
- data/lib/ruby_llm/protocols/cohere.rb +21 -0
- data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
- data/lib/ruby_llm/protocols/converse/chat.rb +695 -0
- data/lib/ruby_llm/protocols/converse/media.rb +178 -0
- data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
- data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
- data/lib/ruby_llm/protocols/converse.rb +54 -0
- data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
- data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
- data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
- data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
- data/lib/ruby_llm/protocols/deepgram.rb +19 -0
- data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
- data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
- data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
- data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
- data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
- data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
- data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
- data/lib/ruby_llm/protocols/files.rb +119 -0
- data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
- data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
- data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
- data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
- data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
- data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
- data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
- data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
- data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
- data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
- data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
- data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
- data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
- data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
- data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
- data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
- data/lib/ruby_llm/protocols/gemini.rb +35 -0
- data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
- data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
- data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
- data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
- data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
- data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
- data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
- data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
- data/lib/ruby_llm/protocols/interactions.rb +29 -0
- data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
- data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
- data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
- data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
- data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
- data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
- data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
- data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
- data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
- data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
- data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
- data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
- data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
- data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
- data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
- data/lib/ruby_llm/protocols/openai/files.rb +42 -0
- data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
- data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
- data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
- data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
- data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
- data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
- data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
- data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
- data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
- data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
- data/lib/ruby_llm/protocols/responses/media.rb +61 -0
- data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
- data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
- data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
- data/lib/ruby_llm/protocols/responses.rb +35 -0
- data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
- data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
- data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
- data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
- data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
- data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
- data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
- data/lib/ruby_llm/protocols/xai/files.rb +30 -0
- data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
- data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
- data/lib/ruby_llm/provider.rb +560 -128
- data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
- data/lib/ruby_llm/providers/anthropic.rb +4 -6
- data/lib/ruby_llm/providers/azure/audio.rb +18 -0
- data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
- data/lib/ruby_llm/providers/azure/chat.rb +2 -9
- data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
- data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
- data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
- data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
- data/lib/ruby_llm/providers/azure/images.rb +22 -0
- data/lib/ruby_llm/providers/azure/media.rb +5 -14
- data/lib/ruby_llm/providers/azure/models.rb +35 -0
- data/lib/ruby_llm/providers/azure/responses.rb +26 -0
- data/lib/ruby_llm/providers/azure/videos.rb +64 -0
- data/lib/ruby_llm/providers/azure.rb +77 -78
- data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
- data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
- data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
- data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
- data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
- data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
- data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
- data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
- data/lib/ruby_llm/providers/bedrock.rb +216 -45
- data/lib/ruby_llm/providers/cohere.rb +31 -0
- data/lib/ruby_llm/providers/deepgram.rb +37 -0
- data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
- data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
- data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
- data/lib/ruby_llm/providers/deepseek.rb +9 -2
- data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
- data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
- data/lib/ruby_llm/providers/gemini.rb +15 -8
- data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
- data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
- data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
- data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
- data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
- data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
- data/lib/ruby_llm/providers/gpustack.rb +35 -11
- data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
- data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
- data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
- data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
- data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
- data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
- data/lib/ruby_llm/providers/mistral/media.rb +6 -18
- data/lib/ruby_llm/providers/mistral/models.rb +57 -23
- data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
- data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
- data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
- data/lib/ruby_llm/providers/mistral.rb +16 -4
- data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
- data/lib/ruby_llm/providers/ollama/media.rb +6 -15
- data/lib/ruby_llm/providers/ollama/models.rb +50 -9
- data/lib/ruby_llm/providers/ollama.rb +9 -8
- data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
- data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
- data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
- data/lib/ruby_llm/providers/openai/models.rb +23 -23
- data/lib/ruby_llm/providers/openai/responses.rb +13 -0
- data/lib/ruby_llm/providers/openai.rb +92 -11
- data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
- data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
- data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
- data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
- data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
- data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
- data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
- data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
- data/lib/ruby_llm/providers/openrouter.rb +78 -20
- data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
- data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
- data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
- data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
- data/lib/ruby_llm/providers/perplexity.rb +28 -20
- data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
- data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
- data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
- data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
- data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
- data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
- data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
- data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
- data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
- data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
- data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
- data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
- data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
- data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
- data/lib/ruby_llm/providers/vertexai.rb +160 -17
- data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
- data/lib/ruby_llm/providers/xai/chat.rb +3 -2
- data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
- data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
- data/lib/ruby_llm/providers/xai/images.rb +91 -0
- data/lib/ruby_llm/providers/xai/models.rb +32 -36
- data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
- data/lib/ruby_llm/providers/xai/responses.rb +52 -0
- data/lib/ruby_llm/providers/xai/speech.rb +45 -0
- data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
- data/lib/ruby_llm/providers/xai/videos.rb +87 -0
- data/lib/ruby_llm/providers/xai.rb +15 -5
- data/lib/ruby_llm/railtie.rb +7 -16
- data/lib/ruby_llm/rerank.rb +105 -0
- data/lib/ruby_llm/research_job.rb +241 -0
- data/lib/ruby_llm/search_results.rb +68 -0
- data/lib/ruby_llm/server_tool_call.rb +73 -0
- data/lib/ruby_llm/speech.rb +159 -0
- data/lib/ruby_llm/speech_chunk.rb +33 -0
- data/lib/ruby_llm/support/deprecator.rb +22 -0
- data/lib/ruby_llm/support/inspectable.rb +49 -0
- data/lib/ruby_llm/support/instrumentation.rb +41 -0
- data/lib/ruby_llm/support/utils.rb +147 -0
- data/lib/ruby_llm/thinking.rb +127 -20
- data/lib/ruby_llm/tokenization.rb +59 -0
- data/lib/ruby_llm/tokens.rb +103 -33
- data/lib/ruby_llm/tool.rb +266 -91
- data/lib/ruby_llm/tool_call.rb +36 -3
- data/lib/ruby_llm/tools/server_tools.rb +109 -0
- data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
- data/lib/ruby_llm/transcription.rb +138 -13
- data/lib/ruby_llm/transcription_chunk.rb +68 -0
- data/lib/ruby_llm/transport/connection.rb +193 -0
- data/lib/ruby_llm/transport/error_middleware.rb +131 -0
- data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
- data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
- data/lib/ruby_llm/uploaded_file.rb +144 -0
- data/lib/ruby_llm/version.rb +2 -1
- data/lib/ruby_llm/video.rb +136 -0
- data/lib/ruby_llm/video_job.rb +150 -0
- data/lib/ruby_llm/workflow.rb +91 -0
- data/lib/ruby_llm.rb +380 -6
- data/lib/tasks/ruby_llm.rake +21 -16
- data/skills/rubyllm/SKILL.md +81 -0
- data/skills/rubyllm/agents/openai.yaml +4 -0
- metadata +339 -97
- data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
- data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
- data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
- data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
- data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
- data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
- data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
- data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
- data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
- data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
- data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
- data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
- data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
- data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
- data/lib/ruby_llm/active_record/model_methods.rb +0 -82
- data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
- data/lib/ruby_llm/aliases.rb +0 -41
- data/lib/ruby_llm/connection.rb +0 -159
- data/lib/ruby_llm/content.rb +0 -91
- data/lib/ruby_llm/deprecator.rb +0 -24
- data/lib/ruby_llm/error_middleware.rb +0 -81
- data/lib/ruby_llm/instrumentation.rb +0 -36
- data/lib/ruby_llm/mime_type.rb +0 -96
- data/lib/ruby_llm/model/info.rb +0 -164
- data/lib/ruby_llm/model_registry.rb +0 -39
- data/lib/ruby_llm/models_schema.json +0 -171
- data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
- data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
- data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
- data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
- data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
- data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
- data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
- data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
- data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
- data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
- data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
- data/lib/ruby_llm/providers/gemini/images.rb +0 -47
- data/lib/ruby_llm/providers/gemini/models.rb +0 -38
- data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
- data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
- data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
- data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
- data/lib/ruby_llm/providers/openai/chat.rb +0 -236
- data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
- data/lib/ruby_llm/providers/openai/images.rb +0 -90
- data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
- data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
- data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
- data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
- data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
- data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
- data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
- data/lib/ruby_llm/stream_accumulator.rb +0 -218
- data/lib/ruby_llm/streaming.rb +0 -179
- data/lib/ruby_llm/tool_concurrency.rb +0 -105
- data/lib/ruby_llm/utils.rb +0 -130
- data/lib/tasks/models.rake +0 -593
- data/lib/tasks/release.rake +0 -94
- data/lib/tasks/vcr.rake +0 -124
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Protocols
|
|
5
|
+
class Responses
|
|
6
|
+
# The Responses input-token counting endpoint.
|
|
7
|
+
module TokenCounting
|
|
8
|
+
COUNT_TOKENS_KEYS = %i[model input instructions tools tool_choice parallel_tool_calls reasoning text].freeze
|
|
9
|
+
|
|
10
|
+
module_function
|
|
11
|
+
|
|
12
|
+
def count_tokens_url
|
|
13
|
+
"#{completion_url}/input_tokens"
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
def render_count_tokens_payload(messages, model:, **options)
|
|
17
|
+
render_payload(messages, model: model, temperature: nil, **options).slice(*COUNT_TOKENS_KEYS)
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
def parse_count_tokens_response(response)
|
|
21
|
+
response.body.fetch('input_tokens')
|
|
22
|
+
end
|
|
23
|
+
end
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
end
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Protocols
|
|
5
|
+
class Responses
|
|
6
|
+
# Tools methods of the OpenAI Responses API. Function definitions are
|
|
7
|
+
# flat rather than nested under a `function` key.
|
|
8
|
+
module Tools
|
|
9
|
+
module_function
|
|
10
|
+
|
|
11
|
+
def tool_for(tool)
|
|
12
|
+
definition = {
|
|
13
|
+
type: 'function',
|
|
14
|
+
name: tool.name,
|
|
15
|
+
description: tool.description,
|
|
16
|
+
parameters: ChatCompletions::Tools.parameters_schema_for(tool),
|
|
17
|
+
strict: false
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
return definition if tool.provider_options.empty?
|
|
21
|
+
|
|
22
|
+
RubyLLM::Support::Utils.deep_merge(definition, tool.provider_options)
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def build_tool_choice(tool_choice)
|
|
26
|
+
case tool_choice
|
|
27
|
+
when :auto, :none, :required
|
|
28
|
+
tool_choice
|
|
29
|
+
else
|
|
30
|
+
{
|
|
31
|
+
type: 'function',
|
|
32
|
+
name: tool_choice
|
|
33
|
+
}
|
|
34
|
+
end
|
|
35
|
+
end
|
|
36
|
+
end
|
|
37
|
+
end
|
|
38
|
+
end
|
|
39
|
+
end
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Protocols
|
|
5
|
+
# The OpenAI Responses API. Overrides the chat surface of Chat Completions;
|
|
6
|
+
# embeddings, images, moderation, and transcription are inherited. Runs
|
|
7
|
+
# stateless (store: false) and replays encrypted reasoning so multi-turn
|
|
8
|
+
# tool calls work without server-side state.
|
|
9
|
+
class Responses < ChatCompletions
|
|
10
|
+
include Responses::Approvals
|
|
11
|
+
include Responses::Chat
|
|
12
|
+
include Responses::Media
|
|
13
|
+
include Responses::Streaming
|
|
14
|
+
include Responses::Tools
|
|
15
|
+
|
|
16
|
+
SERVER_TOOL_ALIASES = {
|
|
17
|
+
web_search: { tool: { type: 'web_search' } },
|
|
18
|
+
file_search: { tool: { type: 'file_search' } },
|
|
19
|
+
code_execution: { tool: { type: 'code_interpreter', container: { type: 'auto' } } },
|
|
20
|
+
code_interpreter: { tool: { type: 'code_interpreter', container: { type: 'auto' } } },
|
|
21
|
+
image_generation: { tool: { type: 'image_generation' } },
|
|
22
|
+
mcp: lambda do |options|
|
|
23
|
+
options = Support::Utils.deep_symbolize_keys(options)
|
|
24
|
+
options[:server_url] = options.delete(:url) if options.key?(:url)
|
|
25
|
+
options[:server_label] = options.delete(:name) if options.key?(:name)
|
|
26
|
+
{ tool: { type: 'mcp' }.merge(options) }
|
|
27
|
+
end
|
|
28
|
+
}.freeze
|
|
29
|
+
|
|
30
|
+
def server_tool_aliases
|
|
31
|
+
SERVER_TOOL_ALIASES
|
|
32
|
+
end
|
|
33
|
+
end
|
|
34
|
+
end
|
|
35
|
+
end
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'stringio'
|
|
4
|
+
|
|
5
|
+
module RubyLLM
|
|
6
|
+
module Protocols
|
|
7
|
+
module VertexAI
|
|
8
|
+
# Shared Vertex AI batchPredictionJobs plumbing. The input row and output
|
|
9
|
+
# result shapes belong to each Vertex protocol.
|
|
10
|
+
module BatchPrediction
|
|
11
|
+
include RubyLLM::Batch::Helpers
|
|
12
|
+
|
|
13
|
+
TERMINAL = %w[
|
|
14
|
+
JOB_STATE_SUCCEEDED
|
|
15
|
+
JOB_STATE_FAILED
|
|
16
|
+
JOB_STATE_CANCELLED
|
|
17
|
+
JOB_STATE_EXPIRED
|
|
18
|
+
JOB_STATE_PARTIALLY_SUCCEEDED
|
|
19
|
+
].freeze
|
|
20
|
+
private_constant :TERMINAL
|
|
21
|
+
|
|
22
|
+
def create_batch(requests)
|
|
23
|
+
model = single_batch_model!(requests, 'vertexai')
|
|
24
|
+
validate_batch_requests!(requests)
|
|
25
|
+
input_uri, output_uri = vertex_batch_storage_uris
|
|
26
|
+
@provider.upload_file(
|
|
27
|
+
StringIO.new(vertex_batch_jsonl(requests)),
|
|
28
|
+
filename: 'input.jsonl',
|
|
29
|
+
uri: input_uri,
|
|
30
|
+
content_type: 'application/jsonl'
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
response = @connection.post("#{@provider.location_path}/batchPredictionJobs",
|
|
34
|
+
vertex_batch_job(model, input_uri, output_uri), idempotent: false)
|
|
35
|
+
|
|
36
|
+
parse_batch_response(response.body)
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def find_batch(id)
|
|
40
|
+
parse_batch_response @connection.get(vertex_batch_name(id)).body
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
def cancel_batch(id)
|
|
44
|
+
@connection.post("#{vertex_batch_name(id)}:cancel", {})
|
|
45
|
+
find_batch(id)
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def batch_results(id)
|
|
49
|
+
job = @connection.get(vertex_batch_name(id)).body
|
|
50
|
+
output_uri = vertex_output_uri(job)
|
|
51
|
+
unless output_uri
|
|
52
|
+
status = parse_batch_status(job['state'], completed: TERMINAL.include?(job['state']))
|
|
53
|
+
return [] if %i[failed cancelled].include?(status)
|
|
54
|
+
|
|
55
|
+
raise Error, 'vertexai batch has no GCS output URI yet'
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
rows = @provider.list_file_uris(output_uri).grep(/\.jsonl\z/).flat_map do |uri|
|
|
59
|
+
@provider.download_file(uri).to_s.each_line.filter_map do |line|
|
|
60
|
+
next if line.strip.empty?
|
|
61
|
+
|
|
62
|
+
JSON.parse(line)
|
|
63
|
+
end
|
|
64
|
+
end
|
|
65
|
+
parse_vertex_batch_results(rows, job:)
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
private
|
|
69
|
+
|
|
70
|
+
def vertex_batch_job(model, input_uri, output_uri)
|
|
71
|
+
{
|
|
72
|
+
displayName: "ruby_llm_#{SecureRandom.hex(8)}",
|
|
73
|
+
model: vertex_batch_model_path(model),
|
|
74
|
+
inputConfig: {
|
|
75
|
+
instancesFormat: 'jsonl',
|
|
76
|
+
gcsSource: { uris: [input_uri] }
|
|
77
|
+
},
|
|
78
|
+
outputConfig: {
|
|
79
|
+
predictionsFormat: 'jsonl',
|
|
80
|
+
gcsDestination: { outputUriPrefix: output_uri }
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
def vertex_batch_jsonl(requests)
|
|
86
|
+
requests.map { |request| JSON.generate(vertex_batch_request(request)) }.join("\n")
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
def vertex_batch_request(_request)
|
|
90
|
+
raise NotImplementedError
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
def validate_batch_requests!(_requests); end
|
|
94
|
+
|
|
95
|
+
def vertex_batch_model_path(model)
|
|
96
|
+
@provider.model_path(model)
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
def vertex_batch_storage_uris
|
|
100
|
+
base = @config.vertexai_batch_gcs_uri.to_s.sub(%r{/+\z}, '')
|
|
101
|
+
if base.empty?
|
|
102
|
+
raise ConfigurationError, 'Set vertexai_batch_gcs_uri to a gs:// bucket prefix for Vertex AI batches'
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
prefix = "#{base}/ruby_llm_batches/#{SecureRandom.hex(8)}"
|
|
106
|
+
["#{prefix}/input.jsonl", "#{prefix}/output"]
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def vertex_batch_name(id)
|
|
110
|
+
id.to_s.start_with?('projects/') ? id : "#{@provider.location_path}/batchPredictionJobs/#{id}"
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
def parse_batch_response(data)
|
|
114
|
+
state = data['state']
|
|
115
|
+
request_counts = data['completionStats']
|
|
116
|
+
|
|
117
|
+
{
|
|
118
|
+
id: data['name'],
|
|
119
|
+
raw_status: state,
|
|
120
|
+
completed: TERMINAL.include?(state),
|
|
121
|
+
request_counts:,
|
|
122
|
+
request_count: data.dig('labels', 'ruby_llm_request_count')&.to_i || request_counts&.values&.sum(&:to_i),
|
|
123
|
+
model: data['model']
|
|
124
|
+
}
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
def parse_batch_status(raw_status, completed:)
|
|
128
|
+
return :pending unless completed
|
|
129
|
+
return :succeeded if %w[JOB_STATE_SUCCEEDED JOB_STATE_PARTIALLY_SUCCEEDED].include?(raw_status)
|
|
130
|
+
return :cancelled if raw_status == 'JOB_STATE_CANCELLED'
|
|
131
|
+
|
|
132
|
+
:failed
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
def vertex_batch_result_index(line, fallback_index)
|
|
136
|
+
key = line['custom_id'] || line.dig('request', 'labels', 'ruby_llm_batch_id')
|
|
137
|
+
key ? batch_result_index(key) : fallback_index
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
def parse_vertex_batch_results(rows, **)
|
|
141
|
+
rows.each_with_index.filter_map { |line, index| parse_vertex_batch_result(line, index) }
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
def parse_vertex_batch_result(_line, _fallback_index)
|
|
145
|
+
raise NotImplementedError
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
def vertex_output_uri(job)
|
|
149
|
+
job.dig('outputInfo', 'gcsOutputDirectory') ||
|
|
150
|
+
job.dig('outputConfig', 'gcsDestination', 'outputUriPrefix')
|
|
151
|
+
end
|
|
152
|
+
end
|
|
153
|
+
end
|
|
154
|
+
end
|
|
155
|
+
end
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Protocols
|
|
5
|
+
module VertexAI
|
|
6
|
+
class EmbeddingPrediction
|
|
7
|
+
# Input identity and model-specific batch embedding request formats.
|
|
8
|
+
module Requests
|
|
9
|
+
private
|
|
10
|
+
|
|
11
|
+
def validate_embedding_request!(request)
|
|
12
|
+
text = request.fetch(:text)
|
|
13
|
+
values = text.is_a?(Array) ? text : [text]
|
|
14
|
+
unless values.any? && values.all? { |value| value.is_a?(String) && !value.empty? }
|
|
15
|
+
raise ArgumentError, 'Vertex AI embedding batches require nonempty text strings'
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
payload = request.fetch(:payload)
|
|
19
|
+
validate_embedding_payload_keys!(payload)
|
|
20
|
+
inputs = embedding_batch_inputs(payload)
|
|
21
|
+
validate_gemini_embedding_options!(payload, inputs) if GEMINI_MODELS.include?(request.fetch(:model))
|
|
22
|
+
return if inputs.size == values.size
|
|
23
|
+
|
|
24
|
+
raise ArgumentError, 'Vertex AI embedding batch payload does not match the number of texts'
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def validate_embedding_payload_keys!(payload)
|
|
28
|
+
supported = if payload.key?(:instances)
|
|
29
|
+
%i[instances parameters]
|
|
30
|
+
elsif payload.key?(:requests)
|
|
31
|
+
[:requests]
|
|
32
|
+
else
|
|
33
|
+
%i[content outputDimensionality taskType title model]
|
|
34
|
+
end
|
|
35
|
+
unknown = payload.keys - supported
|
|
36
|
+
return if unknown.empty?
|
|
37
|
+
|
|
38
|
+
raise ArgumentError, "Unsupported Vertex AI embedding batch options: #{unknown.join(', ')}"
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
def validate_gemini_embedding_options!(payload, inputs)
|
|
42
|
+
parameters = payload.fetch(:parameters, {}).keys - [:outputDimensionality]
|
|
43
|
+
supported = %i[content task_type taskType title outputDimensionality model]
|
|
44
|
+
unsupported = inputs.flat_map(&:keys).uniq - supported
|
|
45
|
+
return if parameters.empty? && unsupported.empty?
|
|
46
|
+
|
|
47
|
+
raise ArgumentError, 'Vertex AI Gemini embedding batches do not support these options: ' \
|
|
48
|
+
"#{(parameters + unsupported).join(', ')}"
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def embedding_batch_inputs(payload)
|
|
52
|
+
return payload.fetch(:instances) if payload.key?(:instances)
|
|
53
|
+
return payload.fetch(:requests) if payload.key?(:requests)
|
|
54
|
+
return [payload] if payload.key?(:content)
|
|
55
|
+
|
|
56
|
+
raise ArgumentError, 'Vertex AI embedding batches require embedding request payloads'
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def render_embedding_batch_rows(request)
|
|
60
|
+
inputs = embedding_batch_inputs(request.fetch(:payload))
|
|
61
|
+
array = request.fetch(:text).is_a?(Array)
|
|
62
|
+
inputs.each_with_index.map do |input, index|
|
|
63
|
+
key = "rllm-#{request.fetch(:custom_id)}-#{index}-#{inputs.size}-#{array ? 'a' : 's'}"
|
|
64
|
+
if LEGACY_MODELS.include?(request.fetch(:model))
|
|
65
|
+
input.merge(key:)
|
|
66
|
+
else
|
|
67
|
+
{ key:, request: render_gemini_embedding_request(input, request.fetch(:payload)) }
|
|
68
|
+
end
|
|
69
|
+
end
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def render_gemini_embedding_request(input, payload)
|
|
73
|
+
content = input[:content]
|
|
74
|
+
content = { parts: [{ text: content }] } if content.is_a?(String)
|
|
75
|
+
config = {
|
|
76
|
+
output_dimensionality: input[:outputDimensionality] || payload.dig(:parameters, :outputDimensionality),
|
|
77
|
+
task_type: input[:task_type] || input[:taskType], title: input[:title]
|
|
78
|
+
}.compact
|
|
79
|
+
{ content:, embed_content_config: config }
|
|
80
|
+
end
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
end
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Protocols
|
|
5
|
+
module VertexAI
|
|
6
|
+
class EmbeddingPrediction
|
|
7
|
+
# Regroups provider output rows into the submitted embedding requests.
|
|
8
|
+
module Results
|
|
9
|
+
private
|
|
10
|
+
|
|
11
|
+
def embedding_row_metadata(row)
|
|
12
|
+
key = row['key'] || row.dig('instance', 'key')
|
|
13
|
+
match = /\Arllm-(\d+)-(\d+)-(\d+)-([as])\z/.match(key.to_s)
|
|
14
|
+
raise Error, "Unknown Vertex AI embedding record key: #{key.inspect}" unless match
|
|
15
|
+
|
|
16
|
+
[Integer(match[1]), Integer(match[2]), Integer(match[3]), match[4] == 'a']
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def parse_embedding_batch_group(index, rows, model:)
|
|
20
|
+
metadata = rows.map { |row| embedding_row_metadata(row) }
|
|
21
|
+
error = embedding_group_error(rows, metadata)
|
|
22
|
+
return [index, nil, batch_failure(index, error)] if error
|
|
23
|
+
return if rows.size < metadata.first[2]
|
|
24
|
+
|
|
25
|
+
ordered = rows.sort_by { |row| embedding_row_metadata(row)[1] }
|
|
26
|
+
parse_embedding_batch_result(index, ordered, model:, array: metadata.first[3])
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
def parse_embedding_batch_result(index, rows, model:, array:)
|
|
30
|
+
vectors = embedding_batch_vectors(rows)
|
|
31
|
+
return [index, nil, batch_failure(index, 'Vertex AI returned no valid embedding')] unless vectors
|
|
32
|
+
|
|
33
|
+
counts = rows.map { |row| embedding_batch_tokens(row) }
|
|
34
|
+
result = Embedding.new(vectors: array ? vectors : vectors.first,
|
|
35
|
+
model:, input_tokens: (counts.sum if counts.all?))
|
|
36
|
+
[index, result]
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def embedding_group_error(rows, metadata)
|
|
40
|
+
unless valid_embedding_metadata?(metadata)
|
|
41
|
+
return 'Invalid or duplicate Vertex AI embedding record positions'
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
rows.filter_map { |row| batch_error_value(row['error']) || batch_error_value(row['status']) }
|
|
45
|
+
.find { |error| !error.empty? }
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def valid_embedding_metadata?(metadata)
|
|
49
|
+
count = metadata.first[2]
|
|
50
|
+
positions = metadata.map { |entry| entry[1] }
|
|
51
|
+
return false unless metadata.first[3] || count == 1
|
|
52
|
+
|
|
53
|
+
metadata.map { |entry| entry[2..] }.uniq.one? && count.positive? &&
|
|
54
|
+
positions.uniq.size == positions.size && positions.max < count
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
def embedding_batch_vectors(rows)
|
|
58
|
+
vectors = rows.map do |row|
|
|
59
|
+
row.dig('response', 'embedding', 'values') || row.dig('predictions', 0, 'embeddings', 'values')
|
|
60
|
+
end
|
|
61
|
+
vectors if vectors.all? { |vector| vector.is_a?(Array) && vector.any? && vector.all?(Numeric) }
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
def embedding_batch_tokens(row)
|
|
65
|
+
value = row.dig('response', 'tokenCount') ||
|
|
66
|
+
row.dig('response', 'usageMetadata', 'promptTokenCount') ||
|
|
67
|
+
row.dig('predictions', 0, 'embeddings', 'statistics', 'token_count')
|
|
68
|
+
Integer(value) unless value.nil?
|
|
69
|
+
end
|
|
70
|
+
end
|
|
71
|
+
end
|
|
72
|
+
end
|
|
73
|
+
end
|
|
74
|
+
end
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Protocols
|
|
5
|
+
module VertexAI
|
|
6
|
+
# Text embeddings over Vertex AI's Cloud Storage batch prediction API.
|
|
7
|
+
class EmbeddingPrediction < Protocol
|
|
8
|
+
include BatchPrediction
|
|
9
|
+
include Requests
|
|
10
|
+
include Results
|
|
11
|
+
|
|
12
|
+
LEGACY_MODELS = %w[text-embedding-004 text-embedding-005 text-multilingual-embedding-002].freeze
|
|
13
|
+
GEMINI_MODELS = %w[gemini-embedding-001 gemini-embedding-2].freeze
|
|
14
|
+
MODELS = (LEGACY_MODELS + GEMINI_MODELS).freeze
|
|
15
|
+
|
|
16
|
+
private
|
|
17
|
+
|
|
18
|
+
def validate_batch_requests!(requests)
|
|
19
|
+
@requests = requests
|
|
20
|
+
model = requests.first.fetch(:model)
|
|
21
|
+
unless MODELS.include?(model)
|
|
22
|
+
raise Error,
|
|
23
|
+
"Vertex AI embedding batches are not supported for #{model.inspect}"
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
requests.each { |request| validate_embedding_request!(request) }
|
|
27
|
+
return unless LEGACY_MODELS.include?(model)
|
|
28
|
+
return if requests.map { |request| request.fetch(:payload).fetch(:parameters, {}) }.uniq.one?
|
|
29
|
+
|
|
30
|
+
raise ArgumentError,
|
|
31
|
+
'Vertex AI legacy embedding batches require the same dimensions and parameters in every request'
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def vertex_batch_job(model, input_uri, output_uri)
|
|
35
|
+
job = super.merge(labels: { ruby_llm_request_count: @requests.size.to_s })
|
|
36
|
+
return job unless LEGACY_MODELS.include?(model)
|
|
37
|
+
|
|
38
|
+
job.merge(instanceConfig: { instanceType: 'object', keyField: 'key' },
|
|
39
|
+
modelParameters: @requests.first.fetch(:payload).fetch(:parameters, {}))
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
def vertex_batch_jsonl(requests)
|
|
43
|
+
rows = requests.flat_map { |request| render_embedding_batch_rows(request) }
|
|
44
|
+
rows.map { |row| JSON.generate(row) }.join("\n")
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
def parse_vertex_batch_results(rows, job:)
|
|
48
|
+
model = job.fetch('model').split('/').last
|
|
49
|
+
rows.group_by { |row| embedding_row_metadata(row).first }.filter_map do |index, group|
|
|
50
|
+
parse_embedding_batch_group(index, group, model:)
|
|
51
|
+
end
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
end
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Protocols
|
|
5
|
+
module VertexAI
|
|
6
|
+
# Google Cloud Storage-backed files for Vertex AI batch input and output.
|
|
7
|
+
class Files < Protocols::Files
|
|
8
|
+
# GCS object names allow spaces and other characters URI() rejects.
|
|
9
|
+
GCS_URI = %r{\Ags://([^/]+)/?(.*)\z}m
|
|
10
|
+
|
|
11
|
+
# rubocop:disable-next Lint/UnusedMethodArgument
|
|
12
|
+
def upload(file, filename: nil, purpose: nil, expires_in: nil, uri: nil, content_type: nil,
|
|
13
|
+
provider_options: {})
|
|
14
|
+
attachment = file_attachment(file, filename:)
|
|
15
|
+
target_uri = uri || storage_uri_for(attachment)
|
|
16
|
+
bucket_name, key = parse_gcs_uri(target_uri)
|
|
17
|
+
|
|
18
|
+
with_file_body(attachment) do |body|
|
|
19
|
+
bucket(bucket_name).create_file(body, key, content_type: content_type || file_content_type(attachment))
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
uploaded_file(
|
|
23
|
+
{ 'uri' => target_uri },
|
|
24
|
+
id: target_uri,
|
|
25
|
+
uri: target_uri,
|
|
26
|
+
filename: attachment.filename,
|
|
27
|
+
byte_size: file_size(attachment),
|
|
28
|
+
mime_type: content_type || file_content_type(attachment)
|
|
29
|
+
)
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
def find(file_id)
|
|
33
|
+
bucket_name, key = parse_gcs_uri(file_id)
|
|
34
|
+
object = bucket(bucket_name).file(key)
|
|
35
|
+
raise Error, "GCS object not found: #{file_id}" unless object
|
|
36
|
+
|
|
37
|
+
uploaded_file(
|
|
38
|
+
{ 'uri' => file_id },
|
|
39
|
+
id: file_id,
|
|
40
|
+
uri: file_id,
|
|
41
|
+
filename: File.basename(key),
|
|
42
|
+
byte_size: object.size,
|
|
43
|
+
created_at: object.created_at,
|
|
44
|
+
mime_type: object.content_type
|
|
45
|
+
)
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def download(file_id)
|
|
49
|
+
bucket_name, key = parse_gcs_uri(file_id)
|
|
50
|
+
object = bucket(bucket_name).file(key)
|
|
51
|
+
raise Error, "GCS object not found: #{file_id}" unless object
|
|
52
|
+
|
|
53
|
+
object.download.string
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
def list_uris(prefix_uri)
|
|
57
|
+
bucket_name, prefix = parse_gcs_uri(prefix_uri)
|
|
58
|
+
uris = []
|
|
59
|
+
bucket(bucket_name).files(prefix: prefix).all do |object|
|
|
60
|
+
uris << "gs://#{bucket_name}/#{object.name}"
|
|
61
|
+
end
|
|
62
|
+
uris
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
private
|
|
66
|
+
|
|
67
|
+
def storage_uri_for(attachment)
|
|
68
|
+
base = @config.vertexai_batch_gcs_uri.to_s.sub(%r{/+\z}, '')
|
|
69
|
+
raise ConfigurationError, 'Set vertexai_batch_gcs_uri to a gs:// bucket prefix' if base.empty?
|
|
70
|
+
|
|
71
|
+
"#{base}/ruby_llm_uploads/#{SecureRandom.hex(8)}/#{attachment.filename}"
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
def storage
|
|
75
|
+
require 'google/cloud/storage'
|
|
76
|
+
|
|
77
|
+
options = { project_id: @config.vertexai_project_id }
|
|
78
|
+
if @config.vertexai_service_account_key
|
|
79
|
+
options[:credentials] =
|
|
80
|
+
JSON.parse(@config.vertexai_service_account_key)
|
|
81
|
+
end
|
|
82
|
+
::Google::Cloud::Storage.new(**options)
|
|
83
|
+
rescue LoadError
|
|
84
|
+
raise Error, 'The google-cloud-storage gem is required for Vertex AI file uploads. ' \
|
|
85
|
+
'Please add it to your Gemfile: gem "google-cloud-storage"'
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
def bucket(name)
|
|
89
|
+
storage.bucket(name) || raise(Error, "GCS bucket not found: #{name}")
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
def parse_gcs_uri(uri)
|
|
93
|
+
match = GCS_URI.match(uri.to_s)
|
|
94
|
+
raise ArgumentError, "Expected a gs:// URI, got: #{uri}" unless match
|
|
95
|
+
|
|
96
|
+
[match[1], match[2]]
|
|
97
|
+
end
|
|
98
|
+
end
|
|
99
|
+
end
|
|
100
|
+
end
|
|
101
|
+
end
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Protocols
|
|
5
|
+
module VertexAI
|
|
6
|
+
# Vertex AI Search's Discovery Engine ranking API.
|
|
7
|
+
class Ranking < Protocol
|
|
8
|
+
def initialize(provider, model = nil)
|
|
9
|
+
super
|
|
10
|
+
@connection = provider.ranking_connection
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
def rerank_url
|
|
14
|
+
"#{@provider.ranking_config}:rank"
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def render_rerank_payload(query, documents, model:, top_n: nil, provider_options: {})
|
|
18
|
+
validate_ranking_input(query, documents, top_n)
|
|
19
|
+
records = documents.each_with_index.map { |document, index| { id: index.to_s, content: document } }
|
|
20
|
+
{ model: model, query: query, records: records, topN: top_n }.compact.merge(provider_options)
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
def parse_rerank_response(response, model:, documents: [])
|
|
24
|
+
records = response.body['records']
|
|
25
|
+
raise Error.new('Vertex AI Search returned no ranking records', response:) unless records.is_a?(Array)
|
|
26
|
+
|
|
27
|
+
seen = []
|
|
28
|
+
results = records.map do |record|
|
|
29
|
+
index = ranking_index(record, documents)
|
|
30
|
+
raise Error.new('Vertex AI Search returned a duplicate document id', response:) if seen.include?(index)
|
|
31
|
+
|
|
32
|
+
seen << index
|
|
33
|
+
Rerank::Result.new(index: index, document: documents[index], score: record.fetch('score'))
|
|
34
|
+
end
|
|
35
|
+
Rerank.new(results: results, model: model, raw: response.body)
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
private
|
|
39
|
+
|
|
40
|
+
def validate_ranking_input(query, documents, top_n)
|
|
41
|
+
unless query.is_a?(String) && !query.empty?
|
|
42
|
+
raise ArgumentError, 'Vertex AI Search reranking requires a nonempty query'
|
|
43
|
+
end
|
|
44
|
+
unless valid_ranking_documents?(documents)
|
|
45
|
+
raise ArgumentError, 'Vertex AI Search reranking accepts between 1 and 1000 nonempty text documents'
|
|
46
|
+
end
|
|
47
|
+
return if top_n.nil? || (top_n.is_a?(Integer) && top_n.positive?)
|
|
48
|
+
|
|
49
|
+
raise ArgumentError, 'top_n must be a positive integer'
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def valid_ranking_documents?(documents)
|
|
53
|
+
documents.is_a?(Array) && (1..1000).cover?(documents.length) &&
|
|
54
|
+
documents.all? { |document| document.is_a?(String) && !document.empty? }
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
def ranking_index(record, documents)
|
|
58
|
+
id = record['id'] if record.is_a?(Hash)
|
|
59
|
+
unless id.is_a?(String) && id.match?(/\A(?:0|[1-9]\d*)\z/) && id.to_i < documents.length &&
|
|
60
|
+
record['score'].is_a?(Numeric)
|
|
61
|
+
raise Error, 'Vertex AI Search returned an invalid document id or score'
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
id.to_i
|
|
65
|
+
end
|
|
66
|
+
end
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
end
|