ruby_llm 1.16.0 → 2.0.0.rc2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.rdoc_options +25 -0
- data/README.md +85 -32
- data/exe/ruby_llm +8 -0
- data/lib/generators/ruby_llm/agent/templates/agent.rb.tt +0 -1
- data/lib/generators/ruby_llm/chat_ui/chat_ui_generator.rb +3 -43
- data/lib/generators/ruby_llm/chat_ui/templates/controllers/chats_controller.rb.tt +11 -3
- data/lib/generators/ruby_llm/chat_ui/templates/controllers/messages_controller.rb.tt +8 -0
- data/lib/generators/ruby_llm/chat_ui/templates/controllers/models_controller.rb.tt +4 -4
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_chat.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/_form.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/index.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/chats/show.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_assistant.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_system.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_tool_calls.html.erb.tt +6 -4
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/_user.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/messages/tool_calls/_default.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/_model.html.erb.tt +5 -6
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/index.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/tailwind/views/models/show.html.erb.tt +5 -5
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_chat.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/_form.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/index.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/chats/show.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_assistant.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_system.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_tool_calls.html.erb.tt +6 -4
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/_user.html.erb.tt +1 -1
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/create.turbo_stream.erb.tt +4 -6
- data/lib/generators/ruby_llm/chat_ui/templates/views/messages/tool_calls/_default.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/views/models/_model.html.erb.tt +5 -6
- data/lib/generators/ruby_llm/chat_ui/templates/views/models/index.html.erb.tt +2 -2
- data/lib/generators/ruby_llm/chat_ui/templates/views/models/show.html.erb.tt +3 -3
- data/lib/generators/ruby_llm/generator_helpers.rb +106 -62
- data/lib/generators/ruby_llm/install/install_generator.rb +3 -11
- data/lib/generators/ruby_llm/install/templates/create_chats_migration.rb.tt +2 -0
- data/lib/generators/ruby_llm/install/templates/create_messages_migration.rb.tt +14 -8
- data/lib/generators/ruby_llm/install/templates/create_ruby_llm_records_migration.rb.tt +117 -0
- data/lib/generators/ruby_llm/install/templates/initializer.rb.tt +0 -8
- data/lib/generators/ruby_llm/provider/cli.rb +175 -0
- data/lib/generators/ruby_llm/provider/scaffold.rb +323 -0
- data/lib/generators/ruby_llm/provider/templates/core/provider.rb.erb +37 -0
- data/lib/generators/ruby_llm/provider/templates/core/provider_spec.rb.erb +34 -0
- data/lib/generators/ruby_llm/provider/templates/gem/archspec.rb.erb +14 -0
- data/lib/generators/ruby_llm/provider/templates/gem/bin/console.erb +15 -0
- data/lib/generators/ruby_llm/provider/templates/gem/bin/setup.erb +5 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_schema_spec.rb.erb +30 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_spec.rb.erb +35 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_streaming_spec.rb.erb +22 -0
- data/lib/generators/ruby_llm/provider/templates/gem/chat_tools_spec.rb.erb +30 -0
- data/lib/generators/ruby_llm/provider/templates/gem/ci.yml.erb +32 -0
- data/lib/generators/ruby_llm/provider/templates/gem/embedding_spec.rb.erb +49 -0
- data/lib/generators/ruby_llm/provider/templates/gem/env.erb +2 -0
- data/lib/generators/ruby_llm/provider/templates/gem/fixtures_gitkeep.erb +1 -0
- data/lib/generators/ruby_llm/provider/templates/gem/flayignore.erb +1 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gemfile.erb +23 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gemspec.erb +28 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gitignore.erb +7 -0
- data/lib/generators/ruby_llm/provider/templates/gem/gitleaks.yml.erb +22 -0
- data/lib/generators/ruby_llm/provider/templates/gem/image_spec.rb.erb +23 -0
- data/lib/generators/ruby_llm/provider/templates/gem/license.erb +21 -0
- data/lib/generators/ruby_llm/provider/templates/gem/models.rb.erb +19 -0
- data/lib/generators/ruby_llm/provider/templates/gem/models_spec.rb.erb +17 -0
- data/lib/generators/ruby_llm/provider/templates/gem/moderation_spec.rb.erb +22 -0
- data/lib/generators/ruby_llm/provider/templates/gem/overcommit.yml.erb +31 -0
- data/lib/generators/ruby_llm/provider/templates/gem/provider.rb.erb +52 -0
- data/lib/generators/ruby_llm/provider/templates/gem/provider_spec.rb.erb +38 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rakefile.erb +38 -0
- data/lib/generators/ruby_llm/provider/templates/gem/readme.md.erb +50 -0
- data/lib/generators/ruby_llm/provider/templates/gem/release.yml.erb +36 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rerank_spec.rb.erb +23 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rspec.erb +2 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rubocop.yml.erb +29 -0
- data/lib/generators/ruby_llm/provider/templates/gem/rubyllm_configuration.rb.erb +14 -0
- data/lib/generators/ruby_llm/provider/templates/gem/spec_helper.rb.erb +26 -0
- data/lib/generators/ruby_llm/provider/templates/gem/speech_spec.rb.erb +25 -0
- data/lib/generators/ruby_llm/provider/templates/gem/vcr_configuration.rb.erb +16 -0
- data/lib/generators/ruby_llm/provider/templates/gem/video_spec.rb.erb +27 -0
- data/lib/generators/ruby_llm/schema/schema_generator.rb +5 -1
- data/lib/generators/ruby_llm/schema/templates/schema.rb.tt +1 -1
- data/lib/generators/ruby_llm/tool/templates/tailwind/tool_call.html.erb.tt +13 -0
- data/lib/generators/ruby_llm/tool/templates/tailwind/tool_result.html.erb.tt +21 -0
- data/lib/generators/ruby_llm/tool/templates/tool.rb.tt +3 -3
- data/lib/generators/ruby_llm/tool/templates/tool_call.html.erb.tt +7 -12
- data/lib/generators/ruby_llm/tool/templates/tool_result.html.erb.tt +5 -2
- data/lib/generators/ruby_llm/tool/tool_generator.rb +25 -59
- data/lib/generators/ruby_llm/upgrade/templates/backfill_v2_data.rb.tt +461 -0
- data/lib/generators/ruby_llm/upgrade/templates/cleanup_v2_upgrade.rb.tt +101 -0
- data/lib/generators/ruby_llm/upgrade/templates/finish_v2_upgrade.rb.tt +215 -0
- data/lib/generators/ruby_llm/upgrade/templates/prepare_v2_upgrade.rb.tt +660 -0
- data/lib/generators/ruby_llm/upgrade/templates/ruby_llm_upgrade.rb.tt +222 -0
- data/lib/generators/ruby_llm/upgrade/templates/upgrade_initializer.rb.tt +13 -0
- data/lib/generators/ruby_llm/upgrade/upgrade_generator.rb +167 -0
- data/lib/generators/ruby_llm/upgrade/upgrade_migration.rb +344 -0
- data/lib/ruby_llm/accounting/usage.rb +245 -0
- data/lib/ruby_llm/active_record/acts_as.rb +94 -111
- data/lib/ruby_llm/active_record/attachment_helpers.rb +186 -0
- data/lib/ruby_llm/active_record/batch.rb +97 -0
- data/lib/ruby_llm/active_record/chat_methods.rb +828 -305
- data/lib/ruby_llm/active_record/message_methods.rb +113 -136
- data/lib/ruby_llm/active_record/model.rb +135 -0
- data/lib/ruby_llm/active_record/payload_helpers.rb +1 -2
- data/lib/ruby_llm/active_record/tool_call.rb +33 -0
- data/lib/ruby_llm/active_record/usage.rb +61 -0
- data/lib/ruby_llm/agent.rb +1065 -151
- data/lib/ruby_llm/aliases.json +268 -100
- data/lib/ruby_llm/attachment.rb +187 -48
- data/lib/ruby_llm/batch.rb +432 -0
- data/lib/ruby_llm/cached_content.rb +112 -0
- data/lib/ruby_llm/chat/tool_concurrency.rb +111 -0
- data/lib/ruby_llm/chat.rb +1127 -198
- data/lib/ruby_llm/chunk.rb +10 -0
- data/lib/ruby_llm/citation.rb +105 -0
- data/lib/ruby_llm/configuration.rb +261 -24
- data/lib/ruby_llm/context.rb +128 -6
- data/lib/ruby_llm/cost.rb +217 -80
- data/lib/ruby_llm/downloaded_file.rb +33 -0
- data/lib/ruby_llm/embedding.rb +121 -17
- data/lib/ruby_llm/embedding_request.rb +53 -0
- data/lib/ruby_llm/error.rb +159 -23
- data/lib/ruby_llm/fallback.rb +133 -0
- data/lib/ruby_llm/files/mime_type.rb +97 -0
- data/lib/ruby_llm/image.rb +153 -32
- data/lib/ruby_llm/message.rb +233 -54
- data/lib/ruby_llm/model/modalities.rb +17 -4
- data/lib/ruby_llm/model/pricing.rb +24 -5
- data/lib/ruby_llm/model/pricing_category.rb +103 -14
- data/lib/ruby_llm/model/pricing_tier.rb +55 -15
- data/lib/ruby_llm/model.rb +244 -2
- data/lib/ruby_llm/models/aliases.rb +41 -0
- data/lib/ruby_llm/models/registry.rb +165 -0
- data/lib/ruby_llm/models/schema.rb +99 -0
- data/lib/ruby_llm/models.json +73261 -33733
- data/lib/ruby_llm/models.rb +477 -215
- data/lib/ruby_llm/moderation.rb +139 -26
- data/lib/ruby_llm/ocr.rb +112 -0
- data/lib/ruby_llm/prompt.rb +79 -0
- data/lib/ruby_llm/protocol/binary_streaming.rb +65 -0
- data/lib/ruby_llm/protocol/stream_accumulator.rb +214 -0
- data/lib/ruby_llm/protocol/streaming.rb +230 -0
- data/lib/ruby_llm/protocol.rb +662 -0
- data/lib/ruby_llm/protocols/anthropic/batches.rb +73 -0
- data/lib/ruby_llm/protocols/anthropic/chat.rb +548 -0
- data/lib/ruby_llm/protocols/anthropic/embeddings.rb +14 -0
- data/lib/ruby_llm/protocols/anthropic/files.rb +38 -0
- data/lib/ruby_llm/protocols/anthropic/media.rb +141 -0
- data/lib/ruby_llm/protocols/anthropic/models.rb +129 -0
- data/lib/ruby_llm/protocols/anthropic/streaming.rb +167 -0
- data/lib/ruby_llm/{providers → protocols}/anthropic/tools.rb +24 -41
- data/lib/ruby_llm/protocols/anthropic.rb +101 -0
- data/lib/ruby_llm/protocols/azure/files.rb +16 -0
- data/lib/ruby_llm/protocols/bedrock/async_videos.rb +109 -0
- data/lib/ruby_llm/protocols/bedrock/batches.rb +129 -0
- data/lib/ruby_llm/protocols/bedrock/files.rb +110 -0
- data/lib/ruby_llm/protocols/bedrock/guardrails.rb +122 -0
- data/lib/ruby_llm/protocols/bedrock/rerank.rb +95 -0
- data/lib/ruby_llm/protocols/chat_completions/batches.rb +32 -0
- data/lib/ruby_llm/protocols/chat_completions/chat.rb +493 -0
- data/lib/ruby_llm/protocols/chat_completions/embedding_batches.rb +35 -0
- data/lib/ruby_llm/protocols/chat_completions/embeddings.rb +60 -0
- data/lib/ruby_llm/protocols/chat_completions/images.rb +127 -0
- data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/media.rb +30 -17
- data/lib/ruby_llm/protocols/chat_completions/models.rb +39 -0
- data/lib/ruby_llm/protocols/chat_completions/moderation.rb +52 -0
- data/lib/ruby_llm/protocols/chat_completions/rerank.rb +56 -0
- data/lib/ruby_llm/protocols/chat_completions/speech.rb +40 -0
- data/lib/ruby_llm/protocols/chat_completions/streaming.rb +69 -0
- data/lib/ruby_llm/{providers/openai → protocols/chat_completions}/tools.rb +15 -17
- data/lib/ruby_llm/protocols/chat_completions/transcription.rb +150 -0
- data/lib/ruby_llm/protocols/chat_completions.rb +21 -0
- data/lib/ruby_llm/protocols/cohere/batch_requests.rb +75 -0
- data/lib/ruby_llm/protocols/cohere/batches.rb +98 -0
- data/lib/ruby_llm/protocols/cohere/chat.rb +227 -0
- data/lib/ruby_llm/protocols/cohere/datasets.rb +102 -0
- data/lib/ruby_llm/protocols/cohere/embeddings.rb +68 -0
- data/lib/ruby_llm/protocols/cohere/media.rb +77 -0
- data/lib/ruby_llm/protocols/cohere/models.rb +92 -0
- data/lib/ruby_llm/protocols/cohere/ocr.rb +63 -0
- data/lib/ruby_llm/protocols/cohere/rerank.rb +52 -0
- data/lib/ruby_llm/protocols/cohere/streaming.rb +105 -0
- data/lib/ruby_llm/protocols/cohere/tokenization.rb +22 -0
- data/lib/ruby_llm/protocols/cohere/tools.rb +132 -0
- data/lib/ruby_llm/protocols/cohere/transcription.rb +41 -0
- data/lib/ruby_llm/protocols/cohere.rb +21 -0
- data/lib/ruby_llm/protocols/converse/batches.rb +55 -0
- data/lib/ruby_llm/protocols/converse/chat.rb +695 -0
- data/lib/ruby_llm/protocols/converse/media.rb +178 -0
- data/lib/ruby_llm/protocols/converse/streaming.rb +432 -0
- data/lib/ruby_llm/protocols/converse/thinking_stream.rb +56 -0
- data/lib/ruby_llm/protocols/converse.rb +54 -0
- data/lib/ruby_llm/protocols/deepgram/models.rb +74 -0
- data/lib/ruby_llm/protocols/deepgram/speech.rb +93 -0
- data/lib/ruby_llm/protocols/deepgram/streaming_transcription.rb +89 -0
- data/lib/ruby_llm/protocols/deepgram/transcription.rb +96 -0
- data/lib/ruby_llm/protocols/deepgram.rb +19 -0
- data/lib/ruby_llm/protocols/deepseek/files.rb +31 -0
- data/lib/ruby_llm/protocols/elevenlabs/assets.rb +36 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows/images.rb +74 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows/media.rb +42 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows/videos.rb +93 -0
- data/lib/ruby_llm/protocols/elevenlabs/flows.rb +14 -0
- data/lib/ruby_llm/protocols/elevenlabs/models.rb +64 -0
- data/lib/ruby_llm/protocols/elevenlabs/speech.rb +67 -0
- data/lib/ruby_llm/protocols/elevenlabs/streaming_transcription.rb +127 -0
- data/lib/ruby_llm/protocols/elevenlabs/transcription.rb +61 -0
- data/lib/ruby_llm/protocols/elevenlabs.rb +15 -0
- data/lib/ruby_llm/protocols/files.rb +119 -0
- data/lib/ruby_llm/protocols/gemini/batches.rb +162 -0
- data/lib/ruby_llm/protocols/gemini/caches.rb +59 -0
- data/lib/ruby_llm/protocols/gemini/chat.rb +453 -0
- data/lib/ruby_llm/protocols/gemini/embedding_batches.rb +86 -0
- data/lib/ruby_llm/protocols/gemini/embeddings.rb +70 -0
- data/lib/ruby_llm/protocols/gemini/file_transcription.rb +32 -0
- data/lib/ruby_llm/protocols/gemini/files.rb +115 -0
- data/lib/ruby_llm/protocols/gemini/images.rb +183 -0
- data/lib/ruby_llm/protocols/gemini/live_transcription.rb +140 -0
- data/lib/ruby_llm/{providers → protocols}/gemini/media.rb +17 -11
- data/lib/ruby_llm/protocols/gemini/models.rb +71 -0
- data/lib/ruby_llm/protocols/gemini/speech.rb +56 -0
- data/lib/ruby_llm/protocols/gemini/streaming.rb +96 -0
- data/lib/ruby_llm/protocols/gemini/tools.rb +157 -0
- data/lib/ruby_llm/{providers → protocols}/gemini/transcription.rb +22 -22
- data/lib/ruby_llm/protocols/gemini/videos.rb +103 -0
- data/lib/ruby_llm/protocols/gemini.rb +35 -0
- data/lib/ruby_llm/protocols/gpustack/responses.rb +111 -0
- data/lib/ruby_llm/protocols/gpustack/tokenization.rb +22 -0
- data/lib/ruby_llm/protocols/gpustack/videos.rb +96 -0
- data/lib/ruby_llm/protocols/interactions/chat.rb +145 -0
- data/lib/ruby_llm/protocols/interactions/content.rb +90 -0
- data/lib/ruby_llm/protocols/interactions/streaming.rb +91 -0
- data/lib/ruby_llm/protocols/interactions/tools.rb +51 -0
- data/lib/ruby_llm/protocols/interactions/transcription.rb +58 -0
- data/lib/ruby_llm/protocols/interactions.rb +29 -0
- data/lib/ruby_llm/protocols/invoke_model/cohere_embeddings.rb +51 -0
- data/lib/ruby_llm/protocols/invoke_model/embedding_batches.rb +111 -0
- data/lib/ruby_llm/protocols/invoke_model/nova_embeddings.rb +50 -0
- data/lib/ruby_llm/protocols/invoke_model/stability_images.rb +103 -0
- data/lib/ruby_llm/protocols/invoke_model/titan_multimodal_embeddings.rb +33 -0
- data/lib/ruby_llm/protocols/invoke_model/titan_text_embeddings.rb +44 -0
- data/lib/ruby_llm/protocols/invoke_model.rb +57 -0
- data/lib/ruby_llm/protocols/mistral/content.rb +49 -0
- data/lib/ruby_llm/protocols/mistral/conversations/chat.rb +160 -0
- data/lib/ruby_llm/protocols/mistral/conversations/images.rb +43 -0
- data/lib/ruby_llm/protocols/mistral/conversations/streaming.rb +83 -0
- data/lib/ruby_llm/protocols/mistral/conversations.rb +30 -0
- data/lib/ruby_llm/protocols/mistral/files.rb +36 -0
- data/lib/ruby_llm/protocols/mistral/multi_completion.rb +160 -0
- data/lib/ruby_llm/protocols/openai/batches.rb +126 -0
- data/lib/ruby_llm/protocols/openai/files.rb +42 -0
- data/lib/ruby_llm/protocols/openrouter/batches.rb +147 -0
- data/lib/ruby_llm/protocols/openrouter/files.rb +24 -0
- data/lib/ruby_llm/protocols/openrouter/responses.rb +53 -0
- data/lib/ruby_llm/protocols/openrouter/transcription.rb +51 -0
- data/lib/ruby_llm/protocols/perplexity/files.rb +48 -0
- data/lib/ruby_llm/protocols/perplexity/router.rb +59 -0
- data/lib/ruby_llm/protocols/responses/approvals.rb +32 -0
- data/lib/ruby_llm/protocols/responses/batches.rb +32 -0
- data/lib/ruby_llm/protocols/responses/chat.rb +476 -0
- data/lib/ruby_llm/protocols/responses/compaction.rb +29 -0
- data/lib/ruby_llm/protocols/responses/media.rb +61 -0
- data/lib/ruby_llm/protocols/responses/streaming.rb +117 -0
- data/lib/ruby_llm/protocols/responses/token_counting.rb +26 -0
- data/lib/ruby_llm/protocols/responses/tools.rb +39 -0
- data/lib/ruby_llm/protocols/responses.rb +35 -0
- data/lib/ruby_llm/protocols/vertexai/batch_prediction.rb +155 -0
- data/lib/ruby_llm/protocols/vertexai/embedding_prediction/requests.rb +85 -0
- data/lib/ruby_llm/protocols/vertexai/embedding_prediction/results.rb +74 -0
- data/lib/ruby_llm/protocols/vertexai/embedding_prediction.rb +56 -0
- data/lib/ruby_llm/protocols/vertexai/files.rb +101 -0
- data/lib/ruby_llm/protocols/vertexai/ranking.rb +69 -0
- data/lib/ruby_llm/protocols/vertexai/research.rb +193 -0
- data/lib/ruby_llm/protocols/xai/files.rb +30 -0
- data/lib/ruby_llm/protocols/xai/streaming_transcription.rb +120 -0
- data/lib/ruby_llm/protocols/xai/tokenization.rb +23 -0
- data/lib/ruby_llm/provider.rb +560 -128
- data/lib/ruby_llm/providers/anthropic/capabilities.rb +5 -7
- data/lib/ruby_llm/providers/anthropic.rb +4 -6
- data/lib/ruby_llm/providers/azure/audio.rb +18 -0
- data/lib/ruby_llm/providers/azure/capabilities.rb +16 -0
- data/lib/ruby_llm/providers/azure/chat.rb +2 -9
- data/lib/ruby_llm/providers/azure/chat_completions/batches.rb +29 -0
- data/lib/ruby_llm/providers/azure/chat_completions.rb +80 -0
- data/lib/ruby_llm/providers/azure/cohere.rb +33 -0
- data/lib/ruby_llm/providers/azure/embeddings.rb +3 -2
- data/lib/ruby_llm/providers/azure/images.rb +22 -0
- data/lib/ruby_llm/providers/azure/media.rb +5 -14
- data/lib/ruby_llm/providers/azure/models.rb +35 -0
- data/lib/ruby_llm/providers/azure/responses.rb +26 -0
- data/lib/ruby_llm/providers/azure/videos.rb +64 -0
- data/lib/ruby_llm/providers/azure.rb +77 -78
- data/lib/ruby_llm/providers/bedrock/auth.rb +60 -41
- data/lib/ruby_llm/providers/bedrock/capabilities.rb +18 -0
- data/lib/ruby_llm/providers/bedrock/mantle/anthropic.rb +39 -0
- data/lib/ruby_llm/providers/bedrock/mantle/chat_completions.rb +23 -0
- data/lib/ruby_llm/providers/bedrock/mantle/responses.rb +24 -0
- data/lib/ruby_llm/providers/bedrock/mantle/voxtral.rb +96 -0
- data/lib/ruby_llm/providers/bedrock/mantle.rb +57 -0
- data/lib/ruby_llm/providers/bedrock/models.rb +193 -41
- data/lib/ruby_llm/providers/bedrock.rb +216 -45
- data/lib/ruby_llm/providers/cohere.rb +31 -0
- data/lib/ruby_llm/providers/deepgram.rb +37 -0
- data/lib/ruby_llm/providers/deepseek/capabilities.rb +4 -51
- data/lib/ruby_llm/providers/deepseek/chat.rb +49 -2
- data/lib/ruby_llm/providers/deepseek/responses.rb +68 -0
- data/lib/ruby_llm/providers/deepseek.rb +9 -2
- data/lib/ruby_llm/providers/elevenlabs.rb +35 -0
- data/lib/ruby_llm/providers/gemini/capabilities.rb +8 -107
- data/lib/ruby_llm/providers/gemini.rb +15 -8
- data/lib/ruby_llm/providers/gpustack/chat.rb +2 -16
- data/lib/ruby_llm/providers/gpustack/embeddings.rb +28 -0
- data/lib/ruby_llm/providers/gpustack/media.rb +17 -16
- data/lib/ruby_llm/providers/gpustack/models.rb +72 -62
- data/lib/ruby_llm/providers/gpustack/speech.rb +15 -0
- data/lib/ruby_llm/providers/gpustack/transcription.rb +29 -0
- data/lib/ruby_llm/providers/gpustack.rb +35 -11
- data/lib/ruby_llm/providers/mistral/capabilities.rb +7 -155
- data/lib/ruby_llm/providers/mistral/chat.rb +37 -61
- data/lib/ruby_llm/providers/mistral/chat_completions/batches.rb +120 -0
- data/lib/ruby_llm/providers/mistral/chat_completions.rb +21 -0
- data/lib/ruby_llm/providers/mistral/conversations.rb +12 -0
- data/lib/ruby_llm/providers/mistral/embeddings.rb +6 -4
- data/lib/ruby_llm/providers/mistral/media.rb +6 -18
- data/lib/ruby_llm/providers/mistral/models.rb +57 -23
- data/lib/ruby_llm/providers/mistral/ocr.rb +47 -0
- data/lib/ruby_llm/providers/mistral/speech.rb +51 -0
- data/lib/ruby_llm/providers/mistral/transcription.rb +62 -0
- data/lib/ruby_llm/providers/mistral.rb +16 -4
- data/lib/ruby_llm/providers/ollama/chat.rb +7 -13
- data/lib/ruby_llm/providers/ollama/media.rb +6 -15
- data/lib/ruby_llm/providers/ollama/models.rb +50 -9
- data/lib/ruby_llm/providers/ollama.rb +9 -8
- data/lib/ruby_llm/providers/ollama_cloud/models.rb +14 -0
- data/lib/ruby_llm/providers/ollama_cloud.rb +40 -0
- data/lib/ruby_llm/providers/openai/capabilities.rb +54 -259
- data/lib/ruby_llm/providers/openai/models.rb +23 -23
- data/lib/ruby_llm/providers/openai/responses.rb +13 -0
- data/lib/ruby_llm/providers/openai.rb +92 -11
- data/lib/ruby_llm/providers/openrouter/chat.rb +130 -108
- data/lib/ruby_llm/providers/openrouter/embeddings.rb +51 -0
- data/lib/ruby_llm/providers/openrouter/images.rb +44 -43
- data/lib/ruby_llm/providers/openrouter/media.rb +34 -0
- data/lib/ruby_llm/providers/openrouter/models.rb +50 -11
- data/lib/ruby_llm/providers/openrouter/speech.rb +32 -0
- data/lib/ruby_llm/providers/openrouter/streaming.rb +31 -38
- data/lib/ruby_llm/providers/openrouter/videos.rb +81 -0
- data/lib/ruby_llm/providers/openrouter.rb +78 -20
- data/lib/ruby_llm/providers/perplexity/chat.rb +2 -9
- data/lib/ruby_llm/providers/perplexity/embeddings.rb +32 -0
- data/lib/ruby_llm/providers/perplexity/media.rb +5 -21
- data/lib/ruby_llm/providers/perplexity/models.rb +80 -13
- data/lib/ruby_llm/providers/perplexity.rb +28 -20
- data/lib/ruby_llm/providers/vertexai/anthropic/batches.rb +52 -0
- data/lib/ruby_llm/providers/vertexai/anthropic.rb +34 -0
- data/lib/ruby_llm/providers/vertexai/capabilities.rb +19 -0
- data/lib/ruby_llm/providers/vertexai/chat_completions/batches.rb +54 -0
- data/lib/ruby_llm/providers/vertexai/chat_completions.rb +15 -0
- data/lib/ruby_llm/providers/vertexai/embed_content.rb +42 -0
- data/lib/ruby_llm/providers/vertexai/embeddings.rb +22 -7
- data/lib/ruby_llm/providers/vertexai/gemini/batches.rb +42 -0
- data/lib/ruby_llm/providers/vertexai/gemini.rb +69 -0
- data/lib/ruby_llm/providers/vertexai/live_transcription.rb +24 -0
- data/lib/ruby_llm/providers/vertexai/mistral.rb +28 -0
- data/lib/ruby_llm/providers/vertexai/models.rb +145 -43
- data/lib/ruby_llm/providers/vertexai/transcription.rb +49 -4
- data/lib/ruby_llm/providers/vertexai/videos.rb +61 -0
- data/lib/ruby_llm/providers/vertexai.rb +160 -17
- data/lib/ruby_llm/providers/xai/capabilities.rb +18 -0
- data/lib/ruby_llm/providers/xai/chat.rb +3 -2
- data/lib/ruby_llm/providers/xai/chat_completions/batches.rb +108 -0
- data/lib/ruby_llm/providers/xai/chat_completions.rb +19 -0
- data/lib/ruby_llm/providers/xai/images.rb +91 -0
- data/lib/ruby_llm/providers/xai/models.rb +32 -36
- data/lib/ruby_llm/providers/xai/reported_cost.rb +18 -0
- data/lib/ruby_llm/providers/xai/responses.rb +52 -0
- data/lib/ruby_llm/providers/xai/speech.rb +45 -0
- data/lib/ruby_llm/providers/xai/transcription.rb +48 -0
- data/lib/ruby_llm/providers/xai/videos.rb +87 -0
- data/lib/ruby_llm/providers/xai.rb +15 -5
- data/lib/ruby_llm/railtie.rb +7 -16
- data/lib/ruby_llm/rerank.rb +105 -0
- data/lib/ruby_llm/research_job.rb +241 -0
- data/lib/ruby_llm/search_results.rb +68 -0
- data/lib/ruby_llm/server_tool_call.rb +73 -0
- data/lib/ruby_llm/speech.rb +159 -0
- data/lib/ruby_llm/speech_chunk.rb +33 -0
- data/lib/ruby_llm/support/deprecator.rb +22 -0
- data/lib/ruby_llm/support/inspectable.rb +49 -0
- data/lib/ruby_llm/support/instrumentation.rb +41 -0
- data/lib/ruby_llm/support/utils.rb +147 -0
- data/lib/ruby_llm/thinking.rb +127 -20
- data/lib/ruby_llm/tokenization.rb +59 -0
- data/lib/ruby_llm/tokens.rb +103 -33
- data/lib/ruby_llm/tool.rb +266 -91
- data/lib/ruby_llm/tool_call.rb +36 -3
- data/lib/ruby_llm/tools/server_tools.rb +109 -0
- data/lib/ruby_llm/transcription/wav_audio.rb +62 -0
- data/lib/ruby_llm/transcription.rb +138 -13
- data/lib/ruby_llm/transcription_chunk.rb +68 -0
- data/lib/ruby_llm/transport/connection.rb +193 -0
- data/lib/ruby_llm/transport/error_middleware.rb +131 -0
- data/lib/ruby_llm/transport/usage_middleware.rb +28 -0
- data/lib/ruby_llm/transport/websocket_connection.rb +220 -0
- data/lib/ruby_llm/uploaded_file.rb +144 -0
- data/lib/ruby_llm/version.rb +2 -1
- data/lib/ruby_llm/video.rb +136 -0
- data/lib/ruby_llm/video_job.rb +150 -0
- data/lib/ruby_llm/workflow.rb +91 -0
- data/lib/ruby_llm.rb +380 -6
- data/lib/tasks/ruby_llm.rake +21 -16
- data/skills/rubyllm/SKILL.md +81 -0
- data/skills/rubyllm/agents/openai.yaml +4 -0
- metadata +339 -97
- data/lib/generators/ruby_llm/install/templates/add_references_to_chats_tool_calls_and_messages_migration.rb.tt +0 -9
- data/lib/generators/ruby_llm/install/templates/create_models_migration.rb.tt +0 -39
- data/lib/generators/ruby_llm/install/templates/create_tool_calls_migration.rb.tt +0 -21
- data/lib/generators/ruby_llm/install/templates/model_model.rb.tt +0 -3
- data/lib/generators/ruby_llm/install/templates/tool_call_model.rb.tt +0 -3
- data/lib/generators/ruby_llm/upgrade_to_v1_10/templates/add_v1_10_message_columns.rb.tt +0 -19
- data/lib/generators/ruby_llm/upgrade_to_v1_10/upgrade_to_v1_10_generator.rb +0 -50
- data/lib/generators/ruby_llm/upgrade_to_v1_14/templates/add_v1_14_tool_call_columns.rb.tt +0 -7
- data/lib/generators/ruby_llm/upgrade_to_v1_14/upgrade_to_v1_14_generator.rb +0 -49
- data/lib/generators/ruby_llm/upgrade_to_v1_7/templates/migration.rb.tt +0 -145
- data/lib/generators/ruby_llm/upgrade_to_v1_7/upgrade_to_v1_7_generator.rb +0 -122
- data/lib/generators/ruby_llm/upgrade_to_v1_9/templates/add_v1_9_message_columns.rb.tt +0 -15
- data/lib/generators/ruby_llm/upgrade_to_v1_9/upgrade_to_v1_9_generator.rb +0 -49
- data/lib/ruby_llm/active_record/acts_as_legacy.rb +0 -597
- data/lib/ruby_llm/active_record/model_methods.rb +0 -82
- data/lib/ruby_llm/active_record/tool_call_methods.rb +0 -18
- data/lib/ruby_llm/aliases.rb +0 -41
- data/lib/ruby_llm/connection.rb +0 -159
- data/lib/ruby_llm/content.rb +0 -91
- data/lib/ruby_llm/deprecator.rb +0 -24
- data/lib/ruby_llm/error_middleware.rb +0 -81
- data/lib/ruby_llm/instrumentation.rb +0 -36
- data/lib/ruby_llm/mime_type.rb +0 -96
- data/lib/ruby_llm/model/info.rb +0 -164
- data/lib/ruby_llm/model_registry.rb +0 -39
- data/lib/ruby_llm/models_schema.json +0 -171
- data/lib/ruby_llm/providers/anthropic/chat.rb +0 -291
- data/lib/ruby_llm/providers/anthropic/content.rb +0 -44
- data/lib/ruby_llm/providers/anthropic/embeddings.rb +0 -20
- data/lib/ruby_llm/providers/anthropic/media.rb +0 -92
- data/lib/ruby_llm/providers/anthropic/models.rb +0 -59
- data/lib/ruby_llm/providers/anthropic/streaming.rb +0 -71
- data/lib/ruby_llm/providers/bedrock/chat.rb +0 -405
- data/lib/ruby_llm/providers/bedrock/media.rb +0 -108
- data/lib/ruby_llm/providers/bedrock/streaming.rb +0 -328
- data/lib/ruby_llm/providers/gemini/chat.rb +0 -542
- data/lib/ruby_llm/providers/gemini/embeddings.rb +0 -37
- data/lib/ruby_llm/providers/gemini/images.rb +0 -47
- data/lib/ruby_llm/providers/gemini/models.rb +0 -38
- data/lib/ruby_llm/providers/gemini/streaming.rb +0 -98
- data/lib/ruby_llm/providers/gemini/tools.rb +0 -234
- data/lib/ruby_llm/providers/gpustack/capabilities.rb +0 -20
- data/lib/ruby_llm/providers/ollama/capabilities.rb +0 -20
- data/lib/ruby_llm/providers/openai/chat.rb +0 -236
- data/lib/ruby_llm/providers/openai/embeddings.rb +0 -33
- data/lib/ruby_llm/providers/openai/images.rb +0 -90
- data/lib/ruby_llm/providers/openai/moderation.rb +0 -34
- data/lib/ruby_llm/providers/openai/streaming.rb +0 -55
- data/lib/ruby_llm/providers/openai/temperature.rb +0 -28
- data/lib/ruby_llm/providers/openai/transcription.rb +0 -71
- data/lib/ruby_llm/providers/perplexity/capabilities.rb +0 -72
- data/lib/ruby_llm/providers/vertexai/chat.rb +0 -14
- data/lib/ruby_llm/providers/vertexai/streaming.rb +0 -14
- data/lib/ruby_llm/stream_accumulator.rb +0 -218
- data/lib/ruby_llm/streaming.rb +0 -179
- data/lib/ruby_llm/tool_concurrency.rb +0 -105
- data/lib/ruby_llm/utils.rb +0 -130
- data/lib/tasks/models.rake +0 -593
- data/lib/tasks/release.rake +0 -94
- data/lib/tasks/vcr.rake +0 -124
data/lib/ruby_llm/cost.rb
CHANGED
|
@@ -1,76 +1,166 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
module RubyLLM
|
|
4
|
-
#
|
|
4
|
+
# A Cost prices token usage in US dollars using pricing from the model
|
|
5
|
+
# registry. Usage entries own accounting events; Message and Chat aggregate
|
|
6
|
+
# calls, and Model#cost_for prices any token usage against a specific model.
|
|
7
|
+
#
|
|
8
|
+
# response = chat.ask "Summarize Ruby's object model."
|
|
9
|
+
# response.cost.total
|
|
10
|
+
#
|
|
11
|
+
# cost = model.cost_for(response.tokens)
|
|
12
|
+
# cost.input
|
|
13
|
+
# cost.output
|
|
14
|
+
#
|
|
15
|
+
# The components are RubyLLM's normalized token buckets: #input, #output,
|
|
16
|
+
# #cache_read, #cache_write, and #thinking. When the registry lacks
|
|
17
|
+
# pricing for tokens that were used, the affected component and #total
|
|
18
|
+
# return +nil+ instead of a false zero.
|
|
19
|
+
#
|
|
20
|
+
# When the provider reports the exact cost of a call, #total returns the
|
|
21
|
+
# reported amount instead of a registry-price estimate, even when
|
|
22
|
+
# registry pricing is missing.
|
|
23
|
+
#
|
|
24
|
+
# Costs are computed when the object is built. Readers do not recalculate
|
|
25
|
+
# them when registry prices change.
|
|
26
|
+
# ::aggregate and ::from_h return the same class, so a single call, a
|
|
27
|
+
# whole chat, and a stored breakdown all read the same way.
|
|
5
28
|
class Cost
|
|
6
|
-
|
|
7
|
-
|
|
29
|
+
include Support::Inspectable
|
|
30
|
+
|
|
31
|
+
COMPONENTS = %i[input output cache_read cache_write thinking].freeze # :nodoc:
|
|
32
|
+
PER_MILLION = 1_000_000.0 # :nodoc:
|
|
33
|
+
|
|
34
|
+
class << self
|
|
35
|
+
# Combines several costs into one Cost that sums each component.
|
|
36
|
+
# Ignores +nil+ entries. A component returns +nil+ when pricing was
|
|
37
|
+
# missing for one of the calls, or when no call has a cost for that
|
|
38
|
+
# component. Pass <tt>complete: false</tt> when some requests are still
|
|
39
|
+
# running or their costs are unknown; #total then remains +nil+.
|
|
40
|
+
#
|
|
41
|
+
# cost = RubyLLM::Cost.aggregate(messages.map(&:cost))
|
|
42
|
+
# cost.total
|
|
43
|
+
#
|
|
44
|
+
def aggregate(costs, complete: true)
|
|
45
|
+
costs = costs.compact.select(&:tokens?)
|
|
46
|
+
|
|
47
|
+
missing = COMPONENTS.select do |component|
|
|
48
|
+
costs.any? { |cost| cost.missing?(component) }
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
amounts = COMPONENTS.to_h do |component|
|
|
52
|
+
values = costs.filter_map { |cost| cost.public_send(component) }
|
|
53
|
+
[component, missing.include?(component) || values.empty? ? nil : values.sum]
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
new(amounts:, missing:, reported: costs.any?, complete:, total: aggregate_total(costs))
|
|
57
|
+
end
|
|
8
58
|
|
|
9
|
-
|
|
59
|
+
# Rebuilds a cost from a stored breakdown Hash, as produced by #to_h
|
|
60
|
+
# and persisted alongside a usage entry. Keys may be Strings or
|
|
61
|
+
# Symbols. The component readers return the recorded amounts and
|
|
62
|
+
# #total equals the recorded +:total+.
|
|
63
|
+
#
|
|
64
|
+
# RubyLLM::Cost.from_h({ input: 0.0001, total: 0.0003 }).total
|
|
65
|
+
#
|
|
66
|
+
def from_h(hash, tokens: nil)
|
|
67
|
+
amounts = COMPONENTS.to_h { |component| [component, hash[component] || hash[component.to_s]] }
|
|
68
|
+
total_recorded = hash.key?(:total) || hash.key?('total')
|
|
69
|
+
total = hash[:total] || hash['total'] if total_recorded
|
|
70
|
+
missing = missing_recorded_components(amounts, tokens, total_recorded)
|
|
71
|
+
|
|
72
|
+
new(amounts:, missing:, reported: recorded_tokens?(amounts, tokens, total_recorded), total:)
|
|
73
|
+
end
|
|
10
74
|
|
|
11
|
-
|
|
12
|
-
costs = costs.compact.select(&:tokens?)
|
|
13
|
-
return new(amounts: {}, has_tokens: false) if costs.empty?
|
|
75
|
+
private
|
|
14
76
|
|
|
15
|
-
|
|
16
|
-
|
|
77
|
+
def aggregate_total(costs)
|
|
78
|
+
totals = costs.map(&:total)
|
|
79
|
+
totals.sum if totals.any? && totals.none?(&:nil?)
|
|
17
80
|
end
|
|
18
81
|
|
|
19
|
-
amounts
|
|
20
|
-
[
|
|
82
|
+
def missing_recorded_components(amounts, tokens, total_recorded)
|
|
83
|
+
return [] if total_recorded
|
|
84
|
+
return COMPONENTS.reject { |component| amounts[component] } unless tokens
|
|
85
|
+
|
|
86
|
+
COMPONENTS.select do |component|
|
|
87
|
+
tokens.public_send(component).to_i.positive? && amounts[component].nil?
|
|
88
|
+
end
|
|
21
89
|
end
|
|
22
90
|
|
|
23
|
-
|
|
91
|
+
def recorded_tokens?(amounts, tokens, total_recorded)
|
|
92
|
+
return true if total_recorded
|
|
93
|
+
return COMPONENTS.any? { |component| !tokens.public_send(component).nil? } if tokens
|
|
94
|
+
|
|
95
|
+
amounts.values.any? { |amount| !amount.nil? }
|
|
96
|
+
end
|
|
24
97
|
end
|
|
25
98
|
|
|
26
|
-
#
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
99
|
+
def initialize(tokens: nil, model: nil, category: :text_tokens, input_details: nil, tier: :standard, # :nodoc:
|
|
100
|
+
amounts: nil, missing: nil, reported: nil, complete: true, total: nil)
|
|
101
|
+
if amounts
|
|
102
|
+
@amounts = amounts
|
|
103
|
+
@missing = missing || []
|
|
104
|
+
@reported = reported
|
|
105
|
+
else
|
|
106
|
+
price_tokens(tokens, model, category, input_details, tier)
|
|
107
|
+
total = reported_total if total.nil?
|
|
108
|
+
@reported ||= !total.nil?
|
|
109
|
+
end
|
|
110
|
+
@complete = complete
|
|
111
|
+
@total = total
|
|
36
112
|
end
|
|
37
|
-
# rubocop:enable Metrics/ParameterLists
|
|
38
113
|
|
|
114
|
+
# Returns the cost of input tokens in US dollars, or +nil+ when the
|
|
115
|
+
# token count or its pricing is unavailable.
|
|
39
116
|
def input
|
|
40
|
-
|
|
117
|
+
@amounts[:input]
|
|
41
118
|
end
|
|
42
119
|
|
|
120
|
+
# Returns the cost of billable output tokens in US dollars, or +nil+
|
|
121
|
+
# when the token count or its pricing is unavailable.
|
|
43
122
|
def output
|
|
44
|
-
|
|
123
|
+
@amounts[:output]
|
|
45
124
|
end
|
|
46
125
|
|
|
126
|
+
# Returns the cost of cache-read input tokens in US dollars, or +nil+
|
|
127
|
+
# when the token count or its pricing is unavailable.
|
|
47
128
|
def cache_read
|
|
48
|
-
|
|
129
|
+
@amounts[:cache_read]
|
|
49
130
|
end
|
|
50
131
|
|
|
132
|
+
# Returns the cost of cache-write input tokens in US dollars, or +nil+
|
|
133
|
+
# when the token count or its pricing is unavailable.
|
|
51
134
|
def cache_write
|
|
52
|
-
|
|
135
|
+
@amounts[:cache_write]
|
|
53
136
|
end
|
|
54
137
|
|
|
138
|
+
# Returns the cost of thinking tokens in US dollars, or +nil+ when
|
|
139
|
+
# the model does not price reasoning output separately from regular
|
|
140
|
+
# output or the token count is unavailable. When not priced
|
|
141
|
+
# separately, thinking tokens are part of #output.
|
|
55
142
|
def thinking
|
|
56
|
-
|
|
143
|
+
@amounts[:thinking]
|
|
57
144
|
end
|
|
58
145
|
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
146
|
+
# Returns the sum of all components in US dollars, or the exact amount
|
|
147
|
+
# the provider reported when it reported one. Returns +nil+ when there
|
|
148
|
+
# is no token usage, or when pricing is missing for tokens that were
|
|
149
|
+
# used and the provider reported no cost.
|
|
64
150
|
def total
|
|
151
|
+
return nil unless @complete
|
|
65
152
|
return nil unless tokens?
|
|
66
|
-
return
|
|
153
|
+
return @total unless @total.nil?
|
|
154
|
+
return nil if @missing.any?
|
|
67
155
|
|
|
68
|
-
|
|
69
|
-
return nil if
|
|
156
|
+
amounts = @amounts.values.compact
|
|
157
|
+
return nil if amounts.empty?
|
|
70
158
|
|
|
71
|
-
|
|
159
|
+
amounts.sum
|
|
72
160
|
end
|
|
73
161
|
|
|
162
|
+
# Returns a hash of component costs in US dollars, plus +:total+,
|
|
163
|
+
# omitting +nil+ values.
|
|
74
164
|
def to_h
|
|
75
165
|
{
|
|
76
166
|
input: input,
|
|
@@ -82,30 +172,39 @@ module RubyLLM
|
|
|
82
172
|
}.compact
|
|
83
173
|
end
|
|
84
174
|
|
|
85
|
-
def tokens?
|
|
86
|
-
|
|
175
|
+
def tokens? # :nodoc:
|
|
176
|
+
@reported
|
|
177
|
+
end
|
|
87
178
|
|
|
88
|
-
|
|
179
|
+
def missing?(component) # :nodoc:
|
|
180
|
+
@missing.include?(component)
|
|
89
181
|
end
|
|
90
182
|
|
|
91
|
-
|
|
92
|
-
return @missing.include?(component) if aggregate?
|
|
93
|
-
return image_input_missing? if component == :input && detailed_image_input?
|
|
94
|
-
return false if component == :thinking && !thinking_priced_separately?
|
|
183
|
+
private
|
|
95
184
|
|
|
96
|
-
|
|
97
|
-
tokens
|
|
185
|
+
def price_tokens(tokens, model, category, input_details, tier)
|
|
186
|
+
@tokens = tokens
|
|
187
|
+
@model = normalize_model(model)
|
|
188
|
+
@category = category.to_sym
|
|
189
|
+
@input_details = input_details
|
|
190
|
+
@tier = tier.to_sym
|
|
191
|
+
@amounts = COMPONENTS.to_h { |component| [component, amount_for(component)] }
|
|
192
|
+
@missing = COMPONENTS.select { |component| missing_component?(component) }
|
|
193
|
+
@reported = COMPONENTS.any? { |component| !tokens_for(component).nil? }
|
|
98
194
|
end
|
|
99
195
|
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
values.empty? ? nil : values.sum
|
|
196
|
+
def reported_total
|
|
197
|
+
@tokens.reported_cost if @tokens.respond_to?(:reported_cost)
|
|
103
198
|
end
|
|
104
199
|
|
|
105
|
-
|
|
200
|
+
def missing_component?(component)
|
|
201
|
+
return image_input_missing? if component == :input && detailed_image_input?
|
|
202
|
+
return false if component == :thinking && !thinking_priced_separately?
|
|
203
|
+
|
|
204
|
+
tokens_for(component).to_i.positive? && price_for(component).nil?
|
|
205
|
+
end
|
|
106
206
|
|
|
107
207
|
def amount_for(component)
|
|
108
|
-
return @amounts[component] if aggregate?
|
|
109
208
|
return image_input_amount if component == :input && detailed_image_input?
|
|
110
209
|
|
|
111
210
|
token_count = tokens_for(component)
|
|
@@ -120,56 +219,88 @@ module RubyLLM
|
|
|
120
219
|
token_count * price / PER_MILLION
|
|
121
220
|
end
|
|
122
221
|
|
|
123
|
-
def aggregate?
|
|
124
|
-
!@amounts.nil?
|
|
125
|
-
end
|
|
126
|
-
|
|
127
222
|
def tokens_for(component)
|
|
128
|
-
return unless tokens
|
|
223
|
+
return unless @tokens
|
|
129
224
|
|
|
130
225
|
case component
|
|
131
226
|
when :input
|
|
132
|
-
tokens.input
|
|
227
|
+
@tokens.input
|
|
133
228
|
when :output
|
|
134
|
-
tokens.output
|
|
229
|
+
@tokens.output
|
|
135
230
|
when :cache_read
|
|
136
|
-
tokens.cache_read
|
|
231
|
+
@tokens.cache_read
|
|
137
232
|
when :cache_write
|
|
138
|
-
tokens.cache_write
|
|
233
|
+
@tokens.cache_write
|
|
139
234
|
when :thinking
|
|
140
|
-
tokens.thinking if thinking_priced_separately?
|
|
235
|
+
@tokens.thinking if thinking_priced_separately?
|
|
141
236
|
end
|
|
142
237
|
end
|
|
143
238
|
|
|
144
239
|
def price_for(component)
|
|
145
240
|
case component
|
|
146
|
-
when :input
|
|
147
|
-
|
|
148
|
-
when :
|
|
149
|
-
|
|
150
|
-
when :
|
|
151
|
-
text_pricing.cache_read_input
|
|
152
|
-
when :cache_write
|
|
153
|
-
text_pricing.cache_write_input
|
|
154
|
-
when :thinking
|
|
155
|
-
text_pricing.reasoning_output
|
|
241
|
+
when :input then input_price
|
|
242
|
+
when :output then output_price
|
|
243
|
+
when :cache_read then text_price(:cache_read_input_per_million, text_pricing.cache_read_input)
|
|
244
|
+
when :cache_write then text_price(:cache_write_input_per_million, text_pricing.cache_write_input)
|
|
245
|
+
when :thinking then text_price(:reasoning_output_per_million, text_pricing.reasoning_output)
|
|
156
246
|
end
|
|
157
247
|
end
|
|
158
248
|
|
|
249
|
+
def input_price
|
|
250
|
+
return text_input_price if @category == :text_tokens || image_cost?
|
|
251
|
+
|
|
252
|
+
category_pricing.input || text_input_price
|
|
253
|
+
end
|
|
254
|
+
|
|
255
|
+
def output_price
|
|
256
|
+
return image_pricing.output || text_output_price if image_cost?
|
|
257
|
+
return text_output_price if @category == :text_tokens
|
|
258
|
+
|
|
259
|
+
category_pricing.output || text_output_price
|
|
260
|
+
end
|
|
261
|
+
|
|
262
|
+
def text_input_price
|
|
263
|
+
text_price(:input_per_million, text_pricing.input)
|
|
264
|
+
end
|
|
265
|
+
|
|
266
|
+
def text_output_price
|
|
267
|
+
text_price(:output_per_million, text_pricing.output)
|
|
268
|
+
end
|
|
269
|
+
|
|
270
|
+
def text_price(attribute, standard_price)
|
|
271
|
+
applicable_text_tier&.public_send(attribute) || standard_price
|
|
272
|
+
end
|
|
273
|
+
|
|
274
|
+
def applicable_text_tier
|
|
275
|
+
return text_pricing.batch if @tier == :batch
|
|
276
|
+
|
|
277
|
+
text_pricing.tier_for(prompt_tokens)
|
|
278
|
+
end
|
|
279
|
+
|
|
280
|
+
def prompt_tokens
|
|
281
|
+
tokens_for(:input).to_i + tokens_for(:cache_read).to_i + tokens_for(:cache_write).to_i
|
|
282
|
+
end
|
|
283
|
+
|
|
159
284
|
def text_pricing
|
|
160
|
-
model&.pricing&.text_tokens || RubyLLM::Model::PricingCategory.new
|
|
285
|
+
@model&.pricing&.text_tokens || RubyLLM::Model::PricingCategory.new
|
|
161
286
|
end
|
|
162
287
|
|
|
163
288
|
def image_pricing
|
|
164
|
-
model&.pricing&.images || RubyLLM::Model::PricingCategory.new
|
|
289
|
+
@model&.pricing&.images || RubyLLM::Model::PricingCategory.new
|
|
165
290
|
end
|
|
166
291
|
|
|
167
|
-
def
|
|
168
|
-
|
|
292
|
+
def category_pricing
|
|
293
|
+
return text_pricing if @category == :text_tokens
|
|
294
|
+
return image_pricing if image_cost?
|
|
295
|
+
|
|
296
|
+
pricing = @model&.pricing
|
|
297
|
+
return RubyLLM::Model::PricingCategory.new unless pricing.respond_to?(@category)
|
|
298
|
+
|
|
299
|
+
pricing.public_send(@category)
|
|
169
300
|
end
|
|
170
301
|
|
|
171
302
|
def image_cost?
|
|
172
|
-
%i[image images].include?(category)
|
|
303
|
+
%i[image images].include?(@category)
|
|
173
304
|
end
|
|
174
305
|
|
|
175
306
|
def detailed_image_input?
|
|
@@ -193,9 +324,10 @@ module RubyLLM
|
|
|
193
324
|
end
|
|
194
325
|
|
|
195
326
|
def image_input_parts
|
|
327
|
+
text_input_price = applicable_text_tier&.input_per_million || text_pricing.input
|
|
196
328
|
[
|
|
197
|
-
[:text, input_detail('text_tokens'),
|
|
198
|
-
[:image, input_detail('image_tokens'), image_pricing.input ||
|
|
329
|
+
[:text, input_detail('text_tokens'), text_input_price],
|
|
330
|
+
[:image, input_detail('image_tokens'), image_pricing.input || text_input_price]
|
|
199
331
|
]
|
|
200
332
|
end
|
|
201
333
|
|
|
@@ -204,10 +336,11 @@ module RubyLLM
|
|
|
204
336
|
end
|
|
205
337
|
|
|
206
338
|
def thinking_priced_separately?
|
|
207
|
-
|
|
339
|
+
tier = applicable_text_tier
|
|
340
|
+
reasoning_price = tier&.reasoning_output_per_million || text_pricing.reasoning_output
|
|
208
341
|
return false unless reasoning_price
|
|
209
342
|
|
|
210
|
-
output_price = text_pricing.output
|
|
343
|
+
output_price = tier&.output_per_million || text_pricing.output
|
|
211
344
|
output_price.nil? || reasoning_price != output_price
|
|
212
345
|
end
|
|
213
346
|
|
|
@@ -220,5 +353,9 @@ module RubyLLM
|
|
|
220
353
|
rescue ModelNotFoundError
|
|
221
354
|
nil
|
|
222
355
|
end
|
|
356
|
+
|
|
357
|
+
def inspect_attributes # :nodoc:
|
|
358
|
+
to_h
|
|
359
|
+
end
|
|
223
360
|
end
|
|
224
361
|
end
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
# A DownloadedFile contains a provider file's bytes. Save it with #save
|
|
5
|
+
# or read the bytes with #to_blob. It is also a String, so parsers and
|
|
6
|
+
# existing string operations work directly on the result.
|
|
7
|
+
#
|
|
8
|
+
# RubyLLM.download(file.id, provider: :openai).save("report.pdf")
|
|
9
|
+
#
|
|
10
|
+
class DownloadedFile < String
|
|
11
|
+
include Support::Inspectable
|
|
12
|
+
|
|
13
|
+
# Returns the raw file bytes as a String.
|
|
14
|
+
def to_blob
|
|
15
|
+
to_s
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
# Writes the file bytes to +path+, expanding it first. Returns +path+.
|
|
19
|
+
#
|
|
20
|
+
# file.save("report.pdf")
|
|
21
|
+
#
|
|
22
|
+
def save(path)
|
|
23
|
+
File.binwrite(File.expand_path(path), to_blob)
|
|
24
|
+
path
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
private
|
|
28
|
+
|
|
29
|
+
def inspect_attributes # :nodoc:
|
|
30
|
+
{ byte_size: bytesize }
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
end
|
data/lib/ruby_llm/embedding.rb
CHANGED
|
@@ -1,59 +1,163 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
module RubyLLM
|
|
4
|
-
#
|
|
4
|
+
# An Embedding is the result of turning text into numerical vectors.
|
|
5
|
+
# RubyLLM.embed returns one:
|
|
6
|
+
#
|
|
7
|
+
# embedding = RubyLLM.embed("Ruby is a programmer's best friend")
|
|
8
|
+
# embedding.vectors # => [0.018, -0.027, ...]
|
|
9
|
+
#
|
|
10
|
+
# Pass an array to embed several texts in one call on supported models:
|
|
11
|
+
#
|
|
12
|
+
# RubyLLM.embed(["Ruby", "Rails"]).vectors # => [[...], [...]]
|
|
13
|
+
#
|
|
14
|
+
# Store the vectors for similarity search. #tokens and #cost report
|
|
15
|
+
# usage when the provider supplies it.
|
|
16
|
+
#
|
|
5
17
|
class Embedding
|
|
6
|
-
|
|
18
|
+
include Support::Inspectable
|
|
19
|
+
include Accounting::Usage::Result
|
|
7
20
|
|
|
8
|
-
|
|
21
|
+
# The embedding vectors. A flat array of floats when a single text
|
|
22
|
+
# was embedded, an array of such arrays when an array of texts was
|
|
23
|
+
# embedded.
|
|
24
|
+
attr_reader :vectors
|
|
25
|
+
|
|
26
|
+
# The sparse vectors, for models that return one beside the dense
|
|
27
|
+
# vector, as a Hash mapping token id to weight. Shaped like #vectors:
|
|
28
|
+
# one Hash for a single text, an array of them for an array of texts.
|
|
29
|
+
# +nil+ on models that return only dense vectors.
|
|
30
|
+
attr_reader :sparse_vectors
|
|
31
|
+
|
|
32
|
+
# The id of the model that produced the vectors, as a String.
|
|
33
|
+
attr_reader :model
|
|
34
|
+
|
|
35
|
+
def initialize(vectors:, model:, sparse_vectors: nil, input_tokens: nil, reported_cost: nil) # :nodoc:
|
|
9
36
|
@vectors = vectors
|
|
37
|
+
@sparse_vectors = sparse_vectors
|
|
10
38
|
@model = model
|
|
11
39
|
@input_tokens = input_tokens
|
|
40
|
+
@reported_cost = reported_cost
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# Returns usage aggregated across every provider attempt.
|
|
44
|
+
def tokens
|
|
45
|
+
return ruby_llm_usage_tokens unless ruby_llm_usage_entries.empty?
|
|
46
|
+
|
|
47
|
+
Tokens.new(input: @input_tokens, reported_cost: @reported_cost)
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
# Returns the embedding cost across every provider attempt.
|
|
51
|
+
def cost
|
|
52
|
+
return ruby_llm_usage_cost unless ruby_llm_usage_entries.empty?
|
|
53
|
+
|
|
54
|
+
Cost.new(tokens:, model: model_info, category: :embeddings)
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
def model_info # :nodoc:
|
|
58
|
+
@model_info ||= RubyLLM.models.find(model)
|
|
59
|
+
rescue ModelNotFoundError
|
|
60
|
+
nil
|
|
12
61
|
end
|
|
13
62
|
|
|
14
|
-
|
|
63
|
+
# Generates embeddings for +text+ and returns an Embedding. +text+
|
|
64
|
+
# may be a single string or, on supported models, an array of strings.
|
|
65
|
+
# An array produces one vector per string in a single API call.
|
|
66
|
+
#
|
|
67
|
+
# RubyLLM.embed "Ruby is a programmer's best friend"
|
|
68
|
+
# RubyLLM.embed ["Ruby", "Python", "JavaScript"]
|
|
69
|
+
# RubyLLM.embed "This is a test sentence",
|
|
70
|
+
# model: "text-embedding-3-large",
|
|
71
|
+
# dimensions: 512
|
|
72
|
+
# RubyLLM.embed "RubyLLM makes provider APIs feel native to Ruby.",
|
|
73
|
+
# model: "gemini-embedding-001",
|
|
74
|
+
# provider: :vertexai,
|
|
75
|
+
# task_type: "RETRIEVAL_DOCUMENT",
|
|
76
|
+
# title: "RubyLLM docs"
|
|
77
|
+
# RubyLLM.embed "The Ruby logo",
|
|
78
|
+
# model: "gemini-embedding-2",
|
|
79
|
+
# with: "logo.png"
|
|
80
|
+
#
|
|
81
|
+
# +model:+ selects the embedding model and defaults to the
|
|
82
|
+
# configured +default_embedding_model+. +provider:+ forces a specific
|
|
83
|
+
# provider, and <tt>assume_model_exists: true</tt> skips the model
|
|
84
|
+
# registry check. +context:+ supplies a Context whose configuration
|
|
85
|
+
# is used instead of the global one. +dimensions:+ requests a
|
|
86
|
+
# specific vector size on models that support it. +task_type:+ names
|
|
87
|
+
# the embedding task in the provider's own vocabulary: Vertex AI and
|
|
88
|
+
# Gemini take values such as <tt>"RETRIEVAL_QUERY"</tt> or
|
|
89
|
+
# <tt>"RETRIEVAL_DOCUMENT"</tt>, while Bedrock Cohere takes an input
|
|
90
|
+
# type such as <tt>"search_document"</tt>. +title:+ labels the
|
|
91
|
+
# document on Vertex AI and Gemini retrieval tasks. Providers that
|
|
92
|
+
# have no task concept ignore both. +with:+ passes one or more media
|
|
93
|
+
# attachments (images, audio, video, PDFs) to embed alongside the
|
|
94
|
+
# text on multimodal embedding models such as Gemini's
|
|
95
|
+
# gemini-embedding-2; providers without multimodal embeddings raise
|
|
96
|
+
# UnsupportedAttachmentError. +provider_options:+ takes options
|
|
97
|
+
# in the provider's request vocabulary and merges them into the
|
|
98
|
+
# request as-is. +metadata:+ is not sent to the provider; it is
|
|
99
|
+
# attached to the emitted +embedding.ruby_llm+ instrumentation event.
|
|
100
|
+
def self.embed(text,
|
|
15
101
|
model: nil,
|
|
16
102
|
provider: nil,
|
|
17
103
|
assume_model_exists: false,
|
|
18
104
|
context: nil,
|
|
19
|
-
dimensions: nil
|
|
105
|
+
dimensions: nil,
|
|
106
|
+
task_type: nil,
|
|
107
|
+
title: nil,
|
|
108
|
+
with: nil,
|
|
109
|
+
provider_options: {},
|
|
110
|
+
metadata: nil)
|
|
20
111
|
config = context&.config || RubyLLM.config
|
|
21
112
|
model ||= config.default_embedding_model
|
|
22
|
-
model, provider_instance = Models.resolve(model, provider: provider,
|
|
113
|
+
model, provider_instance = Models.resolve(model, provider: provider, assume_model_exists: assume_model_exists,
|
|
23
114
|
config: config)
|
|
24
115
|
model_id = model.id
|
|
116
|
+
empty_tokens = Tokens.new
|
|
25
117
|
|
|
26
118
|
payload = {
|
|
27
119
|
provider: provider_instance.slug,
|
|
28
|
-
provider_class: provider_instance.class.
|
|
120
|
+
provider_class: provider_instance.class.display_name,
|
|
29
121
|
model: model_id,
|
|
30
122
|
model_info: model,
|
|
31
123
|
input: text,
|
|
32
|
-
dimensions: dimensions
|
|
124
|
+
dimensions: dimensions,
|
|
125
|
+
task_type: task_type,
|
|
126
|
+
title: title,
|
|
127
|
+
attachment_count: Attachment.wrap(with).size,
|
|
128
|
+
provider_options: provider_options,
|
|
129
|
+
metadata: metadata,
|
|
130
|
+
tokens: empty_tokens,
|
|
131
|
+
cost: Cost.new(tokens: empty_tokens, model:, category: :embeddings)
|
|
33
132
|
}
|
|
34
133
|
|
|
35
134
|
RubyLLM.instrument('embedding.ruby_llm', payload, config: config) do |event|
|
|
36
|
-
result = provider_instance.embed(text, model
|
|
135
|
+
result = provider_instance.embed(text, model:, dimensions:, task_type:, title:, with:, provider_options:)
|
|
37
136
|
event[:result] = result
|
|
38
137
|
event[:response_model] = result.model
|
|
39
|
-
event[:
|
|
138
|
+
event[:tokens] = result.tokens
|
|
139
|
+
event[:cost] = result.cost
|
|
40
140
|
event[:embedding_dimensions] = vector_dimensions(result.vectors)
|
|
41
141
|
event[:embedding_count] = embedding_count(result.vectors)
|
|
42
142
|
result
|
|
43
143
|
end
|
|
44
144
|
end
|
|
45
145
|
|
|
46
|
-
def self.vector_dimensions(vectors)
|
|
47
|
-
return unless vectors.is_a?(Array)
|
|
48
|
-
|
|
146
|
+
private_class_method def self.vector_dimensions(vectors) # :nodoc:
|
|
49
147
|
vector = vectors.first.is_a?(Array) ? vectors.first : vectors
|
|
50
|
-
vector.length
|
|
148
|
+
vector.length
|
|
51
149
|
end
|
|
52
150
|
|
|
53
|
-
def self.embedding_count(vectors)
|
|
54
|
-
return unless vectors.is_a?(Array)
|
|
55
|
-
|
|
151
|
+
private_class_method def self.embedding_count(vectors) # :nodoc:
|
|
56
152
|
vectors.first.is_a?(Array) ? vectors.size : 1
|
|
57
153
|
end
|
|
154
|
+
|
|
155
|
+
def inspect_attributes # :nodoc:
|
|
156
|
+
if vectors&.first.is_a?(Array)
|
|
157
|
+
{ model: model, count: vectors.length, dimensions: vectors.first.length }
|
|
158
|
+
else
|
|
159
|
+
{ model: model, dimensions: vectors&.length }
|
|
160
|
+
end
|
|
161
|
+
end
|
|
58
162
|
end
|
|
59
163
|
end
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
# An EmbeddingRequest is an embedding awaiting a provider-side batch.
|
|
5
|
+
# RubyLLM.embed_later stages one; submit an array of them with
|
|
6
|
+
# RubyLLM.batch, and once the batch completes, collecting the results
|
|
7
|
+
# fills in #result:
|
|
8
|
+
#
|
|
9
|
+
# requests = texts.map { |text| RubyLLM.embed_later(text) }
|
|
10
|
+
# batch = RubyLLM.batch(requests)
|
|
11
|
+
# # later, once batch.refresh.complete?
|
|
12
|
+
# batch.results
|
|
13
|
+
# requests.first.result.vectors
|
|
14
|
+
#
|
|
15
|
+
class EmbeddingRequest
|
|
16
|
+
include Support::Inspectable
|
|
17
|
+
|
|
18
|
+
# The text staged for embedding.
|
|
19
|
+
attr_reader :text
|
|
20
|
+
|
|
21
|
+
# The Model of the embedding model the request targets.
|
|
22
|
+
attr_reader :model
|
|
23
|
+
|
|
24
|
+
# The requested vector size, or +nil+ for the model's default.
|
|
25
|
+
attr_reader :dimensions
|
|
26
|
+
|
|
27
|
+
# The Provider the request will be submitted through.
|
|
28
|
+
attr_reader :provider
|
|
29
|
+
|
|
30
|
+
# The Embedding, once the batch has completed and its results have
|
|
31
|
+
# been collected. +nil+ until then, and for requests that failed.
|
|
32
|
+
# Batch result collection writes it.
|
|
33
|
+
attr_accessor :result
|
|
34
|
+
|
|
35
|
+
def initialize(text, model: nil, provider: nil, dimensions: nil, context: nil) # :nodoc:
|
|
36
|
+
config = context&.config || RubyLLM.config
|
|
37
|
+
model ||= config.default_embedding_model
|
|
38
|
+
@model, @provider = Models.resolve(model, provider:, config:)
|
|
39
|
+
@text = text
|
|
40
|
+
@dimensions = dimensions
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# Returns the request payload this embedding request would send to the
|
|
44
|
+
# provider, in the provider's wire format.
|
|
45
|
+
def render
|
|
46
|
+
@provider.render_embedding(text, model: @model, dimensions: @dimensions)
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
def inspect_attributes # :nodoc:
|
|
50
|
+
{ text: text, model: model.id, dimensions: dimensions, result: result }
|
|
51
|
+
end
|
|
52
|
+
end
|
|
53
|
+
end
|